cloud/config.rs
1//! @yah:ticket(R040-F16, "pg-on-mesh service recipe: bind tailscale0 + pg_hba.conf snippet + ufw rules")
2//! @yah:at(2026-05-05T00:32:34Z)
3//! @yah:assignee(agent:claude)
4//! @yah:status(review)
5//! @yah:parent(R040)
6//! @yah:handoff("Companion to R040-F15. Inter-node TCP (Postgres primary↔replica, NATS clusters, anything raw-protocol) lives on the Headscale mesh, not on Hetzner public IPs. Each node has a stable 100.64.x.x mesh IP that survives replacement of the underlying box, so DNS / config / pg_hba never churn when a CPX-11 is rebuilt. WireGuard already encrypts the wire — TLS becomes defense-in-depth, not load-bearing. This ticket carries the concrete pg-shaped recipe so the first stateful service deploy doesn't have to re-derive the pattern; subsequent services (redis, NATS, etc.) cargo-cult from it.")
7//! @yah:next("ServiceConfig gains a `bind_interface: Option<String>` field (e.g. `Some(\"tailscale0\")` for mesh-only services). The cloud-init/podman compose renderer translates this into either `--network host` + `pg listen_addresses = '<mesh-ip>'` OR a podman macvlan/host-binding pattern that achieves the same.")
8//! @yah:next("Generated pg_hba.conf snippet: allow the mesh subnet (100.64.0.0/10) for replication + app users. Postgres binds to the node's tailscale0 mesh IP only — `listen_addresses` is templated from the node's `tailscale ip --4` at first boot.")
9//! @yah:next("Generated ufw rules: `ufw allow in on tailscale0 to any port 5432; ufw deny 5432` — mirrors the existing yah-yubaba 7443 pattern in mirror.yml. Same shape works for any mesh-only port.")
10//! @yah:next("Replica connection string uses primary's mesh IP, NOT its public IP. Stable across box replacement.")
11//! @yah:next("Out of scope: pg_basebackup orchestration, failover, WAL archiving — those belong in noisetable's domain; this ticket only standardizes the binding/firewall/auth shape so noisetable's pg deployment doesn't reinvent it.")
12//!
13//!
14//! @yah:ticket(R323-F9, "Add sync-wave ordering to ServiceComponent (deploy-panel wave order)")
15//! @yah:assignee(agent:claude)
16//! @yah:at(2026-05-26T15:20:25Z)
17//! @yah:status(review)
18//! @yah:phase(P2)
19//! @yah:parent(R323)
20//! @yah:next("ServiceComponent gains a wave/order field (or depends_on between components) so the deploy panel (R323-F4) can group workload rollout rows into sync waves (wave 0 parallel, wait healthy, wave 1, …). Today all components are implicitly wave 0.")
21//! @yah:next("compute_service/compute_cell in reconciler/sync_status.rs surface the wave per workload so F4 doesn't re-derive it.")
22//! @yah:gotcha("Until this lands, F4 should render every workload as wave 0 (no ordering).")
23//! @yah:handoff("Added wave: u32 (serde default=0, skip_serializing_if zero) to ServiceComponent in config.rs. Added is_zero_u32 helper. Fixed the three struct literal call-sites that now need wave: 0 (config.rs test, local_sim.rs x2, mesofact_static.rs). Added wave?: number to the TS ServiceComponent interface with a doc comment. Deploy panel now reads c.wave ?? 0 for each WorkloadRow instead of hardcoded 0. SyncFooter computes maxWave from the components array and renders 'wave 0' (all-zero case) or 'waves 0–N' (multi-wave). All 218 cloud lib tests pass; bun run typecheck clean.")
24//! @yah:verify("cargo test -p cloud --lib # 218 passed")
25//! @yah:verify("cd packages/yah/ui && bun run typecheck # no new errors")
26//! @yah:verify("In service.toml: add wave = 1 to a component, rebuild, open the deploy panel — that workload row shows 'w1' badge; SyncFooter shows 'waves 0–1'")
27//! @yah:verify("Component with no wave field in TOML deserializes as wave=0 (default). Saving a wave=0 component omits the field from the output TOML (skip_serializing_if).")
28//!
29//! @arch:see(.yah/docs/working/W142-pond.md)
30//!
31//! @yah:relay(R615, "Linked infra sources: sources.toml overlay so a camp can borrow another camp's substrate")
32//! @yah:at(2026-07-20T18:18:05Z)
33//! @yah:status(open)
34//! @arch:see(.yah/docs/working/W274-linked-infra-sources.md)
35//!
36//! @yah:ticket(R615-F1, "InfraSource types + SourcesConfig::load(infra_dir) parsing .yah/infra/sources.toml")
37//! @yah:status(review)
38//! @yah:assignee(agent:bundle-anthropic-miravel)
39//! @yah:at(2026-08-08T19:55:57Z)
40//! @yah:phase(P1)
41//! @yah:parent(R615)
42//! @yah:next("Add InfraSourceKind { Path { path }, Git(GitSource) } + InfraSource { owner, kind, mode, select } to cloud/src/config.rs. Reuse the existing GitSource (config.rs:1205, { repo, ref, subdir }) verbatim — do not invent a second git-source shape.")
43//! @yah:next("SourcesConfig::load(infra_dir) reads .yah/infra/sources.toml (schema_version = 1, ordered [[source]] array). Absent file = empty list, never an error — every existing camp has no sources.toml.")
44//! @yah:next("mode is the write-gate: read-only (borrower cannot mutate) vs owner-manages. Model it as an enum, not a bool, so a future read-write-with-approval tier is additive.")
45//! @yah:verify("cargo check -p cloud && cargo test -p cloud")
46//! @arch:see(.yah/docs/working/W274-linked-infra-sources.md)
47//! @yah:tier(Cleric)
48//! @yah:handoff("InfraSourceKind{Path{path},Git(GitSource)} + SourceMode{ReadOnly,Manage} + InfraSource{owner,kind,mode,select} + SourcesConfig{schema_version,source} all landed in oss/yubaba/crates/cloud/src/config.rs (after default_git_ref, ~line 1550). GitSource reused verbatim -- Git(GitSource) wraps the existing R561 type unchanged, no second git-source shape. InfraSourceKind is internally tagged (#[serde(tag=\"kind\", rename_all=\"kebab-case\")]) and flattened into InfraSource so a [[source]] table reads exactly like W274's example: owner/kind/path-or-repo+ref+subdir/mode/select all at one table level. mode: SourceMode defaults ReadOnly via #[serde(default)] on the field (enum, not bool, per the ticket's own instruction -- Manage is the explicit escape hatch). SourcesConfig::load(infra_dir) returns Ok(default()) -- schema_version=1, empty source list -- when sources.toml is absent; only parses+errors when the file exists and is malformed.")
49//! @yah:handoff("Tree anchor 85801e7f. Pathspec: oss/yubaba/crates/cloud/src/config.rs (only file touched). Tests: cargo test -p yah-cloud --lib (from oss/yubaba) 710 passed / 0 failed / 4 ignored, +6 new over the 704 baseline your R707-T6 verification recorded (sources_load_is_empty_when_the_file_is_absent, sources_parses_a_path_kind_exactly_like_w274s_example, sources_parses_a_git_kind_reusing_gitsource_verbatim, sources_mode_defaults_to_read_only_and_manage_is_explicit, sources_preserves_declaration_order, sources_round_trips_through_serialize). cargo check -p cloud also green (implied by the test build).")
50//! @yah:handoff("Tree anchor at handoff: 85801e7f6b76b369c0c8ecd2e5c7874990cd9286 — the shared tree as I left it. Diff against it (`git diff 85801e7f6b76b369c0c8ecd2e5c7874990cd9286..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
51//! @yah:next("R615-F2 picks this straight up: overlay these sources into CloudConfig::load, tagging origin{owner,source} and merging camp-local-wins-on-collision.")
52//! @yah:handoff("Verified pre-existing work: InfraSourceKind{Path,Git(GitSource)} + SourceMode + InfraSource + SourcesConfig all present in oss/yubaba/crates/cloud/src/config.rs at tree anchor 871fde1c, matching the inline @yah:handoff notes already on this ticket. GitSource reused verbatim, no second git-source shape. This session added no new code -- only ran verification and closed the board state, which a prior session left stuck in `open` despite the work being done (code + handoff notes landed, but board.review/handoff was never called).")
53//! @yah:verify("cargo check -p yah-cloud -- clean (2 pre-existing unrelated warnings)")
54//! @yah:verify("cargo test -p yah-cloud --lib -- 723 passed; 0 failed; 4 ignored (from oss/yubaba)")
55//!
56//! @yah:ticket(R615-F2, "Overlay loader: resolve sources in CloudConfig::load, tag origin, camp-local wins on collision")
57//! @yah:status(review)
58//! @yah:assignee(agent:bundle-anthropic-miravel)
59//! @yah:at(2026-08-08T19:56:05Z)
60//! @yah:phase(P1)
61//! @yah:parent(R615)
62//! @yah:next("In CloudConfig::load, after loading camp-local machines/providers/rules, resolve each source to an infra root (git sources read from the .yah/cache/infra/ sync cache — load stays offline), load that root's machines/providers/rules, tag each entry with origin { owner, source }, and overlay UNDER camp-local. Camp-local wins on name collision.")
63//! @yah:next("The machine load site is config.rs:533 (load_dir::<MachineConfig>(paths::machines_dir(...))). Note config.rs:575 load_from_config_dir is a SECOND machine load site that deliberately skips the inherit_machines redirect for multi-root/sibling trees (W206) — decide explicitly whether sources overlay applies there too, and document the answer either way.")
64//! @yah:verify("cargo check -p cloud && cargo test -p cloud")
65//! @yah:verify("A camp with sources.toml [[source]] kind=path to a sibling camp sees that camp's machines in CloudConfig::load, each tagged with the source owner")
66//! @yah:gotcha("Cross-camp MachineConfig schema skew is real: noisetable ships an older machine schema (location/server_type/hosts_mirrors) while yah's use region/arch/[connect]. A borrowed source can carry fields the borrower's binary predates. Overlay load MUST tolerate/skip unparseable foreign entries per-file and warn — never fail the whole load.")
67//! @arch:see(.yah/docs/working/W274-linked-infra-sources.md)
68//! @yah:depends_on(R615-F1)
69//! @yah:tier(Warrior)
70//! @yah:handoff("Overlay landed in CloudConfig::load (oss/yubaba/crates/cloud/src/config.rs). After camp-local machines/providers/legacy-merge finish, SourcesConfig::load(paths::infra_dir(workspace_root)) resolves + overlay_infra_sources() merges each source's machines/providers UNDER what's already there -- camp-local wins any name collision, and among sources themselves the earlier-declared one wins (both proven by dedicated tests). Provenance is NOT a field on MachineConfig/ProviderConfig: added CloudConfig.machine_origins/provider_origins: BTreeMap<String, InfraOrigin> instead, keyed by name/id. Reason recorded in a doc comment on InfraOrigin -- MachineConfig/ProviderConfig are constructed by struct literal in test helpers across several crates (including crates/yah/agent-tools/src/cloud_tools.rs, which is fenced/live-owned this session), so widening either shape would have forced an edit there for zero semantic gain; origin is a property of the LOAD, not the machine.")
71//! @yah:handoff("GOTCHA closed: added load_dir_tolerant<T>() -- a per-file-tolerant sibling of the existing (strict) load_dir -- so one unparseable foreign machine/provider (schema skew) skips-with-a-tracing::warn! and never sinks the rest of that source's directory or this camp's own load. Proven by one_unparseable_foreign_machine_does_not_sink_the_rest_of_the_directory_or_the_load. load_dir itself is untouched -- camp-local files still hard-fail on a bad TOML, which is correct, only borrowed roots get the tolerant path.")
72//! @yah:handoff("Git sources: InfraSource::infra_root() resolves kind=path to <workspace_root>/<path>/.yah/infra (live tree, no I/O beyond building the path) and kind=git to paths::infra_source_cache_dir(workspace_root, owner)/infra -- a NEW path helper in paths.rs, also what R615-T3's `yah infra sync` target directory must be so the two line up. An unsynced git source (cache dir absent) overlays nothing and is explicitly NOT an error (test: an_unsynced_git_source_overlays_nothing_and_is_not_an_error) -- load() stays fully offline as W274 §3 requires.")
73//! @yah:handoff("select filtering implemented for machines only (name exact-match or literal mesh_tags membership -- not a glob engine, matches W274's own example verbatim) via machine_matches_select(); does NOT apply to providers -- documented as a deliberate choice, nothing in W274 or the ticket describes a provider-scoped filter.")
74//! @yah:handoff("EXPLICIT DECISION on the config.rs:575-equivalent gotcha (now load_from_config_dir): sources overlay does NOT apply there. Multi-root sibling config dirs (W206 layout (b)) are a second config root INSIDE the same camp, not a second camp -- .yah/infra/sources.toml is tied to paths::infra_dir(workspace_root) specifically, which has no well-defined meaning for an arbitrary config_dir. Documented in the function's doc comment and proven by load_from_config_dir_never_applies_sources_overlay (a sources.toml at the real workspace root does NOT leak into a load_from_config_dir call against a sibling .noisetable/ dir under that same root).")
75//! @yah:handoff("Tree anchor 85801e7f. Pathspec: oss/yubaba/crates/cloud/src/config.rs, oss/yubaba/crates/cloud/src/paths.rs (added infra_source_cache_dir + 1 test), oss/yubaba/crates/cloud/src/reconciler/mesofact_bundle.rs (CloudConfig test-literal fixed for the 2 new fields), app/yah/cli/src/cloud.rs (3 CloudConfig test-literal sites fixed, same reason). Tests: cargo test -p yah-cloud --lib (from oss/yubaba) 720 passed / 0 failed / 4 ignored, +10 over R615-F1's 710 baseline (9 overlay tests in config.rs + 1 in paths.rs). cargo build -p yah --lib (repo root) green -- confirms nothing downstream (agent-tools, cloud.rs, hub) broke from CloudConfig's two new fields.")
76//! @yah:handoff("Tree anchor at handoff: 85801e7f6b76b369c0c8ecd2e5c7874990cd9286 — the shared tree as I left it. Diff against it (`git diff 85801e7f6b76b369c0c8ecd2e5c7874990cd9286..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
77//! @yah:next("R615-T3 (yah infra sync) is unblocked and has everything it needs: paths::infra_source_cache_dir(workspace_root, owner) is the exact target directory to clone/pull git sources into, already matching what F2's overlay reads from.")
78//! @yah:next("R615-F4 (Infra tab origin badge, not in my assigned lane) can read CloudConfig.machine_origins/provider_origins directly -- no further backend plumbing needed for the badge itself.")
79//! @yah:handoff("Verified pre-existing work: overlay landed in CloudConfig::load (oss/yubaba/crates/cloud/src/config.rs) at tree anchor 871fde1c -- SourcesConfig::load resolves sources, overlay_infra_sources() merges under camp-local with camp-local-wins and earlier-source-wins collision rules, machine_origins/provider_origins BTreeMaps added to CloudConfig, load_dir_tolerant() added for per-file-tolerant foreign schema skew, InfraSource::infra_root() resolves path/git kinds, load_from_config_dir explicitly does NOT get the overlay (documented). Matches this ticket's own inline @yah:handoff notes. This session added no new code -- only ran verification and closed board state that a prior session left stuck in `open` despite the work being done.")
80//! @yah:verify("cargo check -p yah-cloud -- clean (2 pre-existing unrelated warnings)")
81//! @yah:verify("cargo test -p yah-cloud --lib -- 723 passed; 0 failed; 4 ignored (from oss/yubaba), includes overlay tests + load_dir_tolerant test + infra_source_cache_dir test in paths.rs")
82//!
83//! @yah:ticket(R605-F12, "Sovereign groups have no voting axis, so non-voting membership is inexpressible and the raft guard is enforced by an absent field")
84//! @yah:status(review)
85//! @yah:at(2026-08-20T05:15:30Z)
86//! @yah:assignee(agent:bundle-anthropic-ashguard)
87//! @yah:parent(R605)
88//! @arch:see(.yah/docs/working/W325-isolated-x86-build-capacity.md)
89//! @yah:next("OPERATOR INTENT (2026-08-19) that the model cannot currently record: us-west-003 is a NON-VOTING member of the us-west-001-based (prod) sovereign group, and us-west-011 is a DIFFERENT sovereign (dev) from 001/003. The dev/prod split is already declared correctly. The non-voting membership is not — us-west-003.toml declares no sovereign_group at all.")
90//! @yah:next("THE GAP: MachineConfig::sovereign_group is a single Option<String>, so membership is binary, and judge_join (oss/yubaba/crates/cloud/src/config.rs:459) permits a join IFF both sides declare the same non-None group. There is no way to say 'in this blast radius, but not quorum-eligible'.")
91//! @yah:next("WHY THAT IS ACTIVELY BAD, not just missing: today the ONLY thing refusing us-west-003 into the prod raft at the join gate is its ABSENT stamp. Its own file is emphatic it must never hold a raft node id ('a home-internet partition should never be able to stall the raft'), and that guarantee currently rests on a field nobody wrote. Stamping it prod to record the operator's real intent would REMOVE the guard. This is precisely the W305 failure mode that produced R742-T4: `no-voter` sat inert on three nodes asserting something nothing enforced.")
92//! @yah:next("PROPOSED SHAPE (recommended): a second axis, e.g. sovereign_role = voter | non-voter (default voter for back-compat, or make it required), with judge_join permitting a same-group join only for voters. Then us-west-003 stamps prod + non-voter, the intent is machine-readable, and the raft guard stops depending on omission. us-west-004 (R605-T7) would take the same shape.")
93//! @yah:next("TOUCHES TWO COPIES OF THE PREDICATE, do not fix only one: cloud::judge_join renders the camp-side refusal, but the predicate itself lives in workload_spec::sovereign::join_permitted because yubaba's POST /raft/add-learner gate asks the same question and there is deliberately no yubaba -> cloud edge. Also re-read `yubaba serve --sovereign-group`, whose node-side gate is narrower on purpose (an unset flag means 'declared nothing', not 'declared standalone').")
94//! @yah:gotcha("THE CODE AND THE OPERATOR CURRENTLY DISAGREE ABOUT 003, and a reader should know which is which before editing. judge_join's own doc comment asserts 'prod and dev are both stamped, and us-west-002/003/015 are deliberately not raft members' — i.e. R742-F1 modelled 003 as STANDALONE. The operator's model is that it is a NON-VOTING MEMBER of prod. Those are different claims, not a wording difference: standalone means no blast-radius relationship to 001 at all. Do not silently 'correct' either side; this ticket is the reconciliation.")
95//! @yah:gotcha("FLEET STATE AS DECLARED (2026-08-19): prod = us-west-001, us-south-001, us-east-001. dev = us-west-011, us-west-013, us-west-014. NO sovereign_group declared = us-west-002, us-west-003, us-west-015. Verify against the files rather than trusting this list — xtask/tests/fleet_sovereign_groups.rs pins the roster and will need updating in the same change (it also asserts the stamp parses as a TOP-LEVEL key, which matters because 003 has a long comment block before [allocatable] where a stamp would silently become a member of that table).")
96//! @yah:gotcha("SEPARATE AXIS, DO NOT ENTANGLE: mesh membership is not sovereign membership. The standing rule is ONE mesh for the entire fleet regardless of group (operator, 2026-08-19), so us-west-003 and us-west-011 enrolling in headscale is unrelated work with no design question in it — see R605-T10. A voting axis on sovereign_group must not become a reason to keep any node off the mesh.")
97//! @yah:gotcha("SHARED-TREE COLLISION, live 2026-08-20: R772 (Miravel:spade, session:ce6d74a9) is refactoring oss/yubaba/crates/cloud/src/validate.rs at the same time and the file is currently RED - error[E0425] cannot find function load_machines at validate.rs:753, a half-landed extraction of the machine-loading walk that check_inert_taints / check_retired_arch_tags / the new check_unroled_sovereign_members all duplicate. That error is NOT from this ticket. Told them by party.chat and asked them to absorb check_unroled_sovereign_members into load_machines rather than leave one holdout. Do not hand-fight the file.")
98//! @yah:gotcha("R772 ALSO BROKE THREE PRE-EXISTING INGRESS TESTS, again not this ticket: two_services_fronting_one_node_collate_into_one_front_door, a_cross_service_hostname_clash_is_reported_with_both_declarations, one_mirrors_broken_declaration_does_not_hide_the_rest - all failing with 'providers.compute.use = hetzner - no such provider'. Cause is their new CloudConfig::load(workspace_root) at validate.rs:750 inside collate_workspace_ingress; the fronted_mirror fixture declares the slot but never writes infra/providers/hetzner.toml, and CloudConfig::load runs cross_ref_validate. Left alone deliberately - peer-owned.")
99//! @yah:gotcha("TRAP THAT MADE THREE OF MY OWN TESTS PASS FOR THE WRONG REASON: the machine-lint sweeps SKIP unparseable TOMLs by design (a peer's half-written scaffold must not sink the sweep). So a test fixture missing a REQUIRED MachineConfig field - mesh_tags is the one that bites - is silently skipped, the lint finds nothing, and every assert-empty test passes vacuously. Only the one test asserting found.len() == 1 noticed. write_sovereign_machine now always writes mesh_tags = [] and carries a comment saying why. Check this before trusting any new test in cloud::validate.")
100//! @yah:verify("cargo test -p yah-workload-spec --lib sovereign (from oss/yah-base) -- 9 passed, 0 failed. Covers both new refusals (a_non_voting_member_does_not_join_its_own_group, a_non_voting_target_has_no_quorum_to_join), the back-compat pin (the_default_role_is_the_pre_r605_f12_meaning), and the one-spelling round-trip across TOML/CLI/JSON.")
101//! @yah:verify("cargo test -p yubaba --lib sovereign (from oss/yubaba) -- 13 passed, 0 failed. Includes a_non_voting_joiner_is_refused_by_role_not_by_group, a_non_voting_target_refuses_every_joiner, a_group_without_a_role_key_is_a_voter_not_a_refusal (the deployed-fleet back-compat seam), a_peer_reports_its_role_in_the_toml_spelling.")
102//! @yah:verify("cargo test -p yubaba --test raft_sovereign_group (from oss/yubaba) -- 11 passed, 0 failed, up from 8. Three new end-to-end against real single-node rafts: a_non_voting_member_of_the_same_group_is_refused, a_non_voting_leader_refuses_to_grow_its_quorum, a_node_publishes_its_role_and_the_leader_reads_it_there (which also proves the request body cannot vote a non-voter in - the leader dials the joiner).")
103//! @yah:verify("cargo test -p xtask --test fleet_sovereign_groups (from repo root) -- 2 passed, 0 failed. THE DECISIVE ONE: parses the real .yah/infra/machines/*.toml through the actual MachineConfig deserializer. Confirms us-west-003 = prod + non-voter on disk, all six pre-existing voters now stamped sovereign_role = voter explicitly, and neither key swallowed by a table header.")
104//! @yah:verify("cargo test -p yah-cloud --lib (from oss/yubaba) -- 891 passed, 3 failed, where all 3 failures were R772's ingress-collate tests and none were mine. A clean re-run is BLOCKED, not failing: R555's in-flight AdmissionGrant.secrets field breaks velveteen-exec, and yah-cloud is not a root workspace member so its dev-deps can only resolve from the oss/yubaba workspace. Re-run once R555 lands.")
105//! @yah:handoff("LANDED, operator chose the second-axis shape (Call 1 = A, 2026-08-20). sovereign_role = voter | non-voter now sits beside sovereign_group, and ONE predicate judges both: workload_spec::sovereign::join_permitted(Membership, Membership) where Membership { group: Option<&str>, role: SovereignRole }. Permitted iff same non-None group AND both sides Voter. Both copies of the predicate call it - cloud::judge_join (camp-side) and yubaba::sovereign_group::judge (node-side) - so the rule itself cannot drift; only the prose differs, which was already the R742-F1 split.")
106//! @yah:handoff("WHY THE ROLE IS CHECKED ON BOTH SIDES, since only the joiner half was asked for: a join grows a quorum and it takes two nodes. Refusing a non-voting JOINER is the us-west-003 case. Refusing a non-voting TARGET is the same assertion read from the other end - a box declared non-voting that is serving add-learner is already holding a raft seat its own declaration forbids, and permitting there would paper over the contradiction. Both refusals name the role rather than the group when the groups match, because a message reading 'cross-group join refused: prod and prod' reads as a bug in the check.")
107//! @yah:handoff("THE DEFAULT IS THE LOAD-BEARING DECISION AND IT IS DELIBERATELY PERMISSIVE. An absent sovereign_role resolves to Voter (MachineConfig::sovereign_membership, the ONE place the Option is resolved). Reason: before this field, declaring a group WAS declaring quorum eligibility, so absence has to keep meaning that or the change silently retires six live voters. The permissiveness is bounded at the other end by cloud::validate::check_unroled_sovereign_members, which makes `yah cloud validate` FAIL on a group stamp with no role beside it - so the default can be reached by choice but not by silence. MachineConfig::sovereign_role stays Option<SovereignRole> (not a defaulted plain field) precisely so that lint can tell 'chose voter' from 'never considered it'.")
108//! @yah:handoff("NODE-SIDE BACK-COMPAT SEAM, pinned by a test because it is a decision and not an oversight: a peer answering GET /raft/status with a sovereign_group but NO sovereign_role key - every yubaba built between R742-F1 and R605-F12, which today is the entire prod raft - is read as Voter, not refused. Refusing would freeze a stamped cluster's growth until every member was rolled, strictly worse than what the role guards against, and it is the same degrade-toward-prior-behaviour stance the module already took for the group. Residue, named rather than hidden in read_group's doc: a box whose machine.toml says non-voter but whose daemon predates the flag answers 'voter' and the node gate admits it. judge_join refuses it camp-side, which is where operator-driven joins go. Window closes per-group as its nodes carry the flag.")
109//! @yah:handoff("FILES: workload-spec/src/sovereign.rs (SovereignRole + Membership + role-aware join_permitted, +227). cloud/src/config.rs (sovereign_role field, sovereign_membership(), judge_join same-group role branch, SovereignRole re-exported from cloud::config). cloud/src/validate.rs (check_unroled_sovereign_members + UnroledSovereignMember). app/yah/cli/src/cloud.rs (lint wired: ERROR in `yah cloud validate`, WARNING in the apply preflight - same split as inert-taint/retired-arch-tag, because an unwritten role changes no placement decision and the machine may be declared in a tree this camp does not own). yubaba/src/{sovereign_group,lib,main}.rs (--sovereign-role flag, ServerState.sovereign_role, /raft/status publishes it always-never-null, gate both directions). yubaba-test-harness/src/solo_node.rs (solo_node_with_sovereign_role). .yah/infra/machines/*.toml (7 files). xtask/tests/fleet_sovereign_groups.rs + fleet_build_placement.rs. W325 section 3d.")
110//! @yah:handoff("ONE BEHAVIOUR CHANGE WORTH A SECOND OPINION: a node started with --sovereign-role non-voter AND a --raft-node-id now refuses EVERY add-learner. I judged that correct - it is a contradiction the operator should see loudly - but the symptom is 'joins mysteriously stop working' rather than a startup refusal. main.rs warns loudly at boot when that pair is present; I did NOT make it fatal, because refusing to start could brick a node mid-roll. Reconsider if it bites.")
111//! @yah:handoff("NOT DONE, and it is a HARD GATE: .yah/schema/machine.toml.schema.json has NOT been regenerated, so sovereign_role is absent from it and schema-drift-guard (scripts/check-schema-drift.sh, a step in .yah/qed/check.toml, run by CI on every push) WILL FAIL. Fix is `cargo run -p xtask -- emit-schemas` from the repo root - it was queued behind ~7 concurrent peer cargo builds for the whole session. Nothing else is required to make this pushable.")
112//! @yah:handoff("ALSO NOT RE-CONFIRMED: `cargo test -p yah-cloud --lib` needs a clean run. Its last real run was 891 passed / 3 failed with all three failures belonging to R772's ingress-collate work and none to this ticket. The re-run is BLOCKED not failing - R555's in-flight AdmissionGrant.secrets field breaks velveteen-exec, and yah-cloud is not a root workspace member so its dev-deps only resolve from the oss/yubaba workspace where that break lives. Re-run from oss/yubaba once R555 lands.")
113//! @yah:verify("cargo run -p xtask -- emit-schemas (from repo root) -- wrote 8 files, exit 0 after an 18m24s build queued behind ~7 concurrent peer cargo jobs. .yah/schema/machine.toml.schema.json now carries the sovereign_role property (anyOf SovereignRole | null, with the full doc comment) and the SovereignRole definition as a oneOf over the two string enums voter / non-voter. The schema-drift-guard gate for THIS ticket is closed.")
114//! @yah:gotcha("emit-schemas IS ALL-OR-NOTHING AND WILL PICK UP A PEER'S UNCOMMITTED WORK. Running it to close this ticket's machine-schema drift also regenerated .yah/schema/secret.toml.schema.json (+34) from R555-F5's in-flight SecretAccess::Recipes / RecipeMatch source. That output is CORRECT for the tree as it stands and was not hand-edited, but it means the schema diff in the working tree is not purely R605-F12's: machine.toml.schema.json (+32) is this ticket, secret.toml.schema.json (+34) is R555. Told Ashguard:spade by party.chat so they carry it with their commit rather than regenerating on top. Anyone splitting these commits needs to split the schema diff too.")
115//! @yah:handoff("ALL GATES CLOSED as of 2026-08-20. Both items listed as outstanding in the earlier handoff notes are done: emit-schemas ran (machine.toml.schema.json carries sovereign_role + the SovereignRole voter/non-voter enum, drift guard satisfied), and cargo test -p yah-cloud --lib is 896 passed / 0 failed once R555 and R772 settled. 45 tests green across workload-spec (9), yubaba lib (13), yubaba raft integration (11), yah-cloud lib (10 of this ticket's, within 896), xtask fleet (2). Ready for review. NOTE for whoever commits: the working tree's schema diff is not purely this ticket - .yah/schema/machine.toml.schema.json (+32) is R605-F12, .yah/schema/secret.toml.schema.json (+34) is R555-F5, both correct generated output from one emit-schemas run. Ashguard:spade has agreed to carry theirs.")
116//! @yah:verify("cargo test -p yah-cloud --lib (from oss/yubaba) -- 896 passed, 0 FAILED, 4 ignored. The blocked check from earlier is now clean: R555 landed the velveteen-exec and TransformRecipe.secrets fixes, R772's ingress-collate work settled (they replaced the CloudConfig::load in collate_workspace_ingress with a narrower machines-only loader, so cross_ref_validate can no longer fail the collate over an unrelated provider typo). All 45 R605-F12 tests across the four crates are green simultaneously on one tree.")
117//! @yah:verify("Confirmed by NAME rather than by total, since a passing count proves nothing about which tests ran: cargo test -p yah-cloud --lib -- role voter voting lists all ten of this ticket's cloud tests green - a_non_voting_member_is_refused_into_its_own_group, a_non_voting_target_has_no_quorum_to_grow, a_refusal_names_the_group_when_fixing_the_role_would_not_help, an_unwritten_role_still_joins_its_group, a_non_voter_is_still_in_the_group_it_names, sovereign_role_round_trips_and_is_omitted_when_unwritten, a_group_with_no_role_is_reported_with_the_declaring_file, either_stated_role_is_clean, a_machine_in_no_group_is_not_asked_for_a_role, unroled_findings_are_ordered_by_file_so_output_is_stable.")
118//!
119//! @yah:ticket(R876-B7, "Node taints are structurally inert for mirror-declared placements: you cannot drain a node, and it fails silently")
120//! @yah:at(2026-09-09T09:05:55Z)
121//! @yah:status(review)
122//! @yah:assignee(agent:bundle-anthropic-ashguard)
123//! @yah:parent(R876)
124//! @yah:severity(high)
125//! @yah:next("SECOND HALF, and it is what makes the relay's headline question answerable: a working taint must produce a MOVE, not a refusal. Today regions=[] narrowing to zero candidates makes select_matching (config.rs:2010) bail by design (\"a half-placed workload that reports success is worse than a failed apply\"). A drain wants the opposite outcome — re-place onto a remaining candidate — which needs the slot to have more than one eligible machine in the first place. Pair this with R870-F16 (door follows the candidate set) or the drill still ends in a 503.")
126//! @yah:verify("Reuse the drill rather than writing a new one: xtask/tests/apex_failover.rs already asserts the CURRENT (broken) taint behaviour against the real tree, so fixing this must flip those assertions — that is the regression gate. Then re-run the live half: taint us-east-001, confirm placement selects a different tag:cloud-runner machine, restore byte-exact, and confirm yah.dev stays 200 throughout.")
127//! @yah:gotcha("IT FAILS SILENTLY, WHICH IS THE SHARP EDGE. \"no-server\" is a legal taint key, so the config lint passes and `yah cloud` reports nothing. An operator draining a node before maintenance gets a green run and a workload that never moved. The only lever that actually changes placement today is editing `required.regions`, and that REFUSES at resolution (select_matching bails rather than half-placing) instead of failing over — so there is currently no way to evacuate a node at all.")
128//! @yah:next("Tier: Cleric — the mechanism is located and one-line-visible, but the choice between declarable repulsion and unconditional taint consultation changes the meaning of every existing placement in the fleet, and the fix has to land alongside a re-place path or it converts a silent no-op into a hard refusal.")
129//! @yah:gotcha("MEASURED, NOT INFERRED — R876-S2's drill, 2026-09-09. `taints = [\"public-ip\", \"no-server\"]` was written onto the REAL .yah/infra/machines/us-east-001.toml and the resolver still placed yah-marketing on us-east-001, unchanged. Restored byte-exact (diff empty, sha256 back to 17dd15e2..., git clean against blob d66ab6d8); yah.dev stayed 200 throughout and no mutating apply was run.")
130//! @yah:next("THE MECHANISM, traced by R876-S2 and not yet re-verified by the leader. Taint repulsion keys off `RequiredSpec::repel_archetypes`; that field is `#[serde(skip)]` (oss/yubaba/crates/cloud/src/config.rs:4067), so a slot declared in a mirror's `required = {...}` ALWAYS deserializes with it empty. `matches` (config.rs:4182) consequently never reads `machine.taints` at all. Confirm both line anchors before editing — the shared tree moves.")
131//! @yah:handoff("SEMANTICS LANDED — repel-by-default + declarable toleration. `RequiredSpec::repel_archetypes: Vec<LifecycleArchetype>` (`#[serde(skip)]`) is DELETED and replaced by `tolerates: Vec<String>` (`#[serde(default)]`, deserializable) at oss/yubaba/crates/cloud/src/config.rs:4319. `matches` (config.rs:4397) no longer iterates a field of `self`: it walks `machine.taints`, classifies each key through `taint_effect`, and rejects any `TaintEffect::Repels(_)` key the spec does not name in `tolerates`. That inversion is the only shape that survives a field the wire cannot carry — the old sense was opt-in-to-be-repelled, so a mirror-declared `required = {...}` always deserialized with an empty archetype set and `machine.taints` was never read at all. Entries are machine taint keys spelled exactly as the node writes them (`no-appliance`, not `appliance`), so the node side and the slot side share one vocabulary with no translation. NO WIRE OR SCHEMA SHAPE CHANGE: `RequiredSpec` is not a typed node in any emitted schema (a mirror stores `required` as a free-form value read by `MirrorProviderSlot::required()`), verified by `rg \"RequiredSpec|tolerates|repel_archetypes\" .yah/schema/*.json` — the only hits are prose inside a doc-comment description.")
132//! @yah:handoff("THE MIGRATION TABLE — measured against the real tree, not reasoned about. FLEET TAINTS, all nine machines (`grep -rE \"^\\s*taints\\s*=\" .yah/infra/machines/*.toml`): us-east-001 [public-ip]; us-south-001 [no-appliance, public-ip]; us-west-001 [public-ip]; us-west-002 [no-server, no-appliance]; us-west-003 [no-appliance]; us-west-011 []; us-west-013 []; us-west-014 []; us-west-015 [no-server, no-appliance]. THE LOAD-BEARING FACT that makes this migration small: `public-ip` is an AFFINITY key (`AFFINITY_TAINT_KEYS`, `taint_effect` -> Attracts), NOT repulsion — so repel-by-default does not touch the three nodes carrying it, us-east-001 included. Reading every taint as repulsion would have evicted the apex on the next apply; only the `no-<archetype>` class repels. Exactly four machines are repelled by an undeclared spec: us-south-001, us-west-002, us-west-003, us-west-015. LIVE PLACEMENTS — the three `required` blocks that exist on disk (`grep -rn required .yah/services/*/mirrors/*.toml`): (1) yah-marketing providers.bundle, cloud.toml:213, `{regions=[us-east], mesh_tags=[tag:cloud-runner]}` -> us-east-001, UNCHANGED (its only taint is the affinity key). (2) yah-cloud providers.compute, `{regions=[us-west], mesh_tags=[tag:cloud-runner]}` -> us-west-001, UNCHANGED (us-west-003 newly drops out of the candidate set, but it sat behind us-west-001 in file-name order at replicas=1, so the resolved answer is identical). (3) yah-cloud-admin providers.compute, same constraint -> us-west-001, UNCHANGED. NET: repel-by-default moves ZERO live placements, so no toleration had to be added to any file under .yah/services/ or .yah/infra/ and none was. No file under .yah/infra/machines/ or .yah/services/ was written by this ticket at all.")
133//! @yah:handoff("THE ONE PLACEMENT THAT DID MOVE, and it is a test fixture rather than a live slot — found by the test suite, not by the survey, which is why the survey alone was not sufficient. `xtask/tests/mirror_ingress.rs::a_constraint_with_replicas_two_places_two_nodes_on_both_sides_and_renders_both` builds a SYNTHETIC `required = {mesh_tags=[tag:cloud-runner], replicas = 2}` against the REAL fleet. Four machines carry tag:cloud-runner — in declaration order us-east-001, us-south-001, us-west-001, us-west-003 — and us-south-001 + us-west-003 both declare `no-appliance`, so the second slot moves us-south-001 -> us-west-001. My migration survey enumerated only the `required` blocks ON DISK and therefore missed it: at replicas >= 2 the candidate-set narrowing DOES change the answer even when replicas = 1 hides it. Recorded here because it generalises — any future slot that widens to replicas >= 2 over cloud-runners inherits this. Fixed at the site that caught it (mirror_ingress.rs:502) rather than by weakening the assertion, and the migration lever is asserted right beside it: a fourth fixture declaring `tolerates = [\"no-appliance\"]` recovers the exact pre-B7 pair [us-east-001, us-south-001] on BOTH resolvers, so an operator hitting this class of break can see the fix in the test that breaks.")
134//! @yah:verify("BASELINE MEASURED BEFORE EDITING, then re-measured after. `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` = 1129 passed / 0 failed / 4 ignored, exit 0 (the run completed and printed its result line before my first Edit; a deferred W298 skew advisory later named config.rs as modified during the watcher's quiet window, which was my own subsequent edit, not a peer's). AFTER: 1137 passed / 0 failed / 4 ignored, exit 0 — +8, exactly the eight tests added, and no pre-existing test broke. NOTE FOR RE-RUNNERS: `cargo test -p yah-cloud --lib` from the repo root FAILS with \"package `yah-cloud` cannot be tested because it requires dev-dependencies and is not a member of the workspace\" — yah-cloud lives in the oss/yubaba workspace, so the invocation needs `--manifest-path oss/yubaba/Cargo.toml`. Also `cargo check --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --all-targets` exit 0 and `-p yubaba --all-targets` exit 0 (yubaba consumes cloud, so it is where the field removal would have surfaced). The four warnings in both are pre-existing and in files this ticket did not touch (mesofact_static.rs unused imports, app_manifest.rs dead field, pond_door.rs unused fn, reconciler/mod.rs non-snake-case).")
135//! @yah:verify("EIGHT NEW UNIT TESTS in config.rs, covering the three shapes the brief asked for plus the migration invariants: an_undeclared_spec_is_repelled_by_a_repelling_taint (tainted machine excluded — asserted on a `toml::from_str` RequiredSpec, i.e. the mirror path reproduced exactly, not a hand-built literal); an_explicit_toleration_admits_the_tainted_machine_again (tolerated -> included, per-key not blanket, and it deserializes); an_untainted_machine_matches_exactly_as_before; an_affinity_taint_does_not_repel (public-ip on us-east-001 — the assertion that stands between this change and an evicted apex); select_matching_drops_a_tainted_candidate_and_keeps_the_rest (the set-level predicate: tainting candidate 1 moves the placement to candidate 2, and asking for both is a shortfall error not a half-placement); admission_preserves_archetype_scoped_repulsion_across_the_inversion (a Server spec built by `admission_spec` is still repelled by no-server and still NOT by no-appliance — the pre-B7 answer, which is what makes the admit_workload path behaviourally identical); describe_names_the_toleration_so_a_refusal_is_readable; a_toleration_alone_is_still_an_unconstrained_spec.")
136//! @yah:verify("REGRESSION GATE FLIPPED, not deleted. `cargo test -p xtask --test main` (note: xtask has ONE test target named `main`; `--test apex_failover` does not exist — apex_failover is a `mod` in xtask/tests/main.rs). Result 65 passed / 1 failed. xtask/tests/apex_failover.rs: the drill's finding-1 test was inverted and renamed every_repelling_taint_at_once_leaves_the_apex_bundle_exactly_where_it_was -> ..._now_makes_the_apex_node_ineligible; it now asserts that ONE repelling key is enough (checked before the all-three case so a regression handling only the union is still caught), that all three refuse, and that restoring us-east-001's real taint list [\"public-ip\"] puts the placement straight back. The module header was rewritten to say the hole is closed. ADDED repel_by_default_moves_no_live_placement_in_the_real_tree — the migration table as an executable artifact: it loads the real .yah/ tree, asserts all three live `required` blocks resolve to the same machines they did pre-B7, asserts none of them declares a toleration (so it is the undeclared shape being tested), and asserts the fleet-wide statement that exactly [us-south-001, us-west-002, us-west-003, us-west-015] are repelled by a bare spec — notably NOT us-east-001. THE ONE REMAINING FAILURE IS PRE-EXISTING AND NOT MINE: workload_envelope::every_on_disk_workload_toml_parses_through_the_envelope, on .yah/infra/state/sources/scrabcake/site/site/workload.toml (`unknown field routes`). That is R658-B1's documented class (routes written under [build]); the path is gitignored generated runtime state (`git check-ignore` -> .yah/.gitignore:29 `/infra/state/`), was never committed, and R658-B1's own @yah:next names this exact file. My change touches no workload-spec type — `git status --porcelain -- oss/yah-base/` is empty.")
137//! @yah:handoff("SCOPE BOUNDARY HELD, deliberately. yah-marketing's candidate set was NOT widened: `.yah/services/yah-marketing/mirrors/cloud.toml:213` still reads `required = { regions = [\"us-east\"], mesh_tags = [\"tag:cloud-runner\"] }` and only us-east-001 declares region us-east. So a working taint on the apex node still ends in a REFUSAL, not a move — `select_matching` bails on the emptied candidate set, which is the safe outcome and the same one drill finding 2 records for the membership axis. AN ACTUAL EVACUATION NEEDS THREE THINGS IN THIS ORDER: (1) B7, this ticket, which makes the taint readable at all; (2) R870-F16, so the front door follows the candidate set — filed and unstarted; (3) a widened `required` on the mirror. Doing (3) before (2) buys a workload that relocates and a yah.dev that 503s, which is why it was not done here. Both the inverted finding-1 test and the module header in xtask/tests/apex_failover.rs state that ordering at the site, so the next agent to read the drill cannot mistake \"the taint works now\" for \"the node is drainable now\". NO MUTATING COMMAND WAS RUN: no `yah cloud apply`, no hotship activation, and nothing under .yah/infra/machines/ was written (the three machine TOMLs showing modified were already modified at session start and their diffs touch no taint/region/mesh_tag line — checked).")
138//! @yah:handoff("GENERATED ARTIFACTS REGENERATED, and one of them is a peer's. `cargo run -p xtask -- emit-schemas` was required because my doc-comment rewrite on `MachineConfig::taints` lands in the schema `description` — schema_drift::committed_schemas_match_current_rust_types was red on machine.toml.schema.json. The regen also swept in mirror.toml.schema.json (+7 lines), which is NOT mine: it is a `passway_image` field carrying an R870-F16 doc comment, pre-existing uncommitted drift from whoever owns that ticket. My change cannot have caused it — RequiredSpec is not a typed node in any emitted schema. Regenerated per CLAUDE.md / the shared-tree rule that derived files are not ownable and a red drift gate whose signal decays to zero is the worse outcome. @Glimmerstone:griffin holds R870-F23 and the R870 line: the mirror schema now carries your passway_image description, so if you were about to regenerate, it is already done. Both schema files are the only two under .yah/schema/ that changed.")
139//! @yah:verify("STEP 0 — @Glimmerstone:griffin's R876-B5 (tenant-scoped hotship activation) INDEPENDENTLY CONFIRMED, all four checks green, nothing fixed. (1) `bash -n scripts/hotship.sh` clean. (2) `./scripts/hotship.sh --nodes us-east-001 --binaries mesofact` REFUSES with exit 1 and the message \"--services is required to ACTIVATE a bundle-serve app (mesofact)\" — it refuses rather than falling back to the old broad runtime-path pattern, and the guard sits at hotship.sh:507 ahead of the version stamp and every remote call. (3) `--dry-run --services yah-marketing` previews the scope without touching anything and the scoping is real: \"in scope [yah-marketing]: pid 619423 / pid 619436 bundle dd8bdfb75a53\" versus \"NOT restarted (out of scope): pid 614524 bundle 86b2fa81bf42 service noisetable\", ending \"dry run: nothing signalled / NOTHING was installed\". (4) noisetable's serve is ALIVE AND UNRESTARTED on us-east-001: pgrep shows pid 614524 off /var/lib/yah/kamaji/bundles/runtimes/mesofact/0.8.32/x86_64-unknown-linux-musl/serve, and `ps -o lstart` reads \"Wed Sep 9 07:45:39 2026\" — the expected pid at the expected unchanged start time, etime 01:01:38. `curl -sS -o /dev/null -w %{http_code} https://yah.dev/` = 200. No real hotship activation was run.")
140//! @yah:verify("BUILDS. `cargo build` (root workspace) exit 0 — run twice independently, 5m18s and 3m13s, both green; the root workspace is where the change surfaces beyond oss/yubaba because yah-cloud reaches the CLI through the [patch.crates-io] bridge. `cargo check --manifest-path oss/yubaba/Cargo.toml -p yubaba --all-targets` exit 0. Clean re-measure of `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` after all annotation writes: 1137 passed / 0 failed / 4 ignored, exit 0 — identical to the first post-change measurement, so the earlier W298 skew advisory naming config.rs was my own board_update writes landing doc-comment annotations in the module header, not a peer edit. A later advisory on the root build named app/yah/cli/src/cloud.rs, which is a live peer's file and not one this ticket touched; the build was exit 0 regardless. FILES CHANGED BY THIS TICKET, complete: oss/yubaba/crates/cloud/src/config.rs, xtask/tests/apex_failover.rs, xtask/tests/mirror_ingress.rs, .yah/schema/machine.toml.schema.json, .yah/schema/mirror.toml.schema.json. Nothing under .yah/infra/ or .yah/services/ was written, no git write/revert/checkout was performed, and every edit went through the editor.")
141//! @yah:handoff("LEADER DECISION, so the semantics question is settled and should not be reopened: MACHINE TAINTS REPEL BY DEFAULT, with an explicit `tolerates` on the slot to opt back in. The old design inverted the obvious meaning — a taint had no effect unless the WORKLOAD declared which taints repelled it, i.e. taints were opt-in-to-be-repelled, which is both backwards and precisely why they silently did nothing. `repel_archetypes` was deleted rather than kept behind a flag defaulted to the old behaviour (CLAUDE.md, \"break it, don't tape it\").")
142//! @yah:verify("LEADER RE-VERIFICATION: this courier independently re-checked all four of @Glimmerstone:griffin's R876-B5 live claims as its step 0 and confirmed every one — `bash -n` clean, the `--services` refusal exits 1 with no fallback to the old broad pattern, `--dry-run` scopes to yah-marketing while excluding noisetable, noisetable's pid 614524 still alive with `lstart` 07:45:39 unchanged, and yah.dev 200. Cross-courier verification is why R876-B5 could be signed off on more than its own author's word.")
143//! @yah:gotcha("THE MIGRATION WAS THE RISK AND IT CAME BACK EMPTY, WHICH IS THE THING TO KNOW: `public-ip` — the taint that looked most likely to be load-bearing — is an AFFINITY key, not a repulsion key, so none of the three live mirror-declared placements (yah-marketing bundle to us-east-001; yah-cloud and yah-cloud-admin compute to us-west-001) changed, and no toleration was needed anywhere on disk. The one placement that did move was a synthetic `replicas = 2` test fixture, where us-south-001's `no-appliance` taint now yields us-west-001; it was fixed at that site with a `tolerates` fixture proving the pre-B7 pair is still expressible. Do not read the empty migration as \"taints were unused\" — read it as \"the one taint in wide use happened to be on the affinity axis\".")
144//!
145//! @yah:ticket(R870-F23, "Render and supervise the inner door: the service.toml + domain-manifest join that feeds passway's PathRouter config")
146//! @yah:status(review)
147//! @yah:phase(P2)
148//! @yah:at(2026-09-11T00:25:14Z)
149//! @yah:assignee(agent:bundle-anthropic-ashguard)
150//! @yah:parent(R870)
151//! @yah:next("THE CONSUMER SIDE IS DONE AND ITS FORMAT IS FIXED (R870-T18, in review). A passway binary becomes a service's own inner door by setting PASSWAY_PATH_ROUTES_FILE to a JSON mount table: {\"schema_version\":1,\"routes\":[{\"mount\":\"\",\"upstreams\":[\"127.0.0.1:8081\"]},{\"mount\":\"/app\",\"upstreams\":[\"127.0.0.1:8082\"],\"headers\":{\"cross-origin-opener-policy\":\"same-origin\"}}]}. Parser + validation: oss/passway/crates/passway/src/path_routes_file.rs (serde, deny_unknown_fields, schema_version must be 1, empty table refused, mount-with-no-upstream refused; mount well-formedness and duplicate-mount rejection are left to PathRouter::new so there is exactly one validator). Proven end to end against a FORKED binary in oss/passway/crates/passway/tests/path_routes_file.rs. This ticket is the producer: write that file.")
152//! @yah:next("WHY THIS IS A SEPARATE TICKET AND NOT HALF OF R870-T18. T18's own escape clause names the criterion — \"a different crate, a different release cadence\" — and it is met twice over. (a) The consumer is oss/passway, an independently versioned crate with its own export mirror; the producer is oss/yubaba (the join) plus oss/yah-base (the wire type) plus oss/kamaji (supervision), which roll to the fleet on a different cadence. (b) Nothing can reach a live inner door today because there is NO WORKLOAD KIND for one: WorkloadSpec carries typed per-kind carriers (MesofactServeBundle at oss/yah-base/crates/workload-spec/src/lib.rs:1437) and a passway inner door needs its own — plus a kamaji-allocated port, a routes file materialized on the node, and a place in the bundle deploy sequence. Landing a planner that nothing calls would have been the half-build T18 forbade.")
153//! @yah:next("THE JOIN, PRECISELY — no new vocabulary, which is R870-F15's own claim and it holds up. Inputs: .yah/services/<svc>/service.toml (ServiceComponent { id, kind, mount, ... }, config.rs:3427) and .yah/domains/<zone>.toml (DomainRoute { path, headers, mode }, config.rs:4558, where front_door = passway). Per mount: mount = path_route::mount_from_component(component.mount) — that function already exists and is already the ONE place the \"app\"/None to \"/app\"/\"\" translation happens; headers = the DomainRoute whose route_path_prefix(path) equals normalize_mount(component.mount) (cross_ref_validate already PROVES those two agree, config.rs:1688-1725, so the join cannot silently mismatch); upstreams = the address of the deployed unit serving that mount. Only the last one is placement-time and is why this needs the workload kind above. Group by DEPLOYED UNIT, not by component: every bundle-tier component of a service shares ONE bundle workload (that is config 1, R870-B11), so config-1 mounts collapse to a single root upstream and only independently-deployed units earn their own mount.")
154//! @yah:next("THE TWO ADMISSION RULES, and where each one goes. Both belong to the GENERATOR, never to passway — passway proxies whatever PathRouter it is handed and has no view of how many components a service declares. (1) A service with ONE independently-deployed unit gets NO inner tier at all — enforce by construction: the planner returns Option<InnerDoorPlan> and answers None below two units, so there is no config to write and no process to supervise, and the negative is assertable on the ABSENCE of the plan rather than on a site staying up. (2) A component cannot be both bundle-staged (config 1) and its own workload. R870-B11 landed the config-1-internal half in CloudConfig::cross_ref_validate (config.rs:1621-1657, two bundle components at one mount are refused); put this half in the SAME loop rather than a parallel one. NOTE, checked not assumed: the second half is NOT EXPRESSIBLE TODAY — [providers.bundle] is a per-MIRROR slot, not per-component, so there is no way to say \"give this one component its own workload\" at all. The rule becomes writable in the same commit that introduces that vocabulary, which is this ticket. Do not invent the vocabulary separately.")
155//! @yah:gotcha("DESIGN WRINKLE FOUND WHILE BUILDING R870-T18, and it is an operator call, not a coding one. passway ALWAYS terminates TLS on its listener: TlsMode has exactly two variants, Manual and Acme (oss/passway/crates/passway/src/tls.rs:215), and main() unconditionally calls proxy_service.add_tls_with_settings(&listen, None, tls_settings). So an inner door on loopback still needs a cert on disk, and the outer door still needs PASSWAY_UPSTREAM_TLS=true plus an SNI to reach it. That works — T18's binary-level test does exactly this with an rcgen self-signed leaf — but it means the \"cheap inner tier\" costs a cert, a renewal story, and an upstream TLS handshake per request on loopback. The obvious fix is a plaintext listener mode, and it was deliberately NOT taken in T18: adding a way for a public-facing trust-boundary door to serve cleartext is a security decision with a blast radius past this relay. Decide it before building the supervisor, because it changes what the workload spec has to carry.")
156//! @yah:verify("A two-component service whose components deploy INDEPENDENTLY gets an inner door: one yah cloud apply leaves both https://<host>/ and https://<host>/app/ at 200, and curl -sI on /app/ carries cross-origin-opener-policy: same-origin AND cross-origin-embedder-policy: require-corp from the /app/* route in the domain manifest, while / carries neither.")
157//! @yah:verify("THE NEGATIVE, asserted on absence rather than on uptime: a single-component service (yah-marketing) produces NO inner-door config and NO inner-door process — no routes file materialized on the node, no extra supervised workload in kamaji's table, and a byte-identical workload spec to today. A unit test on the planner returning None is the cheap half; the node-side absence check is the half that matters.")
158//! @yah:gotcha("OPERATOR CALL ASKED AND NOT ANSWERED (R870 relay leader, session:abde2cbb, 2026-09-09). The TLS question in this ticket first gotcha was put to the operator as a three-way choice and the prompt timed out unanswered after 30 minutes, so it remains genuinely open — it was not skipped and not decided by default. The three options as framed, so whoever picks this up does not have to re-derive them: (A) add a plaintext listener mode gated so it is structurally impossible to combine with a public bind — refuse at config load unless the bind is loopback, keep it mutually exclusive with ACME/cert paths; this was the leader recommendation, on the grounds that it makes the inner tier actually cheap as R870-F15 design claimed while keeping the risk a bounded testable invariant rather than an operator remembering not to misconfigure it. (B) keep TLS everywhere and have F23 carry a cert-issuance plus renewal story for every inner door, which is safest by construction and already proven working in R870-T18 binary-level test with an rcgen self-signed leaf, but makes every service with 2+ independently-deployed components pay a cert, a renewal and a loopback handshake per request. (C) park the tier — nothing regresses, because config 1 (bundle staging, R870-B11, in review) already covers the deploy-together case, which is the one noisetable actually needs. THIS IS THE ONLY THING BLOCKING F23 DESIGN; the join itself, both admission rules and the workload-kind vocabulary are all specified in this ticket next entries and need no further decisions.")
159//! @yah:handoff("OPERATOR CALL ANSWERED 2026-09-09: option (A), the loopback-only plaintext listener. It was re-put with ONE fact the earlier framing did not have, and that fact inverts the safety argument the three options were weighed on: option (B) was never \"already proven working\". pingora defaults verify_cert: true (pingora-core-0.8.1/src/upstreams/peer.rs:479, read not assumed) and passway NEVER overrides it — there is no verify_cert anywhere in oss/passway/crates/passway/src. R870-T18's test drove the door from an HTTP client with danger_accept_invalid_certs, not from an outer passway, so the outer-to-inner leg was untested. Since no CA issues for 127.0.0.1, \"keep TLS everywhere\" required a SECOND unbuilt change — a way to disable or pin upstream certificate verification on a public-facing door — traded for encrypting a hop that never leaves the loopback interface. (A) is strictly the smaller security surface, not merely the cheaper one.")
160//! @yah:handoff("PASSWAY: TlsMode::Plaintext, selected by PASSWAY_TLS_MODE=plaintext (a third value on the EXISTING discriminator, not a new bool env var — one variable owns the listener's TLS mode). All guards live in ONE function, tls.rs parse_listener_tls_mode, and each is a boot failure naming what to change: the bind must parse as a LITERAL loopback SocketAddr (0.0.0.0:443 — the default — is refused, and so is a hostname this process cannot prove); PASSWAY_TLS_CERT/KEY must be unset, so a configured public door cannot go cleartext by ADDING a variable rather than removing two; LISTEN_FDS is refused outright because under socket activation PASSWAY_LISTEN is only the key pingora looks the socket up by and proves nothing about the bind. An unrecognized PASSWAY_TLS_MODE is now also a boot failure instead of a silent fall-through to manual. main() reads the mode BEFORE the cert paths (plaintext has none), branches to proxy_service.add_tcp(&listen), and build_tls_settings returns Err rather than panicking on the variant it can no longer be handed.")
161//! @yah:handoff("THE VOCABULARY, and admission rule 2 made UNREPRESENTABLE rather than refused. ServiceComponent gains deploy: DeployTier { Bundle (default), Workload } — oss/yubaba/crates/cloud/src/config.rs. That is the per-component slot [providers.bundle] could not express, and because it is ONE field with two values, \"both bundle-staged and its own workload\" has no spelling at all; there is no rule to enforce. What remained checkable — two components claiming one mount — went into R870-B11's EXISTING cross_ref_validate loop rather than a parallel one. That loop previously filtered on kind and so skipped the workload tier entirely; it now covers both tiers, and only the explanation branches (bundle/bundle = one storage prefix in one bundle; workload/workload = one prefix in the inner-door table; mixed = the mount names two things serving one prefix). skip_serializing_if on the default keeps every existing service.toml byte-identical.")
162//! @yah:handoff("THE JOIN: new module oss/yubaba/crates/cloud/src/inner_door.rs. plan(&ServiceConfig, &domains) -> Result<Option<InnerDoorPlan>>. Rule 1 is by construction — None below two DEPLOYED UNITS, so the negative is assertable on the absence of a plan. Grouping is per unit but mounts are per COMPONENT: N bundle components collapse to one DeployedUnit::Bundle yet keep N mounts, because a bundle sub-mount can carry route headers the root does not and the collapse-to-root shape would silently drop them. Headers come from the DomainRoute whose route_path_prefix equals the component's normalize_mount — cross_ref_validate already PROVES those agree, so the lookup cannot mismatch. Err is reserved for one case: two-plus units with no root mount, which would 503 every unclaimed path. routes_file() refuses an unresolved upstream instead of skipping the mount — a dropped mount does not 503, it falls through to the root and serves the WRONG component with a 200. passway_mount() composes with normalize_mount rather than trimming slashes a second time.")
163//! @yah:handoff("SUPERVISION: WorkloadSpec gains files: Vec<InlineFile { path, content, mode }> (oss/yah-base/crates/workload-spec/src/lib.rs, appended last, serde(default), no skip_serializing_if — postcard is positional, so every pre-existing spec decodes to an empty vec). kamaji's NATIVE backend writes them in spawn_child BEFORE exec and on every respawn (materialize_files, oss/kamaji/crates/kamaji/src/native.rs); containerd/docker/microvm call the new kamaji::reject_unmaterializable_files and REFUSE such a spec by name rather than starting a door against a file that is not there — a silently-skipped route table comes up healthy and routes wrongly, which is worse than not starting. InnerDoorPlan::workload(listen_port, address) renders Workload::Container: argv /usr/local/bin/passway, env PASSWAY_TLS_MODE=plaintext + PASSWAY_LISTEN=127.0.0.1:<port> (the 127.0.0.1 is literal, NOT a parameter, so a wrong port cannot make the door reachable) + PASSWAY_PATH_ROUTES_FILE, and the table itself as the one InlineFile. Not a new Workload variant: TenantPasswayWorkload earns one by carrying config kamaji acts on; an inner door's whole config is an argv, three env vars and a file, so a variant would buy only exhaustive-match churn in peer-owned kamaji-proto (the R572-F1 trade).")
164//! @yah:verify("cargo test -p passway (oss/passway) = 205 lib + 43 + 35 integration, 283 passed / 0 failed, up from the 275 baseline @Ashguard:abde2cbb recorded on R870-T21. 8 new lib tests in tls::tests and 2 new integration tests in tests/path_routes_file.rs. THE END-TO-END ONE IS THE POINT: a_cleartext_inner_door_serves_the_same_mount_table_with_no_certificate forks a REAL passway binary with no PASSWAY_TLS_CERT set at all and asserts the same two-mount split and the same per-mount COOP header over plain http:// — i.e. the tier the operator authorized actually costs a process and nothing else. Its negative, a_cleartext_door_on_a_reachable_bind_refuses_to_start, spawns the binary on 0.0.0.0:0 (the DEFAULT bind, so it is the exact misconfiguration that would make an inner door a public cleartext one) and asserts a non-zero exit whose message names the bind.")
165//! @yah:verify("cargo test -p yah-cloud --lib = 1150 passed / 0 failed (11 new in inner_door::tests, 2 new in config::tests). The cheap half of this ticket's own negative is a_single_unit_service_gets_no_inner_door plus several_bundle_components_are_one_unit_and_still_get_no_door — three components sharing one bundle are still ONE unit and still get no door, which is the case that would be easy to get wrong by counting components. cargo test -p yubaba --lib = 952/0. cargo test -p kamaji --lib --all-features = 208/0 (2 new; the materialization test asserts ORDERING by having the child cat the file into a second path, not merely that the file exists). cargo test -p yah-workload-spec --all-features = 205 + 101, 0 failed. cargo test --workspace --all-features in oss/kamaji = 18+208+5+303, all green in-package.")
166//! @yah:gotcha("ONE PRE-EXISTING FLAKE, DIAGNOSED NOT WAVED THROUGH. kamaji-bin's server::tests::tenant_passway::the_list_reports_the_digest_of_the_spec_it_was_deployed_with FAILS under `cargo test --workspace --all-features` in oss/kamaji, reproducibly, and PASSES 303/303 under `cargo test -p kamaji-bin --lib --all-features` both parallel AND --test-threads=1. So it is cross-PACKAGE contention, not in-package parallelism and not this change: the failing assertion is the second deploy failing to Ack after `free_port()` (server.rs ~:8667) handed back a port another package's test binary had taken between the probe and the bind — a TOCTOU in the helper. Nothing in this ticket adds a port or touches that path; WorkloadSpec::files cannot reach it, since Workload::TenantPassway carries a TenantPasswayWorkload and no WorkloadSpec at all. Worth a real fix (bind-and-hold instead of probe-and-release) but it is not this relay's.")
167//! @yah:gotcha("ROLL ORDER MATTERS AND IS NOT THE USUAL \"JSON IGNORES UNKNOWN KEYS\" ANSWER — flagged by @Ashguard:eclipse (session:e188ccc2, R881-T6) mid-session. All three prod voters now run kamaji+yubaba 0.8.37-h5 (us-south-001 and us-west-001 rolled 2026-09-09; us-east-001 on 0.8.37-h1/h2), all built BEFORE WorkloadSpec::files existed. WorkloadSpec has no deny_unknown_fields, so on the JSON leg an un-rolled node ignores the field exactly as R870-B6's `origin` did. The postcard leg is the one that does NOT forgive: it is positional and non-self-describing, so a new yubaba encoding a spec with a trailing `files` to an old kamaji decoder is a DESYNC, not an ignored key. Before deploying any inner door, confirm which codec that node's kamaji link uses (kamaji-proto/src/codec.rs) and roll kamaji first if it is postcard. Nothing regresses until something actually SETS files — every existing spec encodes an empty vec — but the ordering is a real constraint, not a formality.")
168//! @yah:handoff("WIDER THAN THE TITLE, all mechanical and all compiler-verified. Adding two fields to types this many call sites construct exhaustively meant ~45 initializer repairs across FOUR workspaces: oss/yah-base (workload-spec + local-driver), oss/kamaji (incl. peer-owned kamaji-proto/src/codec.rs and kamaji-containerd-core), oss/yubaba, and the root (crates/yah/hub, app/yah/cli). Each is one line — `files: Vec::new(),` or `deploy: Default::default(),` — with no semantic content; they were driven off E0063 spans, not grep, so none was guessed. NOTE the sweep needs --all-features AND `cargo test --no-run`: `cargo check --all-targets` alone missed sites behind feature gates and in examples/. ALSO REGENERATED (both are pure functions of the tree, so this is not authorship): .yah/schema/{workload,service}.toml.schema.json via `cargo run -p xtask -- emit-schemas` and packages/yah/workload-spec/index.ts via the export-ts bin. Both drift gates still report red because they compare against GIT, and this camp defers commits — they go green with the commit, and the regenerated content is correct.")
169//! @yah:handoff("PHASE 1 DONE — the tier EXISTS and every piece of it is proven in isolation: the cleartext listener (proven through a forked binary), the vocabulary, the join with both admission rules, the wire carrier, and node-side materialization + restart. What is NOT done is the last hop: nothing CALLS plan() yet, so `yah cloud apply` still produces no inner door. That is deliberate rather than abandoned — it is placement work with a live-fleet verify attached, and it is the whole of phase 2.")
170//! @yah:handoff("Tree anchor at handoff: 6f984b53a9dd1a291d29fe7d4cb544b47d4f65e6 — the shared tree as I left it. Diff against it (`git diff 6f984b53a9dd1a291d29fe7d4cb544b47d4f65e6..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
171//! @yah:next("ONE DESIGN QUESTION PHASE 1 LEFT OPEN, stated so it is not rediscovered as a bug. A bundle-tier component at a non-root mount now gets its own entry in the inner-door table pointing at the SAME bundle upstream, purely so the domain manifest's per-path response headers can be applied (see a_bundle_components_sub_mount_keeps_its_headers_and_the_bundle_upstream). That is correct for headers and harmless for routing, but it means the inner door re-states routing the bundle already does internally. If the outer door or the Worker is ALREADY applying those headers for a config-1 service, the inner door would apply them twice — check which tier owns route headers for a passway front door before wiring step 4, because R746 put ROUTE_HEADERS into the Cloudflare Worker and I did not confirm the passway-front-door equivalent.")
172//! @yah:verify("THE LIVE HALF, unrun and needing a fleet: a two-component service whose components deploy independently gets one inner door from one `yah cloud apply`, with https://<host>/ and https://<host>/app/ both 200 and `curl -sI` on /app/ carrying cross-origin-opener-policy: same-origin AND cross-origin-embedder-policy: require-corp while / carries neither. The config-side half of exactly that assertion is already green as inner_door::tests::each_mount_carries_only_its_own_routes_headers, and the transport-side half as the forked-binary cleartext test — what remains unproven is only that apply joins them. THE NEGATIVE'S node-side half is also unrun: for a single-component service (yah-marketing), assert NO routes file is materialized on the node and NO extra workload appears in kamaji's table.")
173//! @yah:next("PHASE 2 IS FIVE STEPS AND EVERY INPUT ALREADY EXISTS. (1) Call inner_door::plan(&svc.service, &cfg.domains) once per service in the apply path; Ok(None) is the common answer and means do nothing at all. (2) Allocate the loopback port. It is deliberately a PARAMETER of InnerDoorPlan::workload rather than config — which port is free is a property of the node — so this is the only genuinely new decision: either take it from kamaji's ledger (oss/kamaji/crates/kamaji/src/ports.rs) or pin one per service. (3) Resolve each DeployedUnit to an address for the `address` closure: DeployedUnit::Bundle is the service's one bundle workload (R870-B11), DeployedUnit::Component(id) is that component's own workload. (4) Deploy the rendered workload in the bundle deploy sequence — BEFORE the outer door is repointed, since the door 503s until its upstream is up. (5) Repoint the outer door's PASSWAY_UPSTREAMS at 127.0.0.1:<port> instead of at the bundle, via IngressPlan::resolve_upstreams (oss/yubaba/crates/cloud/src/reconciler/ingress.rs:397).")
174//! @yah:next("CHECK THE BACKEND BEFORE STEP 4, because getting it wrong is a refused deploy rather than a silent one and you should know which it is. Only kamaji's NATIVE backend materializes WorkloadSpec::files; containerd/docker/microvm call reject_unmaterializable_files and refuse the spec by name. So an inner door must land on a node whose kamaji routes it to Backend::Native. If the fleet's containerd path is where this has to run, the honest fix is to implement the write there (a pre-exec write or a mount), NOT to relax the guard — the guard exists because a door started against an absent route table reports healthy and routes wrongly.")
175//! @yah:handoff("DEFECT IN THIS TICKET'S OWN CHANGE, CAUGHT IN REVIEW BY @Ashguard:eclipse (session:e188ccc2) AND FIXED BEFORE IT LEFT THE TREE. WorkloadSpec rides the postcard `Deploy` frame (kamaji-proto/src/messages.rs:373, and V7's own stanza names `Workload::Container(WorkloadSpec)` as what that frame carries), and kamaji-proto/src/version.rs states the rule twice: every field on a postcard message is mandatory and always encoded, and the only compatibility mechanism is a ProtocolVersion bump. V2/V4/V5/V6 were each exactly \"a field appended to a struct\" and each got one. `files` is that shape and I had not bumped. The reasoning that made me miss it is the one V6's stanza already refutes: `#[serde(default)]` makes an OLD spec decode fine, so the JSON leg really is unaffected — but `default` only affects DEserialization, so a new yubaba still ENCODES a length varint an old kamaji reads as the next field and misparses from there. Now V8, CURRENT = V8, with a stanza naming the wrong reasoning rather than only the rule. cargo test -p kamaji-proto --all-features = 33/0; oss/kamaji workspace = 18+208+5+303+2+2+2+1, 0 failed.")
176//! @yah:gotcha("CORRECTION TO THE FLAKE GOTCHA ABOVE — my characterization was too narrow and would mislead the next reader, so read this one instead. I wrote that the tenant_passway digest test is \"green 303/303 in-package, fails only workspace-wide\". @Ashguard:eclipse measured the counter-example on the same tree: `cargo test -p kamaji -p kamaji-bin --lib --all-features`, in-package and parallel, failed a DIFFERENT test in the same module — deploy_arms_the_declared_socket_and_stop_releases_it (server.rs:8590) — and it failed identically before my sweep. On my own later run the workspace-wide invocation came back 303/0. So the truth is: at least two tests in server::tests::tenant_passway are intermittently flaky in BOTH configurations, the cause is `free_port()` probe-and-release losing the port between the probe and the bind (server.rs ~:8667), and it predates R870-F23. Do NOT read an in-package red there as a regression, and do not read a single green run as proof either. The fix is bind-and-hold; it belongs to neither R870 nor R881 and is unfiled — @Ashguard:eclipse tried and board.open refused for want of a parent relay.")
177//! @yah:gotcha("CONSEQUENCE OF THE V8 BUMP FOR ANY FLEET OPERATION, not just for this relay — relayed by @Ashguard:eclipse (session:e188ccc2) who is holding the fleet on R881-T6, and worth acting on before the next roll. The tree is now ProtocolVersion::V8; EVERY node runs a pre-V8 pair (us-east-001 on 0.8.37-h1/h2, us-south-001 and us-west-001 on 0.8.37-h5, the other six on 0.8.28-0.8.34). Nothing is broken, because each node is internally matched and the protocol is a node-local UDS. What changed is that `hotship --binaries yubaba` ALONE — or `kamaji` alone — is now a footgun on every node: it puts a V8 binary against a V7 sibling, and per version.rs:71 that does not fail cleanly, it misreads every field after the desync, \"which is how a wrong image or a wrong volume mount gets deployed instead of an error\". Ship the PAIR. That was harmless before this ticket and is not now.")
178//! @yah:handoff("PHASE 2 LANDED — `yah cloud apply` now produces an inner door. All five steps, with the call sites. (1) PLAN: `service_inner_door` (app/yah/cli/src/cloud.rs:9212) calls `inner_door::plan` once per service; `Ok(None)` is the answer for every service on disk today and returns before anything else runs. (2) PORT: derived, not allocated — `inner_door::listen_port(service)` (oss/yubaba/crates/cloud/src/inner_door.rs:346). (3) RESOLVE: `InnerDoorPlan::resolve_addresses` (inner_door.rs:418) maps each unit to a mesh ident via `unit_ident` (:398) and looks it up with the new `ServiceRecordFanout::address_for_ident` (reconciler/service_discovery.rs:426). (4) DEPLOY: `deploy_inner_door` (cloud.rs:9274), called at the END of the deploy-phase closure in BOTH apply paths — `reconcile_root` (cloud.rs:11877) and `handle_mirror_up` (cloud.rs:7100) — so it is after every unit registered a record and before the front-door phase repoints anything. (5) REPOINT: `IngressPlan::point_at_inner_door` (reconciler/ingress.rs:438), called from `reconcile_ingress_edge` (cloud.rs:7726).")
179//! @yah:handoff("STEP 2 ANSWERED — the port is DERIVED from the service name, not taken from kamaji's ledger, and the three facts that decided it were read rather than assumed. (a) `LedgerPorts` is node-local (a JSON file beside the supervisor's state dir) and yubaba's HTTP surface exposes no allocation verb at all — yubaba/src/lib.rs routes /workloads/*, /services, /node/*, and nothing for ports — so an apply has no way to ask. (b) A stated number is HONOURED, not rejected, on the path this workload takes: R844-F14's pin rule bites inside `LedgerPorts::resolve_set`, and `NativeRuntime::resolve_declared_ports` (oss/kamaji/crates/kamaji/src/native.rs:280) filters `pin.is_none()` BEFORE calling it. That matters because `PASSWAY_LISTEN` must carry the number, and a number the node picks after the spec is rendered cannot be in it. (c) A collision is not representable: the ledger allocates on the workload's MESH ip, an inner door binds loopback, so 100.64.0.3:14210 and 127.0.0.1:14210 are different sockets. The window is 10000-19999, deliberately below Linux's default ephemeral floor (32768) where `pick_free_port`'s bind(:0) draws from. FNV-1a written out inline rather than `DefaultHasher`, whose stability std does not promise — this number goes into a deployed door's env AND the outer door's upstream list, and a toolchain bump silently moving it would repoint one tier and not the other.")
180//! @yah:handoff("THE OPEN HEADER QUESTION IS ANSWERED, AND THE ANSWER IS NO CHANGE — grounded by reading, not assumed. The question was whether the passway FRONT door also applies per-route response headers. It does not: `PassProxy::response_filter` (oss/passway/crates/passway/src/proxy.rs:700) iterates `ctx.route_headers`, and its own doc at :694 states that vector is empty for `RoutingStrategy::ByHost` — which is what every outer door is. So the outer tier owns no headers and there is no double-apply to resolve there. A THIRD tier the question did not name does apply them, and is worth recording: the mesofact bundle ORIGIN, via `MESOFACT_ROUTE_HEADERS` set by `add_declared_route_headers` (app/yah/cli/src/cloud.rs:8612). For a mount served by `DeployedUnit::Bundle` both that origin and the inner door apply the route's headers — but CONVERGENTLY, not duplicatively: both read the same `.yah/domains` route map, `PathRouter` does not strip the mount prefix (oss/passway/crates/passway/src/path_route.rs has no strip/rewrite), so both match the same request path, and both use insert-semantics (`HeaderMap::insert` in mesofact's `RouteHeaderTable::apply`, `insert_header` in passway) — one header, one value. Do NOT collapse it to one owner. The bundle origin's coverage is strictly WIDER: it applies headers for a declared route that has no component mount (a `/docs/*` route served out of the root bundle's dist), which the inner door has no entry for. And the inner door is the ONLY owner for a `DeployTier::Workload` mount, since nothing hands such a component a header table. The two are complementary; removing either loses headers somewhere.")
181//! @yah:handoff("PLUMBING BUILT BECAUSE STEPS 3 AND 5 NEEDED IT, all three of which did not exist. (1) `inner_door::component_workload_ident(service, component_id)` (inner_door.rs:371) — the mesh identity a workload-tier component registers under. It is a NAMING RULE stated here because nothing else states it: a bundle's ident is a mirror fact (`BundleSlot::workload_name`, renameable with `name = \"...\"`), but a workload-tier component has no slot of its own, since `[providers.*]` is per-kind-per-mirror — the exact gap `DeployTier` was added to close. Folded through `reconciler::native_support::sanitize_ident`, which I widened from private to `pub(crate) mod` (reconciler/mod.rs) rather than writing a second normalizer. Getting the ident wrong fails LOUDLY: `routes_file` refuses a mount whose unit resolved to nothing, naming the unit. (2) `ServiceRecordFanout::address_for_ident` — deliberately SINGULAR where `upstreams_for` is plural. An inner door proxies over loopback to a unit on its own node; handed a fleet-wide set it would dial across the mesh, which is not what the cleartext-listener safety argument assumed. Two nodes, two addresses is ambiguity (None), not load balancing. Port selection follows `port_for`'s discipline exactly (`kamaji::DEFAULT_PORT_NAME` first, then the sole anonymous port) so a unit resolves the same way at both tiers or neither. (3) `IngressPlan::point_at_inner_door` OVERRIDES where `resolve_upstreams`/`resolve_ports` fill in — it clears both halves and then goes through those same two methods, so this stays the only place in the crate writing those fields. The ticket's step 5 named `resolve_upstreams`; used alone it is WRONG, because it skips a rule that already has an `upstream_host` and every mirror on disk pins one. A pin names ONE unit, and fronting a two-unit service from one unit serves half the site and 503s the other half, so the pin has to lose here and nowhere else.")
182//! @yah:handoff("TWO PLACEMENT DECISIONS PHASE 2 HAD TO MAKE, both recorded at the site. (a) The inner door lands on the FRONT DOORS, not the workload nodes — the outer door dials 127.0.0.1, so a door anywhere else is a door the outer tier cannot reach. `ingress_topology` (cloud.rs:9230) recomputes `resolve_ingress_placements` + `plan_ingress` in the deploy phase to learn that set; both are pure, so this costs no network and cannot disagree with the front-door phase's own answer. (b) SELF-DISCOVERY IS TURNED OFF for an inner-door service. `PASSWAY_UPSTREAM_SOURCE=yubaba` makes the door poll for the fronted workload's records and use those INSTEAD of its static set — which would route straight past the inner door to whichever unit registered under the mirror's ident, silently undoing step 5. The rendered note says so in its own words rather than reusing R844-F20's \"NOT self-discoverable ... MANUAL step\" wording, because this is not a degradation: the address is derived and byte-identical on every apply. (c) A mirror with two units and NO declared front door SKIPS with a note rather than failing — `reconcile_mirror_ingress` already returns early on `plans.is_empty()`, so there would be no outer door to repoint and nothing that can 503. Every `shape = \"local\"` dev mirror is in that state; bailing there would have broken `yah mirror up`.")
183//! @yah:handoff("DISCOVERED WORK, FIXED IN THIS PASS, NOT FILED AS A FOLLOWUP. `cargo test -p yah-cloud --lib` was 1161/2 on arrival, and the two reds were NOT mine and NOT a flake: `cloud_init::tests::{rendered_runcmd_entries_are_all_strings, coordinator_prestage_only_for_standalone}`. Cause: oss/yubaba/crates/cloud/templates/mirror.yml:107-108, the two R858-F17 turso-backup-helper runcmd entries, were written as BARE YAML scalars containing a `: ` — which makes the whole entry parse as a Mapping, so cloud-init skips it and the helpers never land on a provisioned node. The file is committed and clean (last touched by a8f0d501, i.e. it regressed AFTER phase 1's 1150/0 measurement), no live peer owns it, and the fix is two lines: double-quote the entries and escape the inner quotes. Both tests are green and the comment at the site names the gate. This is a real provisioning defect, not just a red test — a node provisioned since a8f0d501 has no turso-backup-hydrate / turso-backup-tail, and R858-F17's own design makes durability-declaring workloads refuse to deploy without them. Worth a look at whether any node was provisioned in that window.")
184//! @yah:verify("PHASE 2 MEASURED, every number run by me and read. `cargo test -p yah-cloud --lib` = 1163 passed / 0 failed (baseline 1150; +13 — 6 in inner_door::tests, 4 in service_discovery::tests, 3 in ingress::tests). `cargo test -p yah --lib` = 1549 / 0 (+3 new in a new `inner_door_apply_tests` module). `cargo test -p xtask --test main mirror_ingress` = 13 / 0 (baseline 11; +2). `cargo test -p yubaba --lib` = 952 / 0, exactly the baseline. `cargo test -p passway` in oss/passway = 205 + 43 + 37 = 285 / 0 against the 283 baseline, and `cargo test -p kamaji --lib --all-features` = 217 / 0 against 208 — BOTH deltas are peers', not mine: I touched neither crate. Sweeps: `cargo test --workspace --all-features --no-run` clean, and the same in oss/yubaba clean (only the two pre-existing unused-import warnings in a peer's in-flight mesofact_static.rs). NO SCHEMA REGEN NEEDED — this pass added functions, constants and one module-visibility widening, and no serde-visible field on any generator input, so .yah/schema/*.json and packages/yah/workload-spec/index.ts are untouched by construction.")
185//! @yah:verify("THE NEGATIVE IS ASSERTED IN THREE PLACES, at three different altitudes, because it is the claim the live fleet rests on. (1) `xtask/tests/mirror_ingress.rs::no_service_on_disk_gets_an_inner_door` walks the REAL `.yah/services/` tree and asserts every service plans `None`. That is the strongest form available without a fleet: the only way to be wrong about it is for a service to acquire `deploy = \"workload\"`, at which point the test names the service. Sibling `every_services_derived_inner_door_port_is_distinct` pins the port derivation against the real service list. (2) `cloud::inner_door_apply_tests::a_single_unit_service_leaves_the_outer_door_exactly_as_it_was` builds a two-component fixture that is BYTE-FOR-BYTE the positive test's, with one word changed (`workload` -> `bundle`), and asserts the rendered `PASSWAY_UPSTREAMS` is still the mirror's pinned `noisetable.com=100.64.0.3:8080`. So the difference between the two outcomes is provably that one field. (3) `inner_door::tests::{a_single_unit_service_gets_no_inner_door, several_bundle_components_are_one_unit_and_still_get_no_door}` from phase 1, still green. THE POSITIVE: `a_two_unit_service_repoints_the_outer_door_at_its_inner_door` (outer door renders `noisetable.com=127.0.0.1:<derived>`, and the port is asserted equal to what the door itself binds — two call sites in two phases that must not be able to disagree) and `the_rendered_table_splits_the_mounts_and_carries_only_their_own_headers` (both units addressed, COOP+COEP on /app and ABSENT on the root).")
186//! @yah:gotcha("TRANSIENT BUILD FAILURE SEEN AND DISPROVEN, recorded so the next reader does not re-chase it. The first `cargo test --workspace --all-features --no-run` came back with `can't find crate for 'runner'` / `'agent_tools'` / `'camp_service'` and a linker failing on a dozen absent `.rlib`s (libgif, libzune_jpeg, libimagesize...) in crates this ticket never touched — the exact shape CLAUDE.md's orphan-gc warning describes. Followed that procedure rather than cleaning: `cargo orphan-gc log -n 300` names NONE of the missing artifacts (every entry in the window reads `deleted 0 artifacts`), so orphan-gc is NOT confirmed here. The likelier cause is plain target-dir contention: a `yah-release-check` QED pipeline was holding the same `/Users/leif/ss/yah/target` for 29 minutes alongside this build. Re-ran with nothing else on the key: CLEAN, zero errors. Not reproducible, orphan-gc log does not name it, and the artifacts were never deleted per its own record.")
187//! @yah:next("WHAT REMAINS IS THE LIVE HALF ONLY, and it is an operator call the R870 leader is holding — phase 2 deliberately landed code + tests and touched no node. The two assertions: (a) POSITIVE — a two-component service whose components deploy independently gets one inner door from one `yah cloud apply`, with https://<host>/ and https://<host>/app/ both 200 and `curl -sI` on /app/ carrying cross-origin-opener-policy: same-origin AND cross-origin-embedder-policy: require-corp while / carries neither. (b) NEGATIVE, node-side — for a single-component service (yah-marketing), NO routes file materialized under /var/lib/passway/routes and NO extra workload in kamaji's table. Note that (b) is now also asserted statically against the real tree by `xtask/tests/mirror_ingress.rs::no_service_on_disk_gets_an_inner_door`, so the node-side check is confirmation rather than discovery. BEFORE RUNNING (a): there is no service with `deploy = \"workload\"` on disk, so one has to be declared first — and the R870-B6/V8 roll-order gotcha on this ticket applies the moment anything actually SETS `WorkloadSpec::files`. Confirm the target node's kamaji link codec (kamaji-proto/src/codec.rs) and roll the kamaji+yubaba PAIR first if it is postcard.")
188//! @yah:next("ONE THING PHASE 2 DID NOT BUILD, named so it is not mistaken for done: there is still no FLEET deploy path for a `DeployTier::Workload` component. `reconcile_component` (app/yah/cli/src/cloud.rs:8264) dispatches on `component.kind`, and the only non-bundle arms are `container` — which `ContainerReconciler::up` guards on `MirrorShape::Local`, and `LocalProcessReconciler`, which is the camp/dev tier and registers as `local-process-<service>-<env>-<component>`. So on a real mirror such a component is deployed by hand today (`yah cloud workload deploy`). That is exactly why `inner_door::component_workload_ident` had to STATE the ident rather than look it up. The failure mode is loud rather than silent — a component registered under any other ident leaves its unit unresolved and `routes_file` refuses the whole table, naming the unit — but whoever wires that deploy path must make it register under `component_workload_ident(service, id)`, or change both sides together. Related and already filed: R523-F1 (a component kind that deploys a stateful binary to a fleet node) is the same missing arm seen from the other direction.")
189//! @yah:handoff("PHASE 2 COMPLETE — `yah cloud apply` produces an inner door. Everything above this entry is the detail: the five call sites, the derived-port argument, the header-ownership answer (no change — the outer passway door owns no route headers, proven at proxy.rs:694/700, and the bundle origin's overlap is convergent and strictly wider), the three pieces of plumbing built because steps 3 and 5 needed them, the two placement decisions, and the one mirror.yml provisioning defect fixed on the way through. Nothing was deployed and no node was touched, per the dispatch. Green: yah-cloud 1163/0, yah 1549/0, yubaba 952/0, xtask mirror_ingress 13/0, passway 285/0, kamaji 217/0, both --all-features --no-run sweeps clean. Git policy is `defer`, so nothing is committed — the diff is 6 files: oss/yubaba/crates/cloud/src/{inner_door.rs, reconciler/mod.rs, reconciler/ingress.rs, reconciler/service_discovery.rs}, oss/yubaba/crates/cloud/templates/mirror.yml, app/yah/cli/src/cloud.rs, plus xtask/tests/mirror_ingress.rs.")
190//! @yah:handoff("Tree anchor at handoff: 88533e01f7f578b1520b633d05846973fa47f608 — the shared tree as I left it. Diff against it (`git diff 88533e01f7f578b1520b633d05846973fa47f608..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
191//! @yah:handoff("PHASE 2 ACCEPTED BY THE RELAY LEADER (@Ashguard:hydra, session:39386823). `yah cloud apply` now produces an inner door: all five steps wired, plus three pieces of plumbing that did not exist (`component_workload_ident`, `ServiceRecordFanout::address_for_ident`, `IngressPlan::point_at_inner_door`), the derived-port decision argued from three read facts, and the open header question answered NO CHANGE with the proof at proxy.rs:694/700. Implemented by @Ashguard:blade (session:54a6be05). The detail is in the handoff entries above this one; this entry records only that it was accepted and on what evidence.")
192//! @yah:verify("WHAT IS DELIBERATELY NOT VERIFIED, and it is the operator's call rather than an oversight: the LIVE half. No node was touched, nothing was deployed, nothing committed (git policy is `defer`). Running it needs a service with `deploy = \"workload\"` declared — none exists on disk — and the moment anything actually SETS `WorkloadSpec::files`, this ticket's own V8 roll-order gotcha binds: confirm the target node's kamaji link codec (kamaji-proto/src/codec.rs) and roll the kamaji+yubaba PAIR, never one alone.")
193//! @yah:verify("INDEPENDENTLY RE-RUN BY A SECOND COURIER (@Ashguard:dove, session:60d4f41f) who did not implement it, because a courier's self-report is the inner gate and not the outer one. All six commands reproduced the claimed counts EXACTLY: yah-cloud 1163/0 (4 ignored), yubaba 952/0, passway 285/0 (205+43+37), kamaji --lib --all-features 217/0, yah --lib 1549/0 (1 ignored), workspace --all-features --no-run clean. Every content check held: `point_at_inner_door` at ingress.rs:438; `inner_door::plan` reached from the apply path via `service_inner_door` (cloud.rs:9219) through `deploy_inner_door` (cloud.rs:9274, invoked at 7100 and 11877) and the ingress repoint at 7726; the port confirmed a deterministic per-service pin (FNV-1a into 10000-19999, inner_door.rs:346) and NOT kamaji's ledger, with the native.rs:282 `pin.is_none()` justification verified at the site. The negative is asserted three times, not once. CAVEAT ON THE MEASUREMENT ITSELF: the camp skew detector flagged 4 of 6 runs SUSPECT — peers edited kamaji/src/microvm.rs, kamaji-bin/src/main.rs and cloud/reconciler/mesofact_bundle.rs mid-run — so these are shared-tree numbers, not a frozen-tree measurement.")
194//! @yah:verify("ONE CLAIM CORRECTED AND ONE DEFECT FOUND BY THAT RE-RUN, both recorded rather than smoothed over. (1) CORRECTION: `cargo test -p yah-cloud --lib` does NOT run from the repo root — yah-cloud is not a root workspace member and needs dev-dependencies; it only works from `oss/yubaba`. Anyone reproducing the 1163/0 above must cd there first. (2) DEFECT, pre-existing and NOT caused by this ticket: `embedded_template_matches_workspace_canonical` is green VACUOUSLY — it resolves the workspace root via CARGO_MANIFEST_DIR.ancestors() to oss/yubaba, whose .yah/ holds only a .gitignore, so it takes the bootstrap branch and asserts nothing, while the repo-root twin at .yah/infra/cloud-init/mirror.yml is 128 diff-lines stale and missing the whole R858-F17 turso-backup block. FILED AS R870-B25, not left here. Note `rendered_runcmd_entries_are_all_strings` is a DIFFERENT test, is genuinely green, and is the gate that really catches the colon-space footgun this ticket fixed.")
195
196use anyhow::{bail, Context, Result};
197use serde::{Deserialize, Serialize};
198use std::collections::{BTreeMap, HashMap};
199use std::path::Path;
200use thiserror::Error;
201use workload_spec::secrets::SecretAccess;
202use workload_spec::sovereign::Membership;
203pub use workload_spec::sovereign::SovereignRole;
204use workload_spec::{validate, LifecycleArchetype, Locality, TenantId, WorkloadSpec};
205
206/// Static node capacity declaration on `machine.toml` (R572-F3).
207///
208/// `memory_mb` and `cpu_millis` express the node's *total* hardware budget.
209/// F5's bin-packer subtracts the sum of committed workload requests from
210/// this floor to determine available headroom; an absent `allocatable`
211/// block means no capacity constraint is enforced (any workload fits).
212#[derive(Debug, Clone, Serialize, Deserialize)]
213#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
214pub struct NodeAllocatable {
215 /// Total physical RAM in mebibytes (e.g. 512 for a 512 MB node).
216 pub memory_mb: u32,
217 /// Total CPU in k8s millicores (1000 = 1 core, 250 = 0.25 CPU).
218 pub cpu_millis: u32,
219}
220
221/// `[registration]` — facts **observed** about a running box, written by the
222/// fleet rather than declared by an operator (R707-T1).
223///
224/// The rest of `machine.toml` is *declaration*: intent, operator-authored,
225/// reviewed and diffed like any other source. This block is the other half —
226/// what the box turned out to be once it booted and joined. Keeping the two
227/// apart is what lets the published fleet index (R707-F3) say which half it is
228/// carrying; publishing them under one schema would bake the confusion into a
229/// permanent record.
230///
231/// The split is a **provenance** boundary, not a trust or reach one:
232/// - *Declaration* answers "what did we ask for" — `name`, `region`, `arch`,
233/// `mesh_tags`, `[allocatable]`, and the declared reach in [`ConnectSpec`].
234/// - *Registration* answers "what did we observe" — the hostkey TOFU'd at
235/// attach, the mesh address headscale assigned at join.
236///
237/// It stays in the git-tracked TOML on purpose. Registration is not local
238/// scratch state: every consumer needs the mesh address to dial a node, so it
239/// has to travel with the declaration. (`.yah/infra/state/machines/<name>.json`
240/// — [`crate::state::MachineState`] — remains the *gitignored* sidecar for
241/// provider-side derivatives that nobody but this camp needs.)
242#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq, Eq)]
243#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
244pub struct MachineRegistration {
245 /// Yubaba's ed25519 `/identity` fingerprint, TOFU-recorded by
246 /// `yah cloud machine attach` on first contact (`SHA256:…`). An observed
247 /// property of a running process — not the operator's intent — which is
248 /// why it moved out of the top level here.
249 #[serde(default, skip_serializing_if = "Option::is_none")]
250 pub hostkey_fingerprint: Option<String>,
251 /// Mesh (headscale/tailnet) IPv4 assigned at join, e.g. `"100.64.0.1"`.
252 /// Bare address, not a URL: the *port* is declared reach and lives on
253 /// [`ConnectSpec::yubaba_port`]. [`MachineConfig::yubaba_url`] composes the
254 /// two. Absent until the node has joined the mesh.
255 #[serde(default, skip_serializing_if = "Option::is_none")]
256 pub mesh_ipv4: Option<String>,
257 /// RFC3339 timestamp of the mesh join that produced `mesh_ipv4`. Free-form
258 /// audit; nothing keys off it.
259 #[serde(default, skip_serializing_if = "Option::is_none")]
260 pub joined_at: Option<String>,
261}
262
263impl MachineRegistration {
264 /// True when nothing has been observed yet — used to omit the whole
265 /// `[registration]` table from a serialized machine TOML.
266 pub fn is_empty(&self) -> bool {
267 self.hostkey_fingerprint.is_none() && self.mesh_ipv4.is_none() && self.joined_at.is_none()
268 }
269}
270
271/// Per-machine TOML from `.yah/infra/machines/<name>.toml`.
272///
273/// Two halves, split by provenance (R707-T1): everything here is *declaration*
274/// — operator intent under review and blame — except [`registration`], which
275/// carries what the fleet observed. See [`MachineRegistration`] for why the
276/// boundary is drawn there and what depends on it.
277///
278/// @yah:ticket(R860-T5, "Model per-node native-exec capability as an admission axis (W338 §Placement consequences 3 / R858-T4 gap)")
279/// @yah:status(review)
280/// @yah:phase(P1)
281/// @yah:at(2026-09-05T18:29:19Z)
282/// @yah:assignee(agent:bundle-anthropic-ashguard)
283/// @yah:parent(R860)
284/// @yah:next("Cheapest defensible shape: express it on MachineConfig, which already has the two vocabularies — `mesh_tags: Vec<String>` (config.rs:246, superset match, already carries `arch:`/`os:`/`tag:build-worker`) and `taints: Vec<String>` (config.rs:337). A `native-exec` mesh tag required by any group member whose kind is native is a one-line admission axis in `admission_spec()`. Whichever is chosen, it must be declared in .yah/infra/machines/*.toml for the nodes that actually run kamaji with --native-exec-dir, and `check_inert_taints` (config.rs:703) lints unread taint keys dead — so a taint nobody reads will be flagged.")
285/// @yah:verify("cargo test -p cloud --lib config")
286/// @arch:see(.yah/docs/working/W338-workload-dependencies-and-appliance-composition.md)
287/// @yah:depends_on(R860-T4)
288/// @yah:gotcha("Verified 2026-09-04: native-exec capability is modelled NOWHERE in placement — `rg \"native\" oss/yubaba/crates/cloud/src/config.rs` returns zero hits, and the raft state machine models no member attributes, labels or taints at all (`rg \"taint|capabilit|labels|mesh_tag\"` over raft/{mod,store,network}.rs yields one unrelated comment at raft/store.rs:591). Native-exec is a node-local kamaji startup decision today: `--native-exec-dir` (oss/kamaji/crates/kamaji-bin/src/main.rs:152-156, :51-55) plus the `native-exec` cargo feature (kamaji-bin/src/server.rs:329-330). A node without it refuses the deploy at dispatch time and nothing upstream can see that in advance — which is exactly the deploy-time surprise W338 wants turned into a placement precondition.")
289/// @yah:handoff("NATIVE-EXEC IS NOW A PLACEMENT PRECONDITION, NOT A DISPATCH-TIME SURPRISE. New `pub const NATIVE_EXEC_MESH_TAG: &str = \"cap:native-exec\"` in oss/yubaba/crates/cloud/src/config.rs (declared just above `node_selector_mesh_tags`), and one axis in `admission_spec()` immediately after the R860-T4 group loop: if ANY member of `placement_group(ws, declared)` returns true from `WorkloadSpec::wants_native_exec()`, the tag is appended to the derived `RequiredSpec.mesh_tags` (deduped). No new field on `RequiredSpec`, no signature change anywhere, no wire or serde change — the mesh_tags axis is already an AND-ed superset check against `machine.mesh_tags` in `matches` and is already rendered by `describe`, so a refusal now reads `required.mesh_tags=[...,cap:native-exec]`.")
290/// @yah:handoff("ITEM 1 — HOW A NATIVE WORKLOAD IS DETECTED, settled by opening the type rather than guessing. There is no `kind` on `WorkloadSpec`: on the wire a native workload is still `Workload::Container(WorkloadSpec)`, and the ONLY difference is the annotation `yah.exec = native`, read through `WorkloadSpec::wants_native_exec()` (oss/yah-base/crates/workload-spec/src/lib.rs:2939; consts `NATIVE_EXEC_ANNOTATION` / `NATIVE_EXEC_VALUE` at :3402/:3407). That accessor is what the admission axis calls — matching kamaji, whose `deploy_container` checks the same marker first and routes to `deploy_native_exec` (oss/kamaji/crates/kamaji-bin/src/server.rs). The `yah.exec` key is a substrate selector with a second value, `microvm` (`wants_microvm`, same key, R605-F8), so per-node microVM capability is the obvious sibling axis and is NOT modelled here — see next-steps.")
291/// @yah:handoff("ITEM 2 — DECLARATIONS LANDED ON TWO NODES, FROM READINGS RECORDED IN-REPO, NOT INFERRED. `cap:native-exec` added to `mesh_tags` in .yah/infra/machines/us-west-001.toml and .yah/infra/machines/us-west-003.toml, each with a comment naming its evidence and its re-check condition. us-west-001: the R858 gotcha in its own header records a `ps` reading taken on the box 2026-09-05 — pid 515908 is `/usr/local/bin/kamaji --native-exec-dir /var/lib/yah/kamaji/native`, supervising headscale as a native child. us-west-003: its header's 'THE DEPLOYED KAMAJI PREDATES THE microVM BACKEND' note quotes the box's actual ExecStart, read over ssh 2026-09-01, carrying `--native-exec-dir /var/lib/yah/kamaji/native` (corroborated by .yah/docs/architecture/A043-yah-on-machine-daemons.md's @yah:verify for the same probe). Both comments say plainly that the capability lives in the systemd unit's ExecStart, not in the TOML, so it must be re-checked after any roll.")
292/// @yah:handoff("ITEM 2, THE NEGATIVES — TWO NODES ARE KNOWN NOT TO HAVE IT AND WERE DELIBERATELY LEFT UNSET. us-south-001: kamaji refused headscale there 2026-09-03 with 'native backend not configured — start kamaji with --native-exec-dir' (the R858 chain, quoted in .yah/infra/machines/us-west-001.toml and W267). I did NOT edit us-south-001.toml — it was already dirty in the working tree at the anchor SHA and @Ashguard:eclipse is live on R858, so I left it alone rather than race it; the mechanism fails closed there, which is the correct state. us-west-015 (the sole darwin builder): W254-darwin-build-nodes.md's own next-step records that its kamaji is built/started `--docker` only. I added a comment to us-west-015.toml explaining that the tag is deliberately absent, that this is the node where the axis changes an error message (a darwin build row is native by construction, so it is now refused at ELECTION naming cap:native-exec instead of reaching the box and being refused by kamaji), and the exact enable sequence: rebuild with `--features native-exec`, restart with `--native-exec-dir <dir>`, THEN add the tag. us-west-002/011/013/014 are unestablished from the repo and left unset. THE OPERATOR-FACING ANSWER: the file is `.yah/infra/machines/<node>.toml` and the key is `mesh_tags`; add the literal string `cap:native-exec` to that array, and only after the roll.")
293/// @yah:handoff("DECISIONS THE BRIEF LEFT OPEN, all recorded in doc comments at the site. (1) MESH TAG, NOT TAINT — as recommended, and the doc says why in the terms the brief asked for: mesh tags are positive capability with superset matching ('this node CAN'), which is the claim being made; a taint is repulsion and would have to be inverted to `no-native-exec` on every node LACKING the backend (declaration burden on the majority, and silently wrong for a node nobody has edited) AND taught to `taint_effect`, or `check_inert_taints` would correctly lint the key dead. (2) THE `cap:` NAMESPACE IS NEW. Live prefixes are `tag:` (operator-assigned role), `arch:`/`os:` (silicon and userland facts, emitted as requirements by qed::platform::build_worker_mesh_tags), and `tier:` which R763 RETIRED for architecture and reserved for the environment axis — so reusing any of them would have stated the wrong kind of fact. A capability the daemon was configured with is none of those. Nothing validates tag prefixes (only `check_retired_arch_tags` looks at one), so this costs no wiring. (3) COMPUTED OVER THE GROUP, not the requirer — that is literally W338's sentence ('supply = self specs must be placeable where their requirer lands'), and the second test proves it: an ordinary container requirer with a `local` edge to a native provider is pulled onto a capable node. (4) FAILS CLOSED, accepted deliberately: an undeclared node is simply not a candidate, so an undeclared fleet reports 'no node admits' at election rather than dispatching to a node that refuses. Nothing in `.yah/infra/workloads/` is native-marked today (only yah-cloud-admin.toml exists there), so the only live consumer is the qed darwin build row, where failing closed is strictly the better error.")
294/// @yah:handoff("BLAST RADIUS, MEASURED. `admission_spec` is private and its callers are unchanged: `admit_workload` / `admit_workload_candidates` / `admit_workload_in_group` (config.rs), reached from app/yah/cli/src/cloud.rs (deploy, rolling, topology analyzer), app/yah/cli/src/yubaba_client.rs `elect_node`, and cloud/src/migrate.rs. The headscale appliance path inside yubaba (headscale_appliance.rs) does NOT go through admission — it is node-internal — so nothing eclipse holds on R858 is touched by this. Files edited, in full: oss/yubaba/crates/cloud/src/config.rs; .yah/infra/machines/{us-west-001,us-west-003,us-west-015}.toml. Nothing in oss/yubaba/crates/yubaba/ was opened, and oss/kamaji/crates/kamaji-bin/src/server.rs was READ ONLY (to confirm the marker check), per @Ashguard:hydra's contention triage.")
295/// @yah:handoff("ONE SCOPE ADDITION, stated loudly rather than slipped in: `MachineConfig::mesh_tags` (config.rs:256) had NO doc comment at all — the operator-facing declaration key for four tag namespaces was undocumented. I gave it one enumerating `tag:` / `arch:`+`os:` / the new `cap:` / retired `tier:`, and noting that nothing validates the prefix (which is why the two lints exist). CONSEQUENCE TO KNOW: that field's doc is the source of the `mesh_tags` description in the GENERATED .yah/schema/machine.toml.schema.json, so it is schema-drift-affecting — see the gotcha.")
296/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I found it and left it. Quote this SHA rather than 'HEAD' in any revert/restore instruction; to undo a hunk, read it with `git show 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2:<path>` and put it back with Edit, never `git checkout`/`restore` (they restore whole files and would delete peers' uncommitted work).")
297/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
298/// @yah:next("MICROVM IS THE IDENTICAL UNMODELLED GAP, one line away. `yah.exec` is a substrate selector with a second value: `WorkloadSpec::wants_microvm()` (workload-spec/src/lib.rs, R605-F8), and kamaji constructs MicroVmRuntime only when started with `--microvm-dir` — A043's probe records that us-west-003's deployed kamaji has `--native-exec-dir` but NOT `--microvm-dir`, so a microvm-marked deploy is refused there by exactly the same dispatch-time surprise this ticket removed for native. The shape is `cap:microvm` alongside NATIVE_EXEC_MESH_TAG in the same `if` in `admission_spec`. Not done here because no node in the fleet can host one yet (R605-F14 must land a guest kernel + rootfs first), so declaring the tag anywhere today would be the wrong fact.")
299/// @yah:next("us-south-001 needs `cap:native-exec` DECIDED, not defaulted, and it is the R858 node. It is the one machine the repo positively records as LACKING the backend (kamaji refused headscale there 2026-09-03), so leaving the tag off is correct TODAY — but if R858's fix is 'give us-south-001 a native-capable kamaji' rather than 'stop moving headscale', then the roll and the tag must land together, in that order. I left .yah/infra/machines/us-south-001.toml untouched because it was already dirty at anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 and @Ashguard:eclipse is live on R858.")
300/// @yah:next("R860-T6 (`supply = \"self\"` provisioning) inherits this for free — `admission_spec` already requires the capability of the whole group, so a self-provisioned native member cannot be elected onto a node that cannot run it. What T6 must still not do is re-elect per member: reuse the node URL `elect_node` returned for the requirer, per R860-T4's handoff.")
301/// @yah:verify("BASELINE RECORDED BEFORE EDITING, at tree anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2: `cargo test -p yah-cloud --lib` from oss/yubaba = 1090 passed / 0 failed / 4 ignored, exit 0 — exactly the count the brief predicted. AFTER: 1093 passed / 0 failed / 4 ignored, exit 0 (+3, exactly the three tests added). `cargo check -p yah-cloud --all-targets` exit 0 and `cargo check -p yubaba --all-targets` exit 0 (yubaba consumes cloud, so it is where any signature change would surface — there is none). Every exit code echoed explicitly via an `EXIT=$?` / `${PIPESTATUS[0]}` marker and read back, never inferred from an empty grep. The four `yah-cloud` warnings are all pre-existing and in other files (object-store r2.rs, reconciler/mesofact_static.rs unused imports, app_manifest.rs, reconciler/mod.rs non_snake_case); config.rs contributes none.")
302/// @yah:verify("NEW TESTS (config.rs `mod tests`, R860-T5 section at the end, after the R860-T4 block). (1) a_node_without_the_native_exec_capability_cannot_host_a_native_workload — a `yah.exec = native` spec is refused by a bare node with an error naming `cap:native-exec`, and admitted by a node declaring it, with both nodes in the same fleet so the choice is provably the tag. (2) a_local_edge_to_a_native_provider_makes_the_requirer_need_the_capability — an ordinary container requirer (asserted `!wants_native_exec()`) with a `local` edge to a native provider lands on the capable node, while the SAME spec without the edge still lands on the plain one, so the constraint provably comes from the group. (3) a_group_with_no_native_member_does_not_require_the_capability — the regression guard: the axis is absent from `admission_spec`'s mesh_tags and a group with a local edge between two ordinary specs still admits on a node declaring nothing. Helper `native_spec()` asserts the marker reads back through `wants_native_exec()` before the test uses it, so a typo cannot make the test pass vacuously.")
303/// @yah:verify("Machine-config lints were considered and are unaffected by construction: `check_inert_taints` reads `taints` (I touched none), and `check_retired_arch_tags` flags only the `tier:` prefix. `cap:` is a new namespace and nothing validates prefixes, so no lint fires and no lint needs teaching.")
304/// @yah:gotcha("SCHEMA DRIFT IS EXPECTED FROM THIS TICKET AND WAS ALREADY RED BEFORE IT. `.yah/schema/machine.toml.schema.json` is generated from `cloud::config` by `cargo run -p xtask -- emit-schemas`, and MachineConfig's DOC COMMENT is what the generator emits as its `description` — which means (a) my new `mesh_tags` doc changes it, and (b) so does this very handoff, because R860-T5's @yah: annotation block lives inside MachineConfig's doc at config.rs:201. That file was ALSO already dirty in the working tree at anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2, before I touched anything — `scripts/check-schema-drift.sh` compares the regenerated tree against git, so it is red for any uncommitted schema edit regardless of author. Regenerate with `cargo run -p xtask -- emit-schemas` (or `scripts/check-schema-drift.sh --update`) when the root target dir is not contended; the pre-commit hook no longer does it (disabled 2026-08-15, see CLAUDE.md).")
305/// @yah:verify("SCHEMA REGENERATED IN THIS SESSION, so the drift gate is not left for the next reader: `cargo run --quiet -p xtask -- emit-schemas` exit 0, run from the repo root after the handoff was written (so it captures the annotation text too). Two files moved. `.yah/schema/machine.toml.schema.json`: MachineConfig's `description` grows by this ticket's annotation block, plus a genuinely new `mesh_tags.description` from the doc comment I added. `.yah/schema/workload.toml.schema.json`: +104 lines that are NOT mine — the `Locality` / `Requirement` / `Supply` / `WorkloadSpec.requires` types R860-T1 landed had never been emitted, so the sibling ticket's schema drift was still outstanding and my regen swept it in. Derived artifacts are not ownable (shared-tree doctrine), so this is deliberate rather than accidental; @Ashguard, whoever picks up R860-T1's review should know the schema now describes `requires`.")
306/// @yah:verify("FINAL RE-RUN AFTER THE HANDOFF ANNOTATION WAS WRITTEN INTO config.rs (the board write edits MachineConfig's doc block, so the file changed under the earlier green): `cargo test -p yah-cloud --lib` = 1093 passed / 0 failed / 4 ignored, exit 0. Unchanged. Note for anyone reading the camp build rail's skew warnings on this session: the one `SUSPECT RESULT` it emitted names `oss/yubaba/crates/cloud/src/config.rs` as modified mid-run, and that modification was MY OWN board_handoff annotation write, not a peer — the two authoritative runs (full lib test, and both cargo checks) each came back `Input closure unchanged across the whole run: no skew`.")
307/// @yah:verify("All builds were run with `CARGO_TARGET_DIR=/tmp/r860t5-target` rather than the shared oss/yubaba/target, following R860-T4's recorded gotcha — a peer (session:83093d9d) held the shared target lock for the entire session (20+ minutes of `cargo check -p yubaba --lib`). Costs one cold dep build, then every subsequent run is seconds. Worth reaching for immediately when the queue message says you are behind someone.")
308/// @yah:handoff("LEADER RE-VERIFIED (session:69b18855, independent of the courier's self-report). `cargo test -p yah-cloud --lib` from oss/yubaba: 1093 passed / 0 failed / 4 ignored, exit 0, against the 1090/0/4 baseline this relay's own T4 established — +3 = exactly its new tests. Axis confirmed by content: `NATIVE_EXEC_MESH_TAG = \"cap:native-exec\"` at config.rs:2304, appended to the derived `RequiredSpec.mesh_tags` at :2112-2114 when any `placement_group` member returns true from `WorkloadSpec::wants_native_exec()`. No new `RequiredSpec` field, no signature change, no wire change — it rides the existing AND-ed superset check, so a refusal now reads `required.mesh_tags=[...,cap:native-exec]`.")
309/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
310/// @yah:verify("MACHINE DECLARATIONS AUDITED FOR PROVENANCE, because a wrong capability declaration is worse than an absent one. Both are traceable to measurements ALREADY RECORDED IN-REPO, not inferred: us-west-001 from the `ps` reading at us-west-001.toml:21 (pid 517125, ppid 515908 = `/usr/local/bin/kamaji --native-exec-dir /var/lib/yah/kamaji/native`, cgroup `0::/yubaba.slice/kamaji.service/native`, 2026-09-05); us-west-003 from the actual ExecStart read over ssh 2026-09-01 at us-west-003.toml:141. us-west-015 was deliberately left WITHOUT the tag and carries enable instructions at :207-217 — unknown fails closed, which is the correct direction. No node was guessed at and nothing was probed live.")
311/// @yah:handoff("THIS TICKET MODELS THE EXACT DRIFT THAT CAUSED THE 25-HOUR MESH OUTAGE, which is worth stating because it turns an abstract W338 bullet into a measured one. us-west-001.toml:8 records the root-cause chain: on 2026-09-03T06:03:03Z leadership moved to us-south-001, which tried to deploy headscale and kamaji refused — \\\"workload requests native host execution (yah.exec=native) but no native backend is available (native backend not configured — start kamaji with --native-exec-dir)\\\" — then the systemd fallback failed too, both at WARN, and the mesh had no coordination server for 25 hours. us-west-001.toml:10 names it explicitly as \\\"a silent per-node capability drift that placement does not model\\\". After this ticket, placement models it: a group needing native exec can no longer be admitted onto a node that has not declared `cap:native-exec`.")
312/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
313/// @yah:verify("RE-VERIFIED AT HEAD 00ee20d1 (session:aa5e882d, 2026-09-05). `NATIVE_EXEC_MESH_TAG` present in oss/yubaba/crates/cloud/src/config.rs (declared above `node_selector_mesh_tags`, appended to the derived `RequiredSpec.mesh_tags` when any `placement_group` member wants native exec). `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` = 1141 passed / 0 failed / 4 ignored, exit 0.")
314/// @yah:cleanup("cap:microvm remains the identical unmodelled axis, one line from done in the same `if` in `admission_spec`. Deliberately NOT taken: no node in the fleet can host a microvm until R605-F14 lands a guest kernel + rootfs, so declaring the tag today would assert a false fact. Do it when R605-F14 lands, not before.")
315#[derive(Debug, Clone, Serialize, Deserialize)]
316#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
317pub struct MachineConfig {
318 pub name: String,
319 pub provider: String,
320 /// Who the hardware actually comes from (`"ovh"`, `"vultr"`, `"on-prem"`).
321 ///
322 /// Deliberately *not* [`provider`](Self::provider), which selects the
323 /// auto-provision driver: a box we rented by hand and brought up over SSH
324 /// is `provider = "static"` for its whole life, and writing the vendor
325 /// there instead would flip it driver-backed and make
326 /// [`validate`](Self::validate) demand `location` + `server_type` it has no
327 /// answer for. The two axes genuinely differ — vendor is who bills you,
328 /// `provider` is who yah can call an API against.
329 ///
330 /// Worth recording because vendor-scoped policy is invisible in every other
331 /// field and decides real work: outbound port 25, rDNS/PTR control, IP
332 /// reputation, egress billing. It survived only in TOML prose until now,
333 /// which made it ungreppable at exactly the moment you need it.
334 #[serde(default, skip_serializing_if = "Option::is_none")]
335 pub vendor: Option<String>,
336 /// Human label for the box (`"gamer"`, `"the GEEKOM"`). Free-form and never
337 /// matched on — [`name`](Self::name) stays the identity everywhere. This is
338 /// only so operators and agents can say which box they mean out loud.
339 #[serde(default, skip_serializing_if = "Option::is_none")]
340 pub nickname: Option<String>,
341 /// Provider DC code (e.g. Hetzner `"hil"`). **Provisioning-only**: required
342 /// iff the provider has an auto-provision driver ([`provider_has_machine_driver`]);
343 /// a BYO `static` node we brought up over SSH has no such code. Optional at
344 /// load time so static machine.tomls omit it; [`MachineConfig::validate`]
345 /// enforces presence at the right moment for driver-backed providers.
346 #[serde(default, skip_serializing_if = "Option::is_none")]
347 pub location: Option<String>,
348 /// Provider SKU/size (e.g. Hetzner `"ccx13"`). Provisioning-only, same
349 /// optionality contract as [`location`](Self::location).
350 #[serde(default, skip_serializing_if = "Option::is_none")]
351 pub server_type: Option<String>,
352 /// **Deprecated (R330-F16).** A machine should describe *itself* (region,
353 /// zone, provider, mesh_tags); *which* mirrors run on it is derived by the
354 /// reconciler from each mirror's `required` placement spec, not declared
355 /// here. Now optional + omitted-when-empty so new machine.tomls leave it
356 /// out. The legacy `resolve_mirror_machine` topology fallback still reads
357 /// it until yubaba's reverse-index supersedes the topology.toml path; once
358 /// that lands, this field and its readers are removed wholesale.
359 #[serde(default, skip_serializing_if = "Vec::is_empty")]
360 pub hosts_mirrors: Vec<String>,
361 /// Positive placement facts about this node, matched as a **superset**:
362 /// a workload is admitted only where every tag it requires is present, so
363 /// adding a tag can only ever make a machine match more, never fewer.
364 ///
365 /// Four namespaces are live, and they are not interchangeable:
366 /// - `tag:<role>` — a role the operator assigns (`tag:build-worker`,
367 /// `tag:qed`, `tag:cloud-runner`, `tag:mac-builder`);
368 /// - `arch:<x86|arm>` / `os:<linux|darwin>` — facts about the silicon and
369 /// userland, emitted as *requirements* by
370 /// [`qed::platform::build_worker_mesh_tags`];
371 /// - `cap:<capability>` — something the node's daemons were configured to
372 /// be able to do. Today just [`NATIVE_EXEC_MESH_TAG`] (R860-T5);
373 /// - `tier:` is **retired** for architecture (R763) and reserved for the
374 /// environment axis — [`crate::validate::check_retired_arch_tags`]
375 /// flags a machine still carrying `tier:<arch>`.
376 ///
377 /// Nothing validates the prefix, which is why the lint above exists: a tag
378 /// nobody requires is silently inert, and a *stale* one silently stops
379 /// matching and reports "no node" rather than "wrong tag".
380 pub mesh_tags: Vec<String>,
381 /// Canonical geo region label (latency axis), e.g. `"us-west"`. F16's three
382 /// topology axes are orthogonal: `region` = geo (latency), `zone` = failure
383 /// domain within a region (HA), `provider` = network/cost. `region` is
384 /// distinct from `location` (the provider's DC code, e.g. Hetzner `"hil"`):
385 /// `location` is provider-scoped, `region` is our provider-neutral label.
386 /// Optional for backward-compat; a machine without it never satisfies a
387 /// `required.regions` constraint.
388 #[serde(default, skip_serializing_if = "Option::is_none")]
389 pub region: Option<String>,
390 /// Failure-domain label within a region (HA axis), e.g. `"hil"`. For
391 /// single-DC Hetzner this typically mirrors `location`. F16 placement
392 /// matches `required.zones` against this. Optional for backward-compat.
393 #[serde(default, skip_serializing_if = "Option::is_none")]
394 pub zone: Option<String>,
395 /// Declared CPU architecture (`"x86_64"` / `"aarch64"`). A machine has
396 /// exactly one — it's a first-class property of the box, not a reach
397 /// detail and not a mesh tag. Drives the yubaba release triple. Optional
398 /// only because there's no provider API to probe it (static nodes declare
399 /// it; a driver-backed provider may leave it unset until known).
400 #[serde(default, skip_serializing_if = "Option::is_none")]
401 pub arch: Option<String>,
402 pub bucket: Option<BucketSpec>,
403 /// **Legacy location, superseded by `[registration].hostkey_fingerprint`**
404 /// (R707-T1). Still deserialized so machine TOMLs written before the split
405 /// keep parsing; never *read* directly — go through
406 /// [`MachineConfig::hostkey_fingerprint`], which prefers the registration
407 /// block. [`MachineConfig::normalize`] folds this into `registration`, and
408 /// [`MachineConfig::save`] normalizes before writing, so a load→save cycle
409 /// migrates the file rather than dropping the value.
410 #[serde(
411 rename = "hostkey_fingerprint",
412 default,
413 skip_serializing_if = "Option::is_none"
414 )]
415 pub legacy_hostkey_fingerprint: Option<String>,
416 /// Provider-side SSH-key IDs (Hetzner: from `GET /v1/ssh_keys`)
417 /// authorized for `root` at create time. Defaults to empty for
418 /// backwards-compat with existing machine declarations; an empty
419 /// list yields a Hetzner-emailed random root password (which the
420 /// driver currently discards). Populate this when you want pre-mesh
421 /// SSH access for bootstrap deploys or recovery.
422 #[serde(default, skip_serializing_if = "Vec::is_empty")]
423 pub ssh_keys: Vec<u64>,
424 /// Cloudflare Tunnel ID this machine joins (e.g. `abc123.cfargotunnel.com`).
425 /// `None` → no tunnel (mesh-only node, no public ingress).
426 /// When set, `yah cloud machine provision` reads `cloudflare-tunnel-token`
427 /// from the keys vault and injects the cloudflared install block into
428 /// cloud-init so the new machine connects to CF edge on first boot.
429 #[serde(default, skip_serializing_if = "Option::is_none")]
430 pub cloudflared: Option<String>,
431 /// Provider-issued floating/reserved IP that follows **public-ingress
432 /// ownership** onto this box — R859-F2 (W267 §Tier 1).
433 ///
434 /// The value is the provider's own identifier, opaque here and interpreted
435 /// only by the matching adapter: a Hetzner numeric floating-IP id as a
436 /// string, an OVH Additional-IP address (`"51.81.85.200"`), a Vultr
437 /// reserved-IP UUID. Same "the adapter is the boundary" convention
438 /// [`crate::envoy::floating_ip::FloatingIpAssignInput::ip_id`] documents.
439 ///
440 /// # Why it lives on the machine
441 ///
442 /// [`crate::envoy::floating_ip`] shipped the `floating_ip.*` verbs and
443 /// three provider adapters with no config anywhere saying *which* floating
444 /// IP is "the" ingress IP — the gap R594-F5 recorded and deliberately left.
445 /// This is that field, and it sits beside [`cloudflared`](Self::cloudflared)
446 /// on purpose: that is already the per-node "how the world reaches this
447 /// box" handle, and a floating IP is the sovereign-tier answer to the same
448 /// question. `[[ingress]]`'s
449 /// [`tunnel_id`](crate::config::IngressEdge::tunnel_id) is the *service*
450 /// side of ingress identity — which cohort a given service fronts through —
451 /// and a floating IP is not per-service: one IP moves between boxes, so it
452 /// cannot be partitioned by slot or hostname.
453 ///
454 /// # Absent means "no floating-IP path", never an error
455 ///
456 /// Most machines have none, and that is the normal case: mesh-only nodes,
457 /// boxes behind a Cloudflare tunnel, and every provider without a
458 /// floating-IP adapter. The effector skips such a machine cleanly rather
459 /// than refusing — see
460 /// [`plan_ingress_owner_effect`](crate::provider::floating_ip::plan_ingress_owner_effect).
461 ///
462 /// # The cohort has to agree
463 ///
464 /// Every machine that can hold the same ingress IP must declare the *same*
465 /// id: the IP is one resource that moves, so two ids inside one
466 /// [`sovereign_group`](Self::sovereign_group) means an ownership flip
467 /// silently reassigns a *different* IP than the one currently serving
468 /// traffic. `yah cloud validate` refuses that
469 /// ([`crate::validate::check_ingress_floating_ip`]) rather than leaving it
470 /// to be discovered during a failover.
471 #[serde(default, skip_serializing_if = "Option::is_none")]
472 pub ingress_floating_ip: Option<String>,
473 /// When `true`, this machine hosts operator-bridge workloads (Tailscale
474 /// operator access to mesh-internal services). `yah cloud machine provision`
475 /// will install tailscaled and run `tailscale up` during cloud-init via the
476 /// `{{OPERATOR_BRIDGE_BLOCK}}` placeholder. Defaults to `false` for
477 /// backward-compat with existing machine declarations.
478 #[serde(default)]
479 pub hosts_operator_bridge: bool,
480 /// BYO `static`-node reach descriptor. Static nodes have no provider API to
481 /// probe, so how the camp reaches them (SSH user@host + the yubaba URL,
482 /// which is loopback until the WireGuard mesh lands) is *declared* here.
483 /// `None` for driver-backed providers (Hetzner/Vultr), whose address is
484 /// resolved from the provider API / mesh at provision time.
485 #[serde(default, skip_serializing_if = "Option::is_none")]
486 pub connect: Option<ConnectSpec>,
487 /// Static node capacity (R572-F3). Declares the node's total hardware
488 /// budget; F5's scheduler subtracts committed workload requests from this
489 /// to check whether a new workload fits. Absent means unconstrained.
490 #[serde(default, skip_serializing_if = "Option::is_none")]
491 pub allocatable: Option<NodeAllocatable>,
492 /// Placement taint keys (R572-F3). A repelling key blocks placement by
493 /// default, and a placement opts back in by naming that exact key in
494 /// [`RequiredSpec::tolerates`] (R876-B7).
495 ///
496 /// The `unless` is real now. It was not between R742-T4 and R876-B7: the
497 /// spec side declared which archetypes it *was* rather than which taints it
498 /// tolerated, that field was `#[serde(skip)]`, and so every placement
499 /// declared as `required = {...}` in a mirror TOML read this list as empty
500 /// and could not be drained at all. See [`RequiredSpec::tolerates`].
501 ///
502 /// A key in this list influences placement in exactly one of two ways, and
503 /// [`taint_effect`] is the authority on which:
504 ///
505 /// - **repulsion** — `"no-server"` / `"no-appliance"` / `"no-job"` reject
506 /// any placement that does not tolerate them. The archetype in the key is
507 /// now vocabulary rather than a filter: `matches` does not compare it
508 /// against the workload's class, it checks the toleration list, and
509 /// [`admission_spec`] is what turns a workload's class into the
510 /// tolerations that reproduce the old archetype-scoped behaviour;
511 /// - **affinity** — a key in [`AFFINITY_TAINT_KEYS`] (today just
512 /// `"public-ip"`) that a workload names in
513 /// `yah.placement.requires-taint`, which then *requires* this node.
514 ///
515 /// Anything else is **inert**: it parses, it round-trips, and no scheduler
516 /// decision can ever read it. `yah cloud validate` rejects such keys
517 /// (`validate::check_inert_taints`) rather than letting them sit looking
518 /// load-bearing — which is how `no-voter` spent months asserting a
519 /// falsehood on three nodes. Facts about a node that are not placement
520 /// inputs belong in [`mesh_tags`](Self::mesh_tags) or a comment.
521 #[serde(default, skip_serializing_if = "Vec::is_empty")]
522 pub taints: Vec<String>,
523 /// Which consensus group this node belongs to — W305/R742-F1. `None` means
524 /// standalone: in no group at all, which is us-west-002 and us-west-015.
525 ///
526 /// Membership is not by itself quorum eligibility; that is
527 /// [`sovereign_role`](Self::sovereign_role), added by R605-F12 because
528 /// us-west-003 is in prod's blast radius *and* must never vote in it.
529 ///
530 /// **Not a placement input.** It is deliberately absent from
531 /// [`RequiredSpec::matches`], and adding it there would be a category
532 /// error: a sovereign group is a *blast radius*, not a filter. Nothing
533 /// about "which quorum does this box vote in" should decide where a
534 /// workload runs — that is what made the fleet express three unrelated
535 /// properties through one taint list and get all three wrong (W305).
536 ///
537 /// What it *is* for is refusal. [`judge_join`] answers "may this node join
538 /// that node's cluster", and the answer is no unless both declare the same
539 /// group. Before this field the only guard was a comment in three machine
540 /// TOMLs saying "never run a raft join against this box from a shell
541 /// pointed at prod" — habit, with no mechanism behind it, which is the
542 /// same class of guard W257 §8 admitted to.
543 ///
544 /// # Why `sovereign_group` and not `raft_group`
545 ///
546 /// Raft is today's mechanism (operator, 2026-08-10). A field named for the
547 /// mechanism goes stale the day the mechanism is swapped, and every
548 /// consumer that reads it inherits the lie. `sovereign` names what the
549 /// group *has* — its own authority, its own upgrade cadence, its own
550 /// destruction — which stays true under any consensus protocol.
551 ///
552 /// Note the word already appears in this tree as prose (W267's title, the
553 /// `IngressProvider::Passway` doc comment's "sovereign edge"). That is an
554 /// adjective meaning "self-hosted, not SaaS"; this is the first time it
555 /// carries structure.
556 #[serde(default, skip_serializing_if = "Option::is_none")]
557 pub sovereign_group: Option<String>,
558 /// Whether this node may hold a seat in its group's quorum — R605-F12.
559 /// Meaningless without [`sovereign_group`](Self::sovereign_group): a
560 /// standalone box has no quorum to be eligible for.
561 ///
562 /// **`None` is "not written", not a third role.** Read it through
563 /// [`sovereign_membership`](Self::sovereign_membership), which resolves the
564 /// absence to [`SovereignRole::Voter`] — what declaring a group has always
565 /// meant, so the six nodes stamped before this field keep their seats
566 /// without an edit. The distinction is kept only so
567 /// [`crate::validate::check_unroled_sovereign_members`] can tell an
568 /// operator who *chose* voter from one who never considered the question;
569 /// no join decision reads the `Option` directly.
570 ///
571 /// # Why this is not a taint
572 ///
573 /// It was, once: `no-voter` sat in [`taints`](Self::taints) on three nodes
574 /// for months, read by nothing, and R742-T4 removed it because the taint
575 /// list is a *placement* vocabulary and this is not a placement input (see
576 /// [`taint_effect`]). Nor is it a second group label. It is a modifier on
577 /// the membership this node already declares, which is why it lives beside
578 /// the group and is judged with it in one predicate,
579 /// [`workload_spec::sovereign::join_permitted`].
580 #[serde(default, skip_serializing_if = "Option::is_none")]
581 pub sovereign_role: Option<SovereignRole>,
582 /// `[registration]` — the observed half (R707-T1). Empty until the box has
583 /// been attached / mesh-joined. See [`MachineRegistration`].
584 #[serde(default, skip_serializing_if = "MachineRegistration::is_empty")]
585 pub registration: MachineRegistration,
586}
587
588/// True iff `provider` has an auto-provision driver (create/destroy via API).
589/// Driver-backed providers require `location` + `server_type`; BYO `static`
590/// nodes (brought up over SSH) do not. The cloud-vs-vps distinction the fleet
591/// cares about lives here — at the provider-capability layer — not as a
592/// separate machine type (W242 BYO Phase-0 decision).
593pub fn provider_has_machine_driver(provider: &str) -> bool {
594 matches!(provider, "hetzner" | "vultr" | "digitalocean")
595}
596
597/// Taint keys a workload may name in `yah.placement.requires-taint` to
598/// *require* a node (W305/R742-T4 affinity vocabulary).
599///
600/// This is a closed list on purpose. `WorkloadSpec::requires_taint` returns
601/// free text, but every producer in the tree is code — `passway_ingress.rs`
602/// and `cloudflared_ingress.rs`, both emitting
603/// [`workload_spec::PUBLIC_IP_TAINT`] — and no on-disk `workload.toml` sets the
604/// annotation at all. So the set of keys a node can usefully carry for
605/// affinity is knowable at compile time, which is what lets
606/// [`taint_effect`] call anything outside it inert instead of guessing.
607///
608/// **Adding an affinity key means adding it here**, in the same change that
609/// teaches a workload to require it. That coupling is the point: it makes the
610/// node side and the workload side impossible to land apart.
611pub const AFFINITY_TAINT_KEYS: &[&str] = &[workload_spec::PUBLIC_IP_TAINT];
612
613/// How a key in [`MachineConfig::taints`] can affect placement.
614///
615/// W305 finding 1: before R742-T4 nothing asked this question, so a key that
616/// no scheduler path could read — `"qa"`, `"no-voter"` — parsed, validated,
617/// and quietly did nothing. Both of the findings that cost real fleet state
618/// were invisible for exactly that reason.
619#[derive(Debug, Clone, Copy, PartialEq, Eq)]
620pub enum TaintEffect {
621 /// `"no-<archetype>"`: rejects placement outright unless the constraint
622 /// names this key in [`RequiredSpec::tolerates`]. Read by
623 /// [`RequiredSpec::matches`], which walks `machine.taints` and classifies
624 /// each key through [`taint_effect`] (R876-B7).
625 Repels(LifecycleArchetype),
626 /// A key in [`AFFINITY_TAINT_KEYS`]: a workload naming it in
627 /// `yah.placement.requires-taint` is restricted to nodes carrying it.
628 Attracts,
629 /// Neither. No placement decision can read this key.
630 Inert,
631}
632
633/// Classify one node taint key. See [`TaintEffect`].
634///
635/// The repulsion half is derived from [`LifecycleArchetype::ALL`] rather than
636/// a literal list, so a fourth archetype makes `no-<its key>` live without an
637/// edit here.
638pub fn taint_effect(key: &str) -> TaintEffect {
639 if let Some(arch) = LifecycleArchetype::ALL
640 .into_iter()
641 .find(|a| key == format!("no-{}", a.taint_key()))
642 {
643 return TaintEffect::Repels(arch);
644 }
645 if AFFINITY_TAINT_KEYS.contains(&key) {
646 return TaintEffect::Attracts;
647 }
648 TaintEffect::Inert
649}
650
651/// Every key the scheduler *can* act on, sorted — for error messages that
652/// tell the operator what the legal vocabulary actually is instead of only
653/// what was wrong.
654pub fn live_taint_keys() -> Vec<String> {
655 let mut keys: Vec<String> = LifecycleArchetype::ALL
656 .into_iter()
657 .map(|a| format!("no-{}", a.taint_key()))
658 .chain(AFFINITY_TAINT_KEYS.iter().map(|k| (*k).to_string()))
659 .collect();
660 keys.sort();
661 keys
662}
663
664/// What [`judge_join`] decided about one proposed cluster join.
665///
666/// Shaped like yubaba's `PromotionVerdict` / `GeographyVerdict` and for the
667/// same reason: the rule stays unit-testable without a live cluster, and a
668/// refusal carries its reason from the place that knows it.
669#[derive(Debug, Clone, PartialEq, Eq)]
670pub enum JoinVerdict {
671 /// Both nodes declare the same sovereign group and both are voters. The
672 /// join is within one blast radius and grows a quorum both sides are
673 /// eligible for.
674 Permit,
675 /// The join is refused. Carries an operator-readable reason naming both
676 /// declared values and the file to edit — a refusal that only says
677 /// "invalid" gets worked around rather than fixed.
678 Refuse(String),
679}
680
681/// May `joiner` join the cluster `target` belongs to? — W305/R742-F1.
682///
683/// **A join is permitted iff both nodes declare the same non-`None`
684/// [`sovereign_group`](MachineConfig::sovereign_group) and both are
685/// [`SovereignRole::Voter`].** One rule, no special cases, and it makes the
686/// declaration mandatory before any quorum grows.
687///
688/// The case this exists for is two *different* declared groups: joining a dev
689/// Pi into prod is refused rather than trusted, where today the only guard is
690/// a comment saying not to do it. But an undeclared node is refused too, and
691/// that is the deliberate half — `None` means "in no group", not "unknown", so
692/// growing prod with an unstamped box is exactly as much a cross-group join as
693/// the dev case is. Failing open there would leave the operator believing a
694/// guarantee that was never evaluated, which is the reasoning
695/// `QuorumGeography::judge` already applies to untagged voters.
696///
697/// No legitimate flow pays for that strictness: prod and dev are both stamped,
698/// and us-west-002/015 are deliberately in no group at all. Adding a real
699/// member means declaring it first, which is the point.
700///
701/// # The non-voting refusal (R605-F12)
702///
703/// Same group and still refused, when either side declares
704/// [`SovereignRole::NonVoter`]. This is the case a group label alone could not
705/// express. us-west-003 is a residential-uplink build box the operator counts
706/// as part of prod — same secrets, same upgrade cadence, same destruction — and
707/// which must never hold a prod raft seat, because a home-internet partition
708/// should not be able to stall the quorum. Until R605-F12 the only thing
709/// refusing it was its *absent* stamp, so recording the operator's real intent
710/// (`sovereign_group = "prod"`) would have removed the guard. Now the intent
711/// and the guard are the same two lines.
712///
713/// Note what this is not: the refusal here is about *voting*, and it says
714/// nothing about the mesh. One mesh spans the whole fleet regardless of group
715/// or role (operator, 2026-08-19); a non-voter is reachable, schedulable and
716/// rollable like any other node.
717///
718/// This is the **camp-side** rendering of the rule. The predicate itself lives
719/// in [`workload_spec::sovereign::join_permitted`] because yubaba's
720/// `POST /raft/add-learner` gate asks the same question and cannot see this
721/// crate (there is deliberately no yubaba → cloud edge). Only the prose is
722/// duplicated, and it has to be: a refusal here names
723/// `.yah/infra/machines/<name>.toml`, while the node-side one has no machine
724/// name in hand and must also name `yubaba serve --sovereign-group`.
725///
726/// The node-side gate is *narrower* on purpose, and the difference is worth
727/// knowing when reading either: a daemon started without `--sovereign-group`
728/// has declared nothing rather than declared standalone, so yubaba resolves
729/// that unknown before it judges, and its gate is in force only once the
730/// cluster being joined declares a group. See `yubaba::sovereign_group`.
731pub fn judge_join(joiner: &MachineConfig, target: &MachineConfig) -> JoinVerdict {
732 let stamp_hint = |m: &MachineConfig| {
733 format!(
734 "declare `sovereign_group = \"<group>\"` in .yah/infra/machines/{}.toml",
735 m.name
736 )
737 };
738 let role_hint = |m: &MachineConfig| {
739 format!(
740 "set `sovereign_role = \"voter\"` in .yah/infra/machines/{}.toml",
741 m.name
742 )
743 };
744 if workload_spec::sovereign::join_permitted(
745 joiner.sovereign_membership(),
746 target.sovereign_membership(),
747 ) {
748 return JoinVerdict::Permit;
749 }
750 let (j, t) = (
751 joiner.sovereign_group.as_deref(),
752 target.sovereign_group.as_deref(),
753 );
754 // Everything below is a refusal; the only permitted shape returned above.
755 //
756 // R605-F12: when both sides name the SAME group, the role is the only thing
757 // left that can have refused, and it gets its own message. Falling through
758 // to the arms below would print "cross-group join refused: 'us-west-003' is
759 // in "prod" and 'us-west-001' is in "prod"" — a message that reads as a bug
760 // in the check rather than a decision about the fleet.
761 //
762 // Deliberately not hoisted above the group comparison. A non-voting joiner
763 // whose target is standalone is refused for *both* reasons, and naming the
764 // role there would send the operator to fix a field that would not have
765 // made the join legal anyway.
766 if let (Some(a), Some(b)) = (j, t) {
767 if a == b {
768 for (m, side, other) in [
769 (joiner, "the joiner", &target.name),
770 (target, "the target", &joiner.name),
771 ] {
772 if m.sovereign_membership().role.is_voter() {
773 continue;
774 }
775 return JoinVerdict::Refuse(format!(
776 "join refused: {side} '{}' is a NON-VOTING member of sovereign group {a:?}, \
777 the same group as '{other}'. It is inside that blast radius — same secrets, \
778 same upgrade cadence, same destruction — but declares itself ineligible for \
779 the quorum, so this is refused by declaration rather than by omission. If it \
780 should genuinely vote, {}; if it should not, this refusal is the field doing \
781 its job and the join is the thing to reconsider.",
782 m.name,
783 role_hint(m),
784 ));
785 }
786 }
787 }
788 match (j, t) {
789 (Some(a), Some(b)) => JoinVerdict::Refuse(format!(
790 "cross-group join refused: '{}' is in sovereign group {a:?} and '{}' is in {b:?}. \
791 These are separate blast radii — separate quorums, separate upgrade cadences, \
792 separately destroyable — and merging them is not something a join can undo. If \
793 the move is genuinely intended, restamp '{}' to {b:?} first and treat it as \
794 leaving its old group.",
795 joiner.name,
796 target.name,
797 joiner.name,
798 )),
799 (None, Some(b)) => JoinVerdict::Refuse(format!(
800 "join refused: '{}' declares no sovereign_group, so it is standalone — in no \
801 group — while '{}' is in {b:?}. That is a cross-group join, not an unchecked \
802 one. To make '{}' a member of {b:?}, {}.",
803 joiner.name,
804 target.name,
805 joiner.name,
806 stamp_hint(joiner),
807 )),
808 (Some(a), None) => JoinVerdict::Refuse(format!(
809 "join refused: '{}' is in sovereign group {a:?} but '{}' declares none, so the \
810 target is standalone and has no group to join. Either {}, or found the group on \
811 '{}' rather than growing it.",
812 joiner.name,
813 target.name,
814 stamp_hint(target),
815 joiner.name,
816 )),
817 (None, None) => JoinVerdict::Refuse(format!(
818 "join refused: neither '{}' nor '{}' declares a sovereign_group, so this join \
819 would form a group nobody declared and nothing could later reason about. Name \
820 the group on both boxes first: {}, and the same for '{}'.",
821 joiner.name,
822 target.name,
823 stamp_hint(joiner),
824 target.name,
825 )),
826 }
827}
828
829impl MachineConfig {
830 /// This node's declared place in a sovereign group, as the shared join rule
831 /// wants it — R605-F12.
832 ///
833 /// The one place `sovereign_role`'s `None` is resolved. Absence means
834 /// [`SovereignRole::Voter`], which is what declaring a group meant before
835 /// the role existed; resolving it here rather than at each call site is what
836 /// keeps the camp-side and node-side gates from disagreeing about a node
837 /// that never wrote the field.
838 pub fn sovereign_membership(&self) -> Membership<'_> {
839 Membership {
840 group: self.sovereign_group.as_deref(),
841 role: self.sovereign_role.unwrap_or_default(),
842 }
843 }
844
845 /// Provider DC code, or `""` when omitted (static nodes). Most readers want
846 /// a `&str`; the driver-backed provision/status paths still go through
847 /// [`validate`](Self::validate) which guarantees presence for those.
848 pub fn location(&self) -> &str {
849 self.location.as_deref().unwrap_or("")
850 }
851
852 /// Provider SKU, or `""` when omitted (static nodes).
853 pub fn server_type(&self) -> &str {
854 self.server_type.as_deref().unwrap_or("")
855 }
856
857 /// Enforce the provisioning-only-field contract: a machine whose provider
858 /// has an auto-provision driver MUST declare `location` + `server_type`
859 /// (the driver can't create a server without them). Static nodes may omit
860 /// both. Call this before any provision/diff that assumes a driver.
861 pub fn validate(&self) -> Result<()> {
862 if provider_has_machine_driver(&self.provider) {
863 if self.location.is_none() {
864 anyhow::bail!(
865 "machine '{}' (provider '{}') has an auto-provision driver but no `location`",
866 self.name,
867 self.provider
868 );
869 }
870 if self.server_type.is_none() {
871 anyhow::bail!(
872 "machine '{}' (provider '{}') has an auto-provision driver but no `server_type`",
873 self.name,
874 self.provider
875 );
876 }
877 }
878 Ok(())
879 }
880
881 /// Declared taints that no placement decision can read (W305/R742-T4).
882 ///
883 /// Deliberately **not** folded into [`validate`](Self::validate): that
884 /// guard runs on the provision/diff hot path and answers a different
885 /// question (can the driver create this server). An inert taint is a lint
886 /// — it never breaks an operation in flight, it just means the file is
887 /// asserting something the scheduler will not honour. `yah cloud validate`
888 /// is where the operator asks for that judgement; see
889 /// [`crate::validate::check_inert_taints`].
890 pub fn inert_taints(&self) -> Vec<&str> {
891 self.taints
892 .iter()
893 .filter(|t| taint_effect(t) == TaintEffect::Inert)
894 .map(String::as_str)
895 .collect()
896 }
897
898 /// Yubaba's TOFU'd hostkey fingerprint, from `[registration]` and falling
899 /// back to the pre-R707-T1 top-level field. **The only read path** — a
900 /// caller that reaches for `legacy_hostkey_fingerprint` directly sees
901 /// `None` on every migrated machine.
902 pub fn hostkey_fingerprint(&self) -> Option<&str> {
903 self.registration
904 .hostkey_fingerprint
905 .as_deref()
906 .or(self.legacy_hostkey_fingerprint.as_deref())
907 }
908
909 /// Record (or clear) the observed hostkey fingerprint. Writes
910 /// `[registration]` and drops any pre-R707-T1 top-level value, so the two
911 /// locations can never disagree after a writeback.
912 pub fn set_hostkey_fingerprint(&mut self, fingerprint: Option<String>) {
913 self.registration.hostkey_fingerprint = fingerprint;
914 self.legacy_hostkey_fingerprint = None;
915 }
916
917 /// Mesh (tailnet) IPv4 for this node, or `None` pre-mesh.
918 ///
919 /// Prefers `[registration].mesh_ipv4`; falls back to the host of a legacy
920 /// `[connect].yubaba` URL when that host is in the `100.64.0.0/10` CGNAT
921 /// range the mesh uses. A loopback placeholder (`http://127.0.0.1:7443`,
922 /// meaning "pre-mesh, reachable only through an SSH tunnel") is *not* a
923 /// mesh address and yields `None`.
924 pub fn mesh_ipv4(&self) -> Option<&str> {
925 if let Some(ip) = self.registration.mesh_ipv4.as_deref() {
926 return Some(ip);
927 }
928 let url = self.connect.as_ref()?.yubaba.as_deref()?;
929 mesh_ipv4_from_url(url)
930 }
931
932 /// Base URL for this node's yubaba, or `None` when no reach resolves.
933 ///
934 /// Thin wrapper over [`reach`](Self::reach) for the many call sites that
935 /// only branch on presence. Prefer `reach` anywhere the operator sees the
936 /// outcome — a `None` here throws away a refusal that names exactly which
937 /// address is missing.
938 pub fn yubaba_url(&self) -> Option<String> {
939 self.reach().ok()
940 }
941
942 /// The **one** address automation dials for this node — mesh-only.
943 ///
944 /// `Err` is a *named refusal*, not an absence: a node with no mesh address
945 /// is unresolvable to every automated path, and R605-T10's whole complaint
946 /// is that this used to surface as a connect timeout against an address the
947 /// caller has no route to.
948 ///
949 /// Resolution order:
950 ///
951 /// 1. A declared `[connect].yubaba` on a **private** host (10/8,
952 /// 172.16/12, 192.168/16) is **not dialed** — see below.
953 /// 2. Any other declared `[connect].yubaba` wins verbatim. That includes
954 /// the pre-mesh loopback placeholder (`http://127.0.0.1:7443`, "I have
955 /// no mesh address; reach me through the SSH tunnel to `ssh`"), which is
956 /// a genuine declaration and stays honoured.
957 /// 3. Otherwise `[registration].mesh_ipv4` composed with
958 /// `[connect].yubaba_port`.
959 ///
960 /// **Why a LAN literal loses (R605-T10, operator 2026-08-19).** The LAN
961 /// address is an emergency break-glass route, never an official one, and
962 /// automation must ALWAYS assume the caller is not on that LAN — this camp
963 /// sits on 192.168.22.0/22 with no route to the fleet's 192.168.10.0/24 at
964 /// all. Writing one into the field every resolver dials does not sit beside
965 /// the mesh route, it *overrides* it: R707-T6 made a declared literal beat
966 /// `mesh_ipv4` outright, so us-west-011 (mesh-joined, healthy) was elected
967 /// for every aarch64 build and then dialed at an address that answers only
968 /// from inside bldg-2506.
969 ///
970 /// **What R707-T6 wanted is preserved elsewhere.** Its forcing case was
971 /// identity, not reach: the dev raft group advertises LAN addrs
972 /// (`192.168.10.11:7443`, verified live off `/raft/status` 2026-08-27), and
973 /// `rollout::yubaba::membership_to_nodes` has to map those back to declared
974 /// machines. That match now runs against [`lan_endpoint`](Self::lan_endpoint),
975 /// which is composed from the break-glass `[connect].address` metadata and
976 /// is never dialed — so the two concerns the old precedence rule fused are
977 /// split, and the literal can stop squatting a dialed field.
978 ///
979 /// The LAN address itself STAYS in the machine TOML. It is useful metadata
980 /// and the manual `ssh` path is entitled to it; it is only disconnected
981 /// from every automated process.
982 pub fn reach(&self) -> Result<String, String> {
983 let Some(connect) = self.connect.as_ref() else {
984 return Err(format!(
985 "machine {:?} declares no [connect] block, so nothing knows how to reach it \
986 \u{2192} declare one, or leave it unprovisioned and out of placement",
987 self.name
988 ));
989 };
990 let mesh = || {
991 self.registration
992 .mesh_ipv4
993 .as_deref()
994 .map(|ip| format!("http://{ip}:{}", connect.yubaba_port()))
995 };
996 if let Some(literal) = &connect.yubaba {
997 let Some(lan) = private_ipv4_from_url(literal) else {
998 return Ok(literal.clone());
999 };
1000 return mesh().ok_or_else(|| {
1001 format!(
1002 "machine {:?} is unresolvable to automation: its only declared yubaba reach \
1003 is the private literal {:?} and it has no [registration].mesh_ipv4\n\
1004 \u{2192} a LAN address is an emergency break-glass route, never an official \
1005 one (R605-T10) — every automated path assumes the caller is NOT on {}/24\n\
1006 \u{2192} mesh-join the box and record `mesh_ipv4` under [registration], then \
1007 delete `[connect].yubaba` so the port composes with it",
1008 self.name,
1009 literal,
1010 lan.rsplit_once('.').map(|(net, _)| net).unwrap_or(lan),
1011 )
1012 });
1013 }
1014 mesh().ok_or_else(|| {
1015 format!(
1016 "machine {:?} has no [registration].mesh_ipv4 and declares no \
1017 [connect].yubaba, so no automated path can reach it\n\
1018 \u{2192} mesh-join the box and record its tailnet address, or taint it out of \
1019 placement — do not point `[connect].yubaba` at a LAN address (R605-T10)",
1020 self.name
1021 )
1022 })
1023 }
1024
1025 /// The LAN `host:port` this node's yubaba answers on, composed from the
1026 /// break-glass `[connect].address` metadata plus the declared port.
1027 ///
1028 /// **Identity only — never dial this.** It exists so a raft membership
1029 /// entry that names a node by its LAN address can be mapped back to the
1030 /// declared machine (`rollout::yubaba::membership_to_nodes`) without that
1031 /// address having to live in a field a resolver reads. `None` when the
1032 /// machine is unprovisioned.
1033 pub fn lan_endpoint(&self) -> Option<String> {
1034 let connect = self.connect.as_ref()?;
1035 Some(format!("{}:{}", connect.address, connect.yubaba_port()))
1036 }
1037
1038 /// Fold the pre-R707-T1 top-level `hostkey_fingerprint` into
1039 /// `[registration]`, and lift a mesh IP out of a legacy `[connect].yubaba`
1040 /// URL. Idempotent; a machine already on the split shape is untouched.
1041 ///
1042 /// [`save`](Self::save) calls this, so writing a machine TOML migrates it
1043 /// rather than round-tripping the old shape back out.
1044 pub fn normalize(&mut self) {
1045 if let Some(fp) = self.legacy_hostkey_fingerprint.take() {
1046 self.registration.hostkey_fingerprint.get_or_insert(fp);
1047 }
1048 if self.registration.mesh_ipv4.is_none() {
1049 if let Some(ip) = self
1050 .connect
1051 .as_ref()
1052 .and_then(|c| c.yubaba.as_deref())
1053 .and_then(mesh_ipv4_from_url)
1054 .map(str::to_string)
1055 {
1056 self.registration.mesh_ipv4 = Some(ip);
1057 // The URL was pure derivation from mesh IP + port; keep only
1058 // the declared half so the two can't drift apart.
1059 if let Some(c) = self.connect.as_mut() {
1060 c.yubaba = None;
1061 }
1062 }
1063 }
1064 }
1065
1066 /// Persist to `<cloud_dir>/machines/<name>.toml`, creating the dir if needed.
1067 ///
1068 /// ⚠ Serializes the struct, so **operator comments in the target file are
1069 /// lost**. Pre-existing behaviour, not introduced here, but it is why
1070 /// registration writeback (`yah cloud machine attach`) goes through
1071 /// [`crate::state::MachineState`] and the comment-preserving path in the
1072 /// CLI rather than calling this on a hand-authored inventory file.
1073 pub fn save(&self, cloud_dir: &Path) -> Result<()> {
1074 let dir = cloud_dir.join("machines");
1075 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
1076 let path = dir.join(format!("{}.toml", self.name));
1077 let mut normalized = self.clone();
1078 normalized.normalize();
1079 let s = toml::to_string_pretty(&normalized)
1080 .with_context(|| format!("serializing machine {}", self.name))?;
1081 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
1082 }
1083}
1084
1085/// Host of an `http://host:port` URL iff it is a mesh (headscale) IPv4 in the
1086/// `100.64.0.0/10` CGNAT range. String-level rather than URL-parsed: the
1087/// inventory format is stable and this crate carries no URL dependency (same
1088/// reasoning as `fleet_metrics::extract_host` and
1089/// `hub::coordinator::is_loopback_url`).
1090fn mesh_ipv4_from_url(url: &str) -> Option<&str> {
1091 let host = ipv4_host_of(url)?;
1092 let ip: std::net::Ipv4Addr = host.parse().ok()?;
1093 let [a, b, ..] = ip.octets();
1094 // 100.64.0.0/10 ⇒ first octet 100, second octet 64..=127.
1095 (a == 100 && (64..=127).contains(&b)).then_some(host)
1096}
1097
1098/// Host of an `http://host:port` URL iff it is an **RFC1918 private** IPv4 —
1099/// `10/8`, `172.16/12`, `192.168/16`. `None` for anything else, loopback and
1100/// the `100.64/10` mesh range included: neither is a LAN literal.
1101///
1102/// The judgement R605-T10 turns on. A private literal is only ever reachable
1103/// from inside one building, so it is metadata about where the box physically
1104/// sits and never an address automation may dial — see
1105/// [`MachineConfig::reach`] and [`crate::validate::check_lan_dial_targets`].
1106pub fn private_ipv4_from_url(url: &str) -> Option<&str> {
1107 let host = ipv4_host_of(url)?;
1108 is_private_ipv4(host).then_some(host)
1109}
1110
1111/// Whether a bare host string is an RFC1918 private IPv4 literal.
1112pub fn is_private_ipv4(host: &str) -> bool {
1113 let Ok(ip) = host.parse::<std::net::Ipv4Addr>() else {
1114 return false;
1115 };
1116 ip.is_private()
1117}
1118
1119/// Bare host of a `[scheme://]host[:port][/path]` string.
1120fn ipv4_host_of(url: &str) -> Option<&str> {
1121 let after_scheme = url.split("://").nth(1).unwrap_or(url);
1122 after_scheme.split(['/', ':']).next()
1123}
1124
1125/// Declared **reach** for a BYO `static` node (no provider API). Lives under
1126/// `[connect]` in the machine TOML.
1127///
1128/// Reach only — how the camp gets to the box. *Permission* is a separate axis
1129/// that belongs to cheers' scopes (W295 §"Deliberately deferred"); the two
1130/// collapse in practice today (mesh membership grants everything) and the data
1131/// model must not fuse them, so do not add an authorization field here.
1132///
1133/// `address`, `ssh` and `identity_file` stay whole, literal, operator-authored
1134/// strings even though their values often *look* derived. They are not:
1135/// us-west-001 dials SSH over its public IP while us-west-002 was deliberately
1136/// repointed at its tailnet IP (R608-F10) precisely because the LAN address is
1137/// unreachable off-LAN. Decomposing them into user + host and recomposing
1138/// would silently undo per-machine decisions like that one. `yubaba` is the
1139/// field that *was* derived — mesh IP plus a fixed port, rewritten by
1140/// mesh-join — so that is where R707-T1 cut.
1141#[derive(Debug, Clone, Serialize, Deserialize)]
1142#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
1143pub struct ConnectSpec {
1144 /// Reachable IPv4/host for the box, e.g. `"45.32.194.254"`. Declared: which
1145 /// of a machine's several addresses the camp should use is an operator
1146 /// choice (public IP vs. LAN IP vs. tailnet IP).
1147 pub address: String,
1148 /// SSH target the camp dials for bootstrap + (pre-mesh) tunneled deploys,
1149 /// e.g. `"root@45.32.194.254"` or `"struc@100.64.0.4"`. Declared, whole —
1150 /// see the type doc. Pair with `identity_file` for a copy-pasteable
1151 /// `ssh -i <identity_file> <ssh>`.
1152 pub ssh: String,
1153 /// Private key path the camp uses to authenticate `ssh`, e.g.
1154 /// `"~/.ssh/yah"`. Every node in the fleet uses the same operator key
1155 /// today, but this is declared per-machine rather than assumed globally
1156 /// for the same reason `ssh` is whole rather than decomposed: a future
1157 /// node with a different key should not have to fight a hardcoded
1158 /// default. `~` is not shell-expanded by this crate — callers that shell
1159 /// out to `ssh`/`scp` pass it through `-i`, which expands it itself.
1160 pub identity_file: String,
1161 /// Port yubaba listens on. Declared reach; defaults to 7443 when omitted,
1162 /// which is every machine in the fleet today. Composed with the *observed*
1163 /// [`MachineRegistration::mesh_ipv4`] by [`MachineConfig::yubaba_url`].
1164 #[serde(default, skip_serializing_if = "Option::is_none")]
1165 pub yubaba_port: Option<u16>,
1166 /// Explicit yubaba base URL, overriding the composed form.
1167 ///
1168 /// Two live uses, both genuine declarations: a pre-mesh node saying
1169 /// `"http://127.0.0.1:7443"` — "I have no mesh address; reach me through
1170 /// the SSH tunnel to `ssh`" — and any node whose yubaba is not at
1171 /// `mesh_ipv4:port`. A URL here whose host *is* a mesh IP is the
1172 /// pre-R707-T1 shape; [`MachineConfig::normalize`] lifts it into
1173 /// `[registration].mesh_ipv4` and clears this field so the two cannot
1174 /// drift apart.
1175 #[serde(default, skip_serializing_if = "Option::is_none")]
1176 pub yubaba: Option<String>,
1177}
1178
1179/// Default yubaba listen port, used when `[connect].yubaba_port` is omitted.
1180pub const DEFAULT_YUBABA_PORT: u16 = 7443;
1181
1182impl ConnectSpec {
1183 /// Declared yubaba port, defaulting to [`DEFAULT_YUBABA_PORT`].
1184 pub fn yubaba_port(&self) -> u16 {
1185 self.yubaba_port.unwrap_or(DEFAULT_YUBABA_PORT)
1186 }
1187}
1188
1189#[derive(Debug, Clone, Serialize, Deserialize)]
1190#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
1191pub struct BucketSpec {
1192 pub name: String,
1193 pub public_read: bool,
1194}
1195
1196/// Per-camp mirror declaration from `.yah/cloud/mirrors/<id>/mirror.toml`
1197/// (folder form) or the legacy `.yah/cloud/mirrors/<id>.toml` (flat form).
1198///
1199/// The folder form is preferred for new mirrors so that per-mirror secrets
1200/// and override files can sit next to `mirror.toml` without polluting the
1201/// top-level `mirrors/` directory.
1202#[derive(Debug, Clone, Serialize, Deserialize)]
1203pub struct LegacyMirrorConfig {
1204 /// Logical camp name this mirror hosts, e.g. `"yah"` or `"noisetable"`.
1205 ///
1206 /// Serialised as `camp`; accepts the legacy `rig` spelling for files that
1207 /// predate the R137 rig→camp rename (one-time migration: `sed -i ''
1208 /// 's/^rig = /camp = /' ~/.yah/cloud/mirrors/*.toml`).
1209 #[serde(rename = "camp", alias = "rig")]
1210 pub camp: String,
1211 pub regions: Vec<String>,
1212 /// Workload names deployed as part of this mirror (references `workloads/<name>.toml`).
1213 /// Renamed from `services` in R092-F1; use `yah cloud config migrate-services-to-workloads`
1214 /// on repos that still have the old `services/` layout.
1215 #[serde(alias = "services")]
1216 pub workloads: Vec<String>,
1217 /// Base domain for Cloudflare-fronted services on this mirror's machines.
1218 /// Combined with the machine's `location` to build virtual-host names:
1219 /// e.g. `cloud_domain = "cloud.noisetable.example"` on machine in location
1220 /// `pdx` → Caddyfile site address `pdx.cloud.noisetable.example`.
1221 /// Optional: if unset the Caddyfile falls back to `:port` listeners.
1222 #[serde(default, skip_serializing_if = "Option::is_none")]
1223 pub cloud_domain: Option<String>,
1224}
1225
1226/// Error from loading or validating a single workload TOML file.
1227#[derive(Debug, Error)]
1228pub enum WorkloadConfigError {
1229 #[error("reading {path}: {source}")]
1230 Io {
1231 path: String,
1232 source: std::io::Error,
1233 },
1234 #[error("parsing {path}: {source}")]
1235 Toml {
1236 path: String,
1237 source: toml::de::Error,
1238 },
1239 #[error("invalid WorkloadSpec in {path}: {source}")]
1240 Shape {
1241 path: String,
1242 source: validate::ShapeError,
1243 },
1244}
1245
1246/// A workload declaration loaded from `.yah/cloud/workloads/<name>.toml`.
1247///
1248/// Each file is the human-authored TOML serialization of a [`WorkloadSpec`].
1249/// On load, the spec is validated against the shape layer; failures surface as
1250/// a [`CloudConfigError::Workload`] with the file path and field path.
1251#[derive(Debug, Clone, Serialize, Deserialize)]
1252pub struct WorkloadConfig {
1253 /// The validated spec.
1254 #[serde(flatten)]
1255 pub spec: WorkloadSpec,
1256}
1257
1258impl WorkloadConfig {
1259 /// Persist to `<cloud_dir>/workloads/<name>.toml`, creating the dir if needed.
1260 pub fn save(&self, cloud_dir: &Path) -> Result<()> {
1261 let dir = cloud_dir.join("workloads");
1262 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
1263 let path = dir.join(format!("{}.toml", self.spec.name));
1264 let s = toml::to_string_pretty(self)
1265 .with_context(|| format!("serializing workload {}", self.spec.name))?;
1266 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
1267 }
1268}
1269
1270/// Error surfaced by [`CloudConfig::load`] when a workload TOML fails validation.
1271#[derive(Debug, Error)]
1272pub enum CloudConfigError {
1273 #[error(transparent)]
1274 Anyhow(#[from] anyhow::Error),
1275 #[error("workload validation failed: {0}")]
1276 Workload(WorkloadConfigError),
1277}
1278
1279/// Mirror-to-machine assignment table from `.yah/cloud/topology.toml`.
1280///
1281/// Declares which logical mirror names are assigned to which machines.
1282/// This is the source-canonical placement until yubaba raft observes it
1283/// (per the migration tracker in the arch doc).
1284#[derive(Debug, Clone, Serialize, Deserialize, Default)]
1285pub struct TopologyConfig {
1286 /// Mirror→machine assignments.
1287 #[serde(default)]
1288 pub assignments: Vec<MirrorAssignment>,
1289 /// Declared buckets, logged by `yah cloud bucket create`.
1290 /// Source-canonical until yubaba raft observes actual placement.
1291 #[serde(default, skip_serializing_if = "Vec::is_empty")]
1292 pub buckets: Vec<BucketLogEntry>,
1293}
1294
1295impl TopologyConfig {
1296 /// Load from a `topology.toml` file, returning `Default` when absent.
1297 pub fn load(path: &Path) -> Result<Self> {
1298 if !path.exists() {
1299 return Ok(Self::default());
1300 }
1301 let s =
1302 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
1303 toml::from_str(&s).with_context(|| format!("parsing {}", path.display()))
1304 }
1305
1306 /// Persist to `topology.toml`, creating parent dirs if needed.
1307 pub fn save(&self, path: &Path) -> Result<()> {
1308 if let Some(parent) = path.parent() {
1309 std::fs::create_dir_all(parent)
1310 .with_context(|| format!("creating {}", parent.display()))?;
1311 }
1312 let s = toml::to_string_pretty(self).context("serializing topology")?;
1313 std::fs::write(path, s).with_context(|| format!("writing {}", path.display()))
1314 }
1315
1316 /// Find a declared bucket by name.
1317 pub fn bucket_by_name(&self, name: &str) -> Option<&BucketLogEntry> {
1318 self.buckets.iter().find(|b| b.name == name)
1319 }
1320
1321 /// Find a mutable declared bucket by name.
1322 pub fn bucket_by_name_mut(&mut self, name: &str) -> Option<&mut BucketLogEntry> {
1323 self.buckets.iter_mut().find(|b| b.name == name)
1324 }
1325
1326 /// Returns true if the bucket is declared as cross-machine (no owning machine).
1327 pub fn is_cross_machine_bucket(&self, name: &str) -> bool {
1328 self.buckets
1329 .iter()
1330 .any(|b| b.name == name && b.machine.is_none())
1331 }
1332}
1333
1334/// One mirror→machine placement entry in `topology.toml`.
1335#[derive(Debug, Clone, Serialize, Deserialize)]
1336pub struct MirrorAssignment {
1337 /// Logical mirror name, e.g. `"noisetable-pdx"`.
1338 pub mirror: String,
1339 /// Machine that hosts this mirror, e.g. `"noisetable-pdx-1"`.
1340 pub machine: String,
1341}
1342
1343/// A bucket declaration logged in `topology.toml` by `yah cloud bucket create`.
1344#[derive(Debug, Clone, Serialize, Deserialize)]
1345pub struct BucketLogEntry {
1346 pub name: String,
1347 /// Machine that owns this bucket. `None` marks it as cross-machine
1348 /// (no single-machine ownership; requires an explicit declaration in
1349 /// `topology.toml` before `yah cloud bucket create` will proceed without
1350 /// `--machine`).
1351 #[serde(default, skip_serializing_if = "Option::is_none")]
1352 pub machine: Option<String>,
1353 /// Logical location of the bucket, e.g. `"pdx"`.
1354 pub location: String,
1355 /// Current declared policy: `"private"` | `"public-read"` | `"signed-only"`.
1356 #[serde(default = "default_bucket_policy")]
1357 pub policy: String,
1358}
1359
1360fn default_bucket_policy() -> String {
1361 "private".to_string()
1362}
1363
1364/// Per-service config from `.yah/cloud/services/<name>.toml`.
1365///
1366/// **Deprecated.** The `services/` layout was replaced by `workloads/` in R092-F1.
1367/// Kept to allow in-place reads for repos that haven't migrated yet; use
1368/// `yah cloud config migrate-services-to-workloads` to upgrade.
1369#[derive(Debug, Clone, Serialize, Deserialize)]
1370pub struct LegacyServiceConfig {
1371 pub name: String,
1372 pub image: String,
1373 pub version: String,
1374 #[serde(default)]
1375 pub env: HashMap<String, String>,
1376 #[serde(default)]
1377 pub ports: Vec<PortMapping>,
1378 #[serde(default)]
1379 pub mesh_only: bool,
1380 /// Network interface this service binds to exclusively (e.g. `"tailscale0"`).
1381 ///
1382 /// When set the compose renderer emits `network_mode: "host"` and the
1383 /// service is NOT joined to the shared compose bridge network. The service
1384 /// process must bind its listen socket to the named interface's IP — for
1385 /// Postgres this means setting `POSTGRES_LISTEN_ADDRESSES` to the node's
1386 /// `tailscale ip --4` output at first boot. See [`crate::mesh_service`] for
1387 /// the standard pg_hba.conf snippet and ufw rules to pair with this field.
1388 #[serde(default, skip_serializing_if = "Option::is_none")]
1389 pub bind_interface: Option<String>,
1390
1391 /// Tenant this service belongs to (W206 isolation axis). Absent in the
1392 /// service TOML → [`TenantId::singleton`], keeping single-tenant machines
1393 /// on one shared compose network. When a machine hosts services from two
1394 /// or more distinct tenants, the compose renderer (R558-T2) splits them
1395 /// into per-tenant `<tenant>-<tier>` networks so cross-tenant stacks on the
1396 /// same host are not bridged together.
1397 #[serde(default = "TenantId::singleton")]
1398 pub tenant: TenantId,
1399}
1400
1401#[derive(Debug, Clone, Serialize, Deserialize)]
1402pub struct PortMapping {
1403 pub host: u16,
1404 pub container: u16,
1405}
1406
1407/// A loaded service plus its per-environment mirrors.
1408///
1409/// Wraps the `service.toml` body and the directory of `mirrors/<env>.toml`
1410/// files that project the service onto concrete infra.
1411#[derive(Debug, Clone, Serialize, Deserialize)]
1412pub struct ServiceWithMirrors {
1413 pub service: ServiceConfig,
1414 /// Mirrors keyed by environment name (file stem of `mirrors/<env>.toml`).
1415 pub mirrors: BTreeMap<String, MirrorConfig>,
1416 /// Transform recipe names keyed by component id. Populated from each
1417 /// static-asset component's `workload.toml` at load time — not stored
1418 /// in service.toml. Only present for components that declare
1419 /// `[asset.derive.transform] recipe = "..."`.
1420 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
1421 pub component_transform_recipes: BTreeMap<String, String>,
1422 /// Nodes each mirror's passway front door is placed on, keyed by env —
1423 /// exactly what [`MirrorConfig::passway_machines`] returns, with the envs
1424 /// that declare no passway edge left out.
1425 ///
1426 /// Derived at load time like `component_transform_recipes` above: it is
1427 /// stored in no TOML file. It exists so that a consumer of this wire type —
1428 /// the desktop `service_list` command, and through it the Services tab's
1429 /// custom-domain panel — never reconciles the two `ingress` spellings
1430 /// itself. An env present here with an **empty** list is a passway edge
1431 /// whose placement is co-located rather than declared; see
1432 /// [`MirrorConfig::passway_machines`] for why that is a different answer
1433 /// from being absent.
1434 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
1435 pub passway_machines: BTreeMap<String, Vec<String>>,
1436}
1437
1438/// All cloud config loaded from a workspace root (the parent of `.yah/`).
1439///
1440/// Reads two trees:
1441/// - `.yah/infra/` — `machines/`, `providers/`
1442/// - `.yah/services/<svc>/` — `service.toml` + `mirrors/<env>.toml`
1443///
1444/// Pre-R215 fields (`legacy_mirrors`, `legacy_services`, `workloads`,
1445/// `topology`) are still populated from `.yah/cloud/` when present so
1446/// pre-R215 callers (compose.rs, bucket commands) keep compiling — they
1447/// just see empty collections in a post-B1 workspace where the legacy
1448/// data was deleted. These fields are scheduled for removal in B3-T3.
1449#[derive(Debug)]
1450pub struct CloudConfig {
1451 /// Workspace root that was loaded — useful for path-resolving
1452 /// component references on a [`ServiceComponent`].
1453 pub workspace_root: std::path::PathBuf,
1454
1455 // ─── R215+ tree ────────────────────────────────────────────────────────
1456 /// `.yah/infra/machines/<name>.toml`
1457 pub machines: Vec<MachineConfig>,
1458 /// `.yah/infra/providers/<id>.toml`
1459 pub providers: Vec<ProviderConfig>,
1460 /// Provenance for every entry in `machines` that came from a linked
1461 /// `.yah/infra/sources.toml` source rather than this camp's own
1462 /// `.yah/infra/machines/` (R615-F2 / W274). Keyed by
1463 /// [`MachineConfig::name`]; a name absent here is camp-local. Empty from
1464 /// [`CloudConfig::load_from_config_dir`] — see its doc for why sources
1465 /// don't apply to multi-root sibling trees.
1466 pub machine_origins: BTreeMap<String, InfraOrigin>,
1467 /// Same as [`machine_origins`](Self::machine_origins), keyed by
1468 /// [`ProviderConfig::id`].
1469 pub provider_origins: BTreeMap<String, InfraOrigin>,
1470 /// `.yah/services/<svc>/` — service.toml plus mirrors/<env>.toml.
1471 pub services: BTreeMap<String, ServiceWithMirrors>,
1472 /// `.yah/domains/<name>.toml` — public-facing routing manifests
1473 /// (R347). Single file per domain; no nested per-env tree because
1474 /// domains themselves aren't projected onto infra — they describe
1475 /// how a Worker bundle ingresses requests onto services.
1476 pub domains: BTreeMap<String, DomainConfig>,
1477
1478 // ─── Pre-R215 legacy (slated for removal in B3-T3) ────────────────────
1479 /// Legacy mirrors from `.yah/cloud/mirrors/`.
1480 pub legacy_mirrors: Vec<LegacyMirrorConfig>,
1481 /// Workloads from `.yah/cloud/workloads/*.toml` (R092-F1 schema).
1482 pub workloads: Vec<WorkloadConfig>,
1483 /// Topology from `.yah/cloud/topology.toml` (mirror→machine assignments).
1484 pub topology: TopologyConfig,
1485 /// Legacy services from `.yah/cloud/services/*.toml` (pre-R092 layout).
1486 pub legacy_services: Vec<LegacyServiceConfig>,
1487}
1488
1489impl CloudConfig {
1490 /// Load all cloud config rooted at `workspace_root` (the parent of `.yah/`).
1491 ///
1492 /// Reads the R215+ tree (`.yah/infra/`, `.yah/services/<svc>/`) eagerly
1493 /// and the pre-R215 `.yah/cloud/` tree opportunistically. Returns `Err`
1494 /// immediately if any TOML fails to parse or a workload TOML fails
1495 /// shape validation; the error includes the file path and field path.
1496 ///
1497 /// Cross-ref validation runs after both trees finish loading: every
1498 /// `mirror.providers.X.use = "<id>"` must resolve to a real provider
1499 /// declared under `.yah/infra/providers/`.
1500 ///
1501 /// R844-B7 — **a missing `.yah/` is a wrong-root error, not an empty
1502 /// fleet.** Every sub-loader below tolerates a missing directory by
1503 /// returning empty, so before this check a call against the wrong
1504 /// directory produced a perfectly valid `CloudConfig` with zero machines,
1505 /// zero services and zero providers. Nothing downstream can tell that
1506 /// apart from a camp that genuinely declares nothing, so the failure
1507 /// surfaces as an operation that silently does nothing to nothing: a
1508 /// collate that renders no backends, a fanout that asks no nodes, a
1509 /// rollout that plans against an empty fleet. It was found the hard way —
1510 /// a live-fleet test in `app/yah/cli` called this with `"."`, which under
1511 /// `cargo test` is the *package* root, and passed while measuring nothing.
1512 ///
1513 /// The line is drawn at `.yah/` and only there: a workspace whose
1514 /// `.yah/infra/machines/` is absent or empty is a real, if unusual, camp
1515 /// with an empty fleet and still loads. `unknown` is not `answered with
1516 /// none`.
1517 pub fn load(workspace_root: &Path) -> Result<Self> {
1518 let yah_dir = crate::paths::yah_dir(workspace_root);
1519 if !yah_dir.is_dir() {
1520 anyhow::bail!(
1521 "not a yah workspace: no {} — expected the camp root (the parent \
1522 of `.yah/`), got {}. This is a wrong-root error, not an empty \
1523 fleet; a camp with no machines declared still has a `.yah/`.",
1524 yah_dir.display(),
1525 workspace_root.display(),
1526 );
1527 }
1528
1529 let mut providers = load_providers(&crate::paths::providers_dir(workspace_root))?;
1530 let services = load_services(&crate::paths::services_dir(workspace_root), workspace_root)?;
1531 let domains = load_domains(&crate::paths::domains_dir(workspace_root))?;
1532
1533 Self::cross_ref_validate(&providers, &services, &domains)?;
1534
1535 // Legacy `.yah/cloud/` reads — empty in post-B1 workspaces. Wrapped in
1536 // a helper so a missing tree is silent (no error, no warning).
1537 let cloud_dir = crate::paths::legacy_cloud_dir(workspace_root);
1538 let (legacy_mirrors, legacy_workloads, topology, legacy_services) = if cloud_dir.exists() {
1539 (
1540 load_mirrors(cloud_dir.join("mirrors"))?,
1541 load_workloads(cloud_dir.join("workloads"))?,
1542 load_topology(cloud_dir.join("topology.toml"))?,
1543 load_dir::<LegacyServiceConfig>(cloud_dir.join("services"))?,
1544 )
1545 } else {
1546 Default::default()
1547 };
1548
1549 // Workloads come from `.yah/infra/workloads/` (R215+). R568-T7: before
1550 // that path was read here, this field was populated *only* from the
1551 // legacy tree above — which R222-B1 emptied — so `cfg.workload(name)`
1552 // resolved nothing in every post-R215 camp and `yah cloud workload
1553 // deploy` could not find any declaration at all. The bug survived
1554 // because the only workloads ever deployed were forge/QED runs, which
1555 // build their spec in memory and never come through here. Same
1556 // dedupe-by-name shape as machines below: R215+ wins.
1557 let mut workloads = load_workloads(crate::paths::workloads_dir(workspace_root))?;
1558 let workload_names: std::collections::HashSet<String> =
1559 workloads.iter().map(|w| w.spec.name.clone()).collect();
1560 for w in legacy_workloads {
1561 if !workload_names.contains(&w.spec.name) {
1562 workloads.push(w);
1563 }
1564 }
1565
1566 // R870-B13: machines are resolved by [`resolve_fleet_inventory`] —
1567 // camp-local, the pre-R215 legacy tree, and every machine borrowed
1568 // through `.yah/infra/sources.toml`, in that precedence. This used to
1569 // be spelled out inline here, which made `CloudConfig::load` the only
1570 // reader that saw borrowed machines at all; the two *resolution*
1571 // callers in `validate`/`reconciler::domain` read a camp-local-only
1572 // loader and could not see a borrowing camp's fleet. There is now one
1573 // implementation and three callers.
1574 let fleet = resolve_fleet_inventory(workspace_root)?;
1575
1576 // Providers overlay here rather than inside `resolve_fleet_inventory`:
1577 // that function answers "which machines does this camp have", which is
1578 // the question with three readers. Providers have exactly one reader —
1579 // this load — so hoisting them would build a seam nothing crosses.
1580 let mut provider_origins = BTreeMap::new();
1581 overlay_source_providers(
1582 workspace_root,
1583 &fleet.sources,
1584 &mut providers,
1585 &mut provider_origins,
1586 );
1587
1588 Ok(Self {
1589 workspace_root: workspace_root.to_path_buf(),
1590 machines: fleet.machines,
1591 providers,
1592 machine_origins: fleet.origins,
1593 provider_origins,
1594 services,
1595 domains,
1596 legacy_mirrors,
1597 workloads,
1598 topology,
1599 legacy_services,
1600 })
1601 }
1602
1603 /// Load the R215+ tree (`infra/`, `services/`, `domains/`) rooted at an
1604 /// arbitrary config directory instead of the hardcoded `.yah/`. This is the
1605 /// building block for multi-root deployments (W206 config layout (b), sibling
1606 /// `.noisetable/` trees) — see [`crate::multi_root`]. Part of R558-F4.
1607 ///
1608 /// `config_dir` is the `.X/` directory itself (e.g. `<parent>/.noisetable`);
1609 /// `workspace_root` remains the camp dir (the config dir's parent) so a
1610 /// component's `path` reference resolves against the same tree the classic
1611 /// [`CloudConfig::load`] uses. The legacy `.yah/cloud/` reads are skipped —
1612 /// multi-root deployments are post-R215 by construction — so `legacy_*`,
1613 /// `workloads`, and `topology` come back empty. Machines are read from
1614 /// `config_dir/infra/machines` directly (sibling trees declare their own
1615 /// inventory or none).
1616 ///
1617 /// R615-F2 decision, explicit rather than silent: **sources.toml overlay
1618 /// does NOT apply here.** This function
1619 /// exists specifically because a multi-root sibling tree (W206 layout
1620 /// (b), e.g. `.noisetable/`) is a *second config root inside the same
1621 /// camp*, not a second camp — `config_dir` is already wherever the
1622 /// caller decided this tree's infra lives, and `.yah/infra/sources.toml`
1623 /// (singular, tied to `paths::infra_dir(workspace_root)`) has no
1624 /// well-defined meaning for an arbitrary `config_dir` that isn't that
1625 /// path. A sibling tree that wants borrowed infra declares its own
1626 /// `sources.toml` under whichever root actually calls
1627 /// [`CloudConfig::load`] for it; `machine_origins`/`provider_origins`
1628 /// come back empty here, not wrong — there is nothing to overlay.
1629 pub fn load_from_config_dir(config_dir: &Path, workspace_root: &Path) -> Result<Self> {
1630 let providers = load_providers(&config_dir.join("infra").join("providers"))?;
1631 let services = load_services(&config_dir.join("services"), workspace_root)?;
1632 let domains = load_domains(&config_dir.join("domains"))?;
1633
1634 Self::cross_ref_validate(&providers, &services, &domains)?;
1635
1636 let machines = load_dir::<MachineConfig>(config_dir.join("infra").join("machines"))?;
1637
1638 Ok(Self {
1639 workspace_root: workspace_root.to_path_buf(),
1640 machines,
1641 providers,
1642 machine_origins: BTreeMap::new(),
1643 provider_origins: BTreeMap::new(),
1644 services,
1645 domains,
1646 legacy_mirrors: vec![],
1647 workloads: vec![],
1648 topology: TopologyConfig::default(),
1649 legacy_services: vec![],
1650 })
1651 }
1652
1653 /// Cross-reference validation shared by [`CloudConfig::load`] and
1654 /// [`CloudConfig::load_from_config_dir`]: every mirror `providers.X.use =
1655 /// "<id>"` must resolve to a declared provider, and every domain route's
1656 /// `component = "<service>/<component-id>"` must resolve to a real component.
1657 fn cross_ref_validate(
1658 providers: &[ProviderConfig],
1659 services: &BTreeMap<String, ServiceWithMirrors>,
1660 domains: &BTreeMap<String, DomainConfig>,
1661 ) -> Result<()> {
1662 // Mirror `use = "<id>"` slots must resolve to a declared provider.
1663 let provider_ids: std::collections::HashSet<&str> =
1664 providers.iter().map(|p| p.id.as_str()).collect();
1665 for (svc_name, svc) in services {
1666 for (env, mirror) in &svc.mirrors {
1667 for (slot, body) in &mirror.providers {
1668 if let Some(id) = body.provider_id() {
1669 if !provider_ids.contains(id) {
1670 anyhow::bail!(
1671 "services/{svc_name}/mirrors/{env}.toml: \
1672 providers.{slot}.use = \"{id}\" — no such provider; \
1673 declare it at infra/providers/{id}.toml"
1674 );
1675 }
1676 }
1677 }
1678 // An `[[ingress]]` edge's own `use` is the same kind of
1679 // reference (R845) and gets the same check: a typo there is
1680 // otherwise invisible until `yah cloud apply` reaches the
1681 // Cloudflare arm and fails on a missing provider file.
1682 for (idx, edge) in mirror.ingress_edge_slice().iter().enumerate() {
1683 if let Some(id) = edge.provider_id.as_deref() {
1684 if !provider_ids.contains(id) {
1685 anyhow::bail!(
1686 "services/{svc_name}/mirrors/{env}.toml: \
1687 ingress[{idx}].use = \"{id}\" — no such provider; \
1688 declare it at infra/providers/{id}.toml"
1689 );
1690 }
1691 }
1692 }
1693 }
1694 }
1695
1696 // R870-B11. Two bundle-tier components sharing a mount would stage
1697 // into the same `app/dist/<mount>/` prefix inside the service's one
1698 // assembled bundle and silently clobber each other on disk — the
1699 // exact failure class this ticket exists to fix, one level down
1700 // (there it was two components silently overwriting the same
1701 // *workload*; here it would be two components silently overwriting
1702 // the same *path inside* the workload). A mount is owned by exactly
1703 // one component; refuse the config before the clobber happens.
1704 //
1705 // R870-F23 widens the same loop to the workload tier rather than
1706 // adding a parallel one. A mount is owned by exactly one component
1707 // whichever tier serves it: two workload-tier components at one mount
1708 // would hand the inner door two upstream sets for one prefix, and two
1709 // components in *different* tiers at one mount is the same clobber
1710 // read from the routing side — the request reaches whichever of the
1711 // bundle and the workload the mount table happened to name. So the
1712 // rule is now "one component per mount, service-wide", and only the
1713 // explanation branches on tier.
1714 for (svc_name, svc) in services {
1715 let mut owner_by_mount: BTreeMap<String, (&str, DeployTier)> = BTreeMap::new();
1716 for component in &svc.service.components {
1717 let bundle_tier =
1718 component.kind == "mesofact-static" || component.kind == "mesofact-spa";
1719 if !bundle_tier && component.deploy != DeployTier::Workload {
1720 continue;
1721 }
1722 let mount = component
1723 .mount
1724 .as_deref()
1725 .map(normalize_mount)
1726 .unwrap_or_default();
1727 if let Some((existing, existing_tier)) =
1728 owner_by_mount.insert(mount.clone(), (&component.id, component.deploy))
1729 {
1730 let where_ = if mount.is_empty() {
1731 "the service root (no `mount`)".to_string()
1732 } else {
1733 format!("mount = \"/{mount}\"")
1734 };
1735 let why = if existing_tier == component.deploy {
1736 match component.deploy {
1737 DeployTier::Bundle => {
1738 "a bundle-tier component's mount is a storage prefix inside the \
1739 service's single assembled bundle (app/dist/<mount>/), so two \
1740 components at the same mount would stage into the same path and \
1741 silently overwrite each other"
1742 }
1743 DeployTier::Workload => {
1744 "a workload-tier component's mount is its prefix in the service's \
1745 inner-door route table, so two components at the same mount would \
1746 claim one prefix and requests would reach whichever the table \
1747 named"
1748 }
1749 }
1750 } else {
1751 "one is staged into the service bundle and the other deploys as its own \
1752 workload, so the mount names two different things that serve one prefix \
1753 — the inner door can only route it to one of them"
1754 };
1755 anyhow::bail!(
1756 "services/{svc_name}/service.toml: components \"{existing}\" and \
1757 \"{}\" both declare {where_} — {why}. Give one of them a distinct \
1758 `mount`.",
1759 component.id,
1760 );
1761 }
1762 }
1763 }
1764
1765 // Every domain route's `component = "<service>/<component-id>"` must
1766 // resolve to a real component.
1767 for (dom_name, dom) in domains {
1768 for (idx, route) in dom.routes.iter().enumerate() {
1769 let Some(component_ref) = route.mode.component() else {
1770 continue; // redirects don't reference components
1771 };
1772 let Some((svc_name, comp_id)) = split_component_ref(component_ref) else {
1773 anyhow::bail!(
1774 "domains/{dom_name}.toml: routes[{idx}].component = \
1775 \"{component_ref}\" — expected \"<service>/<component-id>\""
1776 );
1777 };
1778 let Some(svc) = services.get(svc_name) else {
1779 anyhow::bail!(
1780 "domains/{dom_name}.toml: routes[{idx}].component = \
1781 \"{component_ref}\" — no such service \"{svc_name}\" \
1782 under services/"
1783 );
1784 };
1785 let Some(component) = svc.service.components.iter().find(|c| c.id == comp_id)
1786 else {
1787 anyhow::bail!(
1788 "domains/{dom_name}.toml: routes[{idx}].component = \
1789 \"{component_ref}\" — service \"{svc_name}\" has no \
1790 component with id \"{comp_id}\""
1791 );
1792 };
1793
1794 // R746: a mounted component must be routed where it publishes.
1795 // The publisher writes its bundle under the mount and the front
1796 // door looks a request up by its own path, so a route path and
1797 // a mount that disagree produce a 404 with its cause two files
1798 // away. Checked in both directions, since either one alone is
1799 // the same silent miss.
1800 //
1801 // Static routes only: `mount` is a *storage* prefix, and a
1802 // backend route proxies to an origin that owns its own paths.
1803 if !matches!(route.mode, RouteMode::Static { .. }) {
1804 continue;
1805 }
1806 let mount = component.mount.as_deref().map(normalize_mount);
1807 let route_prefix = route_path_prefix(&route.path);
1808 if let Some(mount) = mount {
1809 if mount != route_prefix {
1810 anyhow::bail!(
1811 "domains/{dom_name}.toml: routes[{idx}].path = \
1812 \"{path}\" serves \"{component_ref}\", which \
1813 declares mount = \"/{mount}\" — a mounted \
1814 component publishes under its mount, so the route \
1815 must be \"/{mount}\" or \"/{mount}/*\" (or drop \
1816 the mount to serve from the service root)",
1817 path = route.path,
1818 );
1819 }
1820 } else if !route_prefix.is_empty() {
1821 anyhow::bail!(
1822 "domains/{dom_name}.toml: routes[{idx}].path = \
1823 \"{path}\" serves \"{component_ref}\", which declares \
1824 no `mount` — its bundle publishes at the service root, \
1825 so nothing is stored under \"/{route_prefix}\". Set \
1826 mount = \"/{route_prefix}\" on the component, or route \
1827 it at \"/*\"",
1828 path = route.path,
1829 );
1830 }
1831 }
1832 }
1833 Ok(())
1834 }
1835
1836 /// Look up a domain manifest by name (file stem under `.yah/domains/`).
1837 pub fn domain(&self, name: &str) -> Option<&DomainConfig> {
1838 self.domains.get(name)
1839 }
1840
1841 pub fn machine(&self, name: &str) -> Option<&MachineConfig> {
1842 self.machines.iter().find(|m| m.name == name)
1843 }
1844
1845 /// Look up a provider by id (matches `provider.id`, not the file stem).
1846 pub fn provider(&self, id: &str) -> Option<&ProviderConfig> {
1847 self.providers.iter().find(|p| p.id == id)
1848 }
1849
1850 /// Look up a service by name (matches `service.toml`'s `name` field).
1851 pub fn service(&self, name: &str) -> Option<&ServiceWithMirrors> {
1852 self.services.get(name)
1853 }
1854
1855 /// Look up a legacy mirror by camp name (pre-R215 .yah/cloud/mirrors/).
1856 pub fn legacy_mirror(&self, camp: &str) -> Option<&LegacyMirrorConfig> {
1857 self.legacy_mirrors.iter().find(|m| m.camp == camp)
1858 }
1859
1860 pub fn workload(&self, name: &str) -> Option<&WorkloadConfig> {
1861 self.workloads.iter().find(|w| w.spec.name == name)
1862 }
1863
1864 /// Every machine declaring `sovereign_group == group`, in declaration order.
1865 ///
1866 /// W305/R742-F3. A sovereign group has no file of its own — it exists only
1867 /// as the set of machines that name the same string — so "which boxes are
1868 /// the dev cluster" has to be *derived*, and before this it was not derived
1869 /// anywhere: `yah cloud rollout plan` still takes a hand-listed
1870 /// `--voter us-west-011 --voter us-west-013 …` for a fact the machine TOMLs
1871 /// already state (W314 gap 1).
1872 ///
1873 /// **This is not placement.** Resolving a group to its members is a
1874 /// *lookup*, and it stays outside [`RequiredSpec`] on purpose — see
1875 /// [`MachineConfig::sovereign_group`]. `migrate` calls this to pick the
1876 /// candidate set it then admits a workload against; nothing here filters
1877 /// scheduling, and adding `sovereign_group` to `matches` would still be the
1878 /// category error that doc warns about.
1879 ///
1880 /// An empty result means no machine declares `group`, which is
1881 /// indistinguishable from a typo — callers should say so with
1882 /// [`Self::declared_sovereign_groups`] rather than reporting "no
1883 /// candidates".
1884 pub fn machines_in_group(&self, group: &str) -> Vec<&MachineConfig> {
1885 self.machines
1886 .iter()
1887 .filter(|m| m.sovereign_group.as_deref() == Some(group))
1888 .collect()
1889 }
1890
1891 /// Every distinct `sovereign_group` declared by any machine, sorted.
1892 ///
1893 /// Exists so a bad `--to` names the real vocabulary instead of complaining
1894 /// abstractly — the same fail-loud shape [`taint_effect`]'s legal-key list
1895 /// gives `check_inert_taints`. Standalone machines (`None`) contribute
1896 /// nothing: "in no group" is not a group you can migrate *to*.
1897 pub fn declared_sovereign_groups(&self) -> Vec<&str> {
1898 let mut groups: Vec<&str> = self
1899 .machines
1900 .iter()
1901 .filter_map(|m| m.sovereign_group.as_deref())
1902 .collect();
1903 groups.sort_unstable();
1904 groups.dedup();
1905 groups
1906 }
1907
1908 /// F16 placement v1: the first machine satisfying every hard axis of `req`
1909 /// (region/zone/provider membership + mesh_tags superset). Declaration order
1910 /// in `.yah/infra/machines/` decides ties — deterministic-greedy, no
1911 /// backtracking. A fully-unconstrained `req` matches the first machine.
1912 ///
1913 /// Fails loud with the constraint summary and the candidate machine names
1914 /// when nothing matches, so `yah cloud apply` surfaces *why* placement
1915 /// failed instead of a silent empty set.
1916 pub fn resolve_machine(&self, req: &RequiredSpec) -> Result<&MachineConfig> {
1917 resolve_machine_among(&self.machines, req)
1918 }
1919
1920 /// F16 placement at horizontal scale: the first
1921 /// [`RequiredSpec::replica_count`] machines satisfying every hard axis of
1922 /// `req`, in declaration order (R844-F8).
1923 ///
1924 /// The N-valued form of [`Self::resolve_machine`], which is the N=1 case of
1925 /// this and not a different selector — both land in [`select_matching`].
1926 /// That shared bottom is what makes the deploy resolver
1927 /// (`reconciler::mesofact_bundle::resolve_bundle_machines`, which calls
1928 /// this) and the ingress planner's
1929 /// (`reconciler::ingress::resolve_ingress_placements`, which calls
1930 /// [`resolve_machines_among`] over the same `machines` slice) agree on the
1931 /// same N machines **by construction**. They must agree set-for-set, not
1932 /// merely in count: a front door aimed at nodes the workload was never
1933 /// deployed to renders a *subset* of the backends, which is the failure that
1934 /// looks like it worked.
1935 pub fn resolve_machines(&self, req: &RequiredSpec) -> Result<Vec<&MachineConfig>> {
1936 resolve_machines_among(&self.machines, req)
1937 }
1938
1939 /// F16 placement: first machine whose `mesh_tags` is a superset of
1940 /// `required`. Declaration order in `.yah/infra/machines/` decides ties.
1941 /// Empty `required` matches the first machine; callers should treat
1942 /// empty-required as "no constraint" and skip this lookup.
1943 ///
1944 /// Back-compat thin wrapper over [`CloudConfig::resolve_machine`] for the
1945 /// mesh-tags-only call sites that predate the topology axes.
1946 pub fn resolve_machine_by_mesh_tags(&self, required: &[String]) -> Option<&MachineConfig> {
1947 let req = RequiredSpec {
1948 mesh_tags: required.to_vec(),
1949 ..Default::default()
1950 };
1951 self.resolve_machine(&req).ok()
1952 }
1953
1954 /// Admission: resolve the target machine for a remote [`WorkloadSpec`],
1955 /// honoring the R594 mesh-tag node-selector annotation
1956 /// (`velveteen_exec::remote::NODE_SELECTOR_MESH_TAGS_ANNOTATION` =
1957 /// `yah.node-selector.mesh-tags`, comma-joined).
1958 ///
1959 /// The producer side (`velveteen_exec::remote::build_workload_spec`, R594) writes
1960 /// `TaskLocation::RemoteAny.mesh_tags` — e.g. `[tag:build-worker, arch:x86]`
1961 /// from [`qed::platform::build_worker_mesh_tags`] — into the workload's
1962 /// annotations. This is the consumer: candidates are restricted to machines
1963 /// whose `mesh_tags` are a **superset** of the requested set, so an amd64
1964 /// build lands on the `arch:x86` build-worker (us-west-002) and an arm64
1965 /// build on a `arch:arm` Pi5. Declaration order in `.yah/infra/machines/`
1966 /// breaks ties.
1967 ///
1968 /// An absent or empty annotation means "no mesh-tag constraint" — pre-R594
1969 /// behavior (any node), matching [`RequiredSpec::is_unconstrained`].
1970 ///
1971 /// This is the single admission seam: R572-F5 extends it with the capacity
1972 /// floor (workload request fits node allocatable−committed) and taint
1973 /// repulsion/affinity by enriching [`RequiredSpec::matches`] /
1974 /// [`Self::resolve_machine`]. Do not fork a second selector.
1975 pub fn admit_workload(&self, ws: &WorkloadSpec) -> Result<&MachineConfig> {
1976 self.resolve_machine(&admission_spec(ws, &self.workloads))
1977 }
1978
1979 /// Every machine that admits `ws`, in declaration order — the *pool*
1980 /// [`Self::admit_workload`] returns the head of (R605-T14).
1981 ///
1982 /// # Why a pool and not just the winner
1983 ///
1984 /// `tag:build-worker` is a statement that the tagged boxes are
1985 /// **interchangeable**: a build is booked against the tag, not against
1986 /// `us-west-002`. Returning one machine forced every caller to act as if it
1987 /// were booked against a name, and admission has no liveness input — so a
1988 /// tagged box that is asleep won the file-name tie-break and its builds
1989 /// failed rather than landing on the identical box next to it. That is
1990 /// exactly what happened on 2026-09-03 when `us-west-002` regained the tag.
1991 ///
1992 /// The fix is **not** to teach this function about liveness. It stays a pure
1993 /// function of the declared inventory (see `xtask/tests/fleet_build_placement.rs`
1994 /// on why a placement pin that needs the network is a flake). It hands the
1995 /// dispatcher the whole interchangeable set instead, and the dispatcher —
1996 /// which has the network — probes and fails over within it:
1997 /// `app/yah/cli/src/yubaba_client.rs`'s `MeshYubabaClient::deploy`.
1998 ///
1999 /// Order is the declaration order `admit_workload` already used, and callers
2000 /// should preserve it as their preference order rather than load-balancing
2001 /// across it: a retried build wants the node still holding its warm
2002 /// `target/`, which is the same reason [`first_match`] is deliberately
2003 /// first-fit.
2004 ///
2005 /// `Err` — never `Ok(vec![])` — when nothing admits `ws`, carrying the same
2006 /// message [`Self::admit_workload`] would have produced. "No node admits
2007 /// this" and "the pool is empty" are the same failure and must read the same.
2008 pub fn admit_workload_candidates(&self, ws: &WorkloadSpec) -> Result<Vec<&MachineConfig>> {
2009 let req = admission_spec(ws, &self.workloads);
2010 let all: Vec<&MachineConfig> = self.machines.iter().collect();
2011 let matched = matching(&all, &req);
2012 if matched.is_empty() {
2013 // Delegate the wording so the two paths cannot drift apart.
2014 return Err(first_match(&all, &req, DECLARED_POOL, EMPTY_DECLARED_POOL)
2015 .expect_err("matching() found nothing, so first_match cannot succeed"));
2016 }
2017 Ok(matched)
2018 }
2019
2020 /// [`Self::admit_workload`] restricted to the machines of one sovereign
2021 /// group (W305/R742-F3, `yah cloud migrate --to <group>`).
2022 ///
2023 /// Same [`RequiredSpec`], same [`RequiredSpec::matches`], same
2024 /// declaration-order tie-break — only the candidate *set* differs. That is
2025 /// the whole reason this is a narrowing of the admission seam rather than a
2026 /// second selector: a workload that cannot be scheduled onto a group's
2027 /// boxes must fail here for exactly the reason it would fail anywhere else,
2028 /// and `no-appliance` on the dev Pis (W305 finding 2) is precisely the case
2029 /// that must not be silently routed around by a migration verb.
2030 ///
2031 /// `Err` when the group has no members *or* when no member admits `ws`; the
2032 /// two are different mistakes, so callers wanting to tell them apart should
2033 /// check [`Self::machines_in_group`] first.
2034 pub fn admit_workload_in_group(
2035 &self,
2036 ws: &WorkloadSpec,
2037 group: &str,
2038 ) -> Result<&MachineConfig> {
2039 let members = self.machines_in_group(group);
2040 let empty_pool = format!(
2041 "(no machine declares sovereign_group = \"{group}\" — declared groups: {})",
2042 match self.declared_sovereign_groups().as_slice() {
2043 [] => "(none)".to_string(),
2044 gs => gs.join(", "),
2045 }
2046 );
2047 first_match(
2048 &members,
2049 &admission_spec(ws, &self.workloads),
2050 &format!("machines in sovereign group '{group}'"),
2051 &empty_pool,
2052 )
2053 }
2054}
2055
2056/// **The** placement selector: the first candidate satisfying every axis of
2057/// `req`, declaration order breaking ties, deterministic-greedy with no
2058/// backtracking.
2059///
2060/// Every path that picks a machine goes through here, and the only thing any
2061/// of them varies is *which machines are candidates* — never the predicate.
2062/// [`CloudConfig::resolve_machine`] passes the whole fleet;
2063/// [`CloudConfig::admit_workload_in_group`] passes one sovereign group's
2064/// members. That split is the point: a candidate-set narrowing composes with
2065/// the [`RequiredSpec`] axes for free, whereas expressing the same narrowing
2066/// *as* an axis would put facts like blast radius into a filter they must
2067/// never be in (see [`MachineConfig::sovereign_group`]).
2068///
2069/// So a new placement scope is a new candidate set plus a `pool` label, and a
2070/// new placement *constraint* is a field on [`RequiredSpec`] — those are the
2071/// two extension points, and neither is a second selector. `pool` and
2072/// `empty_pool` exist only so the failure names the set it actually searched;
2073/// a refusal that says "no candidates" without saying *among what* is one the
2074/// operator has to reconstruct by hand.
2075/// F16 placement v1 resolution over an explicit machine list — the
2076/// `.machines`-only half of [`CloudConfig::resolve_machine`], for callers that
2077/// have loaded just the machines tree rather than the whole cross-ref-validated
2078/// config.
2079///
2080/// R772: `resolve_ingress_placements` (`reconciler::ingress`) is the reason
2081/// this is `pub(crate)` rather than staying folded into
2082/// `CloudConfig::resolve_machine` — ingress collation walks every mirror in
2083/// the workspace and has no business hard-failing over an unrelated mirror's
2084/// `providers.X.use = "<id>"` typo, which is what going through
2085/// `CloudConfig::load`'s cross-ref validation would do. "Do not fork a second
2086/// selector" (see the module doc above) still holds: this is the *same*
2087/// [`first_match`], just handed a narrower candidate set than `self.machines`.
2088pub(crate) fn resolve_machine_among<'a>(
2089 machines: &'a [MachineConfig],
2090 req: &RequiredSpec,
2091) -> Result<&'a MachineConfig> {
2092 let all: Vec<&MachineConfig> = machines.iter().collect();
2093 first_match(&all, req, DECLARED_POOL, EMPTY_DECLARED_POOL)
2094}
2095
2096/// R844-F8: [`resolve_machine_among`] widened to the constraint's own replica
2097/// count — the first [`RequiredSpec::replica_count`] matching machines, in the
2098/// same declaration order, from the same candidate slice.
2099///
2100/// **The one entry point both resolvers share.**
2101/// `reconciler::ingress::resolve_ingress_placements` calls this directly and
2102/// `reconciler::mesofact_bundle::resolve_bundle_machines` reaches it through
2103/// [`CloudConfig::resolve_machines`], both over `cfg.machines` — so the ingress
2104/// planner and the deployer cannot pick different subsets. That is a structural
2105/// guarantee, not a tested coincidence, and it has to be: discovery aimed at a
2106/// node the bundle was never placed on publishes a hostname with a dead
2107/// backend behind it, and at scale > 1 the front door still answers from the
2108/// nodes that *did* get it.
2109///
2110/// Determinism is therefore part of correctness here. `machines` arrives in
2111/// file-name order (`load_dir`, pinned by
2112/// `machines_load_in_file_name_order_not_read_dir_order`), and selection is a
2113/// stable prefix of that order — so "the first two matching" is the same two
2114/// on both sides of the same tree.
2115pub(crate) fn resolve_machines_among<'a>(
2116 machines: &'a [MachineConfig],
2117 req: &RequiredSpec,
2118) -> Result<Vec<&'a MachineConfig>> {
2119 let all: Vec<&MachineConfig> = machines.iter().collect();
2120 select_matching(
2121 &all,
2122 req,
2123 req.replica_count(),
2124 DECLARED_POOL,
2125 EMPTY_DECLARED_POOL,
2126 )
2127}
2128
2129const DECLARED_POOL: &str = "declared machines";
2130const EMPTY_DECLARED_POOL: &str = "(no machines declared under .yah/infra/machines/)";
2131
2132fn first_match<'a>(
2133 candidates: &[&'a MachineConfig],
2134 req: &RequiredSpec,
2135 pool: &str,
2136 empty_pool: &str,
2137) -> Result<&'a MachineConfig> {
2138 Ok(select_matching(candidates, req, 1, pool, empty_pool)?
2139 .into_iter()
2140 .next()
2141 .expect("select_matching errors rather than returning short"))
2142}
2143
2144/// The N-selecting core of the placement selector: the first `want` candidates
2145/// satisfying `req`, in candidate order (R844-F8).
2146///
2147/// [`first_match`] is this with `want = 1`, which is why widening a caller to a
2148/// replica count cannot introduce a second selector — the predicate, the
2149/// ordering and the failure vocabulary are all one implementation.
2150///
2151/// **A shortfall is an error.** Matching one machine when two were asked for
2152/// returns `Err` naming both numbers and the pool searched, never a one-element
2153/// vec: a half-placed workload that reports success is worse than a failed
2154/// apply, because the front door then publishes a hostname whose backend set is
2155/// quietly smaller than declared. `want = 0` is the same mistake spelled
2156/// differently and is refused for the same reason.
2157fn select_matching<'a>(
2158 candidates: &[&'a MachineConfig],
2159 req: &RequiredSpec,
2160 want: usize,
2161 pool: &str,
2162 empty_pool: &str,
2163) -> Result<Vec<&'a MachineConfig>> {
2164 let names = || {
2165 if candidates.is_empty() {
2166 empty_pool.to_string()
2167 } else {
2168 candidates
2169 .iter()
2170 .map(|m| m.name.as_str())
2171 .collect::<Vec<_>>()
2172 .join(", ")
2173 }
2174 };
2175
2176 if want == 0 {
2177 anyhow::bail!(
2178 "replicas = 0 places {} on nothing — a placement that deploys to no machine is \
2179 a typo, not a scale-down; remove the slot instead",
2180 req.describe()
2181 );
2182 }
2183
2184 let mut matched = matching(candidates, req);
2185 if matched.len() >= want {
2186 matched.truncate(want);
2187 return Ok(matched);
2188 }
2189
2190 if want == 1 {
2191 anyhow::bail!(
2192 "no candidates matching {} — {pool}: {}",
2193 req.describe(),
2194 names()
2195 );
2196 }
2197 anyhow::bail!(
2198 "only {} of {want} machines match {} — placing fewer than the declared \
2199 `replicas = {want}` would publish a smaller backend set than the mirror asks for; \
2200 {pool}: {}",
2201 matched.len(),
2202 req.describe(),
2203 names()
2204 )
2205}
2206
2207/// The predicate itself, applied to every candidate in order — the one place
2208/// `req.matches` is called on a set.
2209///
2210/// [`select_matching`] takes a prefix of this; [`CloudConfig::admit_workload_candidates`]
2211/// takes all of it. Keeping both on this function is what makes "the pool the
2212/// dispatcher failed over within" and "the machine admission picked" the same
2213/// answer by construction rather than by two filters that happen to agree.
2214fn matching<'a>(candidates: &[&'a MachineConfig], req: &RequiredSpec) -> Vec<&'a MachineConfig> {
2215 candidates
2216 .iter()
2217 .copied()
2218 .filter(|m| req.matches(m))
2219 .collect()
2220}
2221
2222/// The [`RequiredSpec`] a workload is admitted against — the single place the
2223/// axes are derived from a [`WorkloadSpec`].
2224///
2225/// Extracted from [`CloudConfig::admit_workload`] so that
2226/// [`CloudConfig::admit_workload_in_group`] narrows the candidate set without
2227/// restating the axes. Forking that derivation is how the two paths would
2228/// silently disagree about whether a workload fits a node.
2229///
2230/// # It admits a group, not a workload (R860-T4 / W338)
2231///
2232/// The axes come from [`placement_group`] — `ws` plus the transitive closure of
2233/// its `local` requirement edges — because those members are placed together or
2234/// not at all. Capacity is their **sum**, archetype repulsion their **union**,
2235/// and mesh tags their union too. `prefer-local` and `anywhere` edges bind
2236/// nothing: a spec with neither `requires` nor `depends_on` local edges has a
2237/// group of exactly itself and resolves byte-identically to the pre-R860 axes.
2238///
2239/// This is the **only** gate. Node election is CLI-side
2240/// (`MeshYubabaClient::elect_node`, which picks a live member of the pool this
2241/// produces); the yubaba node process accepts whatever it is handed and never
2242/// re-checks placement, so a wrong group here is not caught downstream.
2243///
2244/// @yah:ticket(R860-T4, "Admission: place the transitive closure of `local` edges as one group, not one workload")
2245/// @yah:status(review)
2246/// @yah:phase(P1)
2247/// @yah:at(2026-09-05T18:29:13Z)
2248/// @yah:assignee(agent:bundle-anthropic-ashguard)
2249/// @yah:parent(R860)
2250/// @yah:next("W338 §Placement consequences 1 and 2. `admission_spec()` (config.rs:1974-1998) derives its axes from ONE spec; it must derive them from the group — the transitive closure of `local` requirement edges over `effective_requirements()`. `prefer-local` and `anywhere` edges do NOT bind the group. Three consequences: memory/cpu floor becomes the SUM of the group's requests, not the requirer's alone; `repel_archetype` becomes the union over members (so a group containing an Appliance is repelled by `no-appliance` even if the requirer is a Server); and the group is non-drainable if ANY member is an Appliance, which today is a per-workload check at yubaba/src/lib.rs:3117-3128 and now has to be computed over a set.")
2251/// @yah:verify("cargo test -p cloud --lib config")
2252/// @yah:gotcha("Node election is CLI-side, not cluster-side: `MeshYubabaClient::elect_node` (app/yah/cli/src/yubaba_client.rs:235-268) calls `admit_workload_candidates` (config.rs:1753), picks one node, and POSTs the deploy there. The yubaba node process never decides placement — it accepts whatever it is handed. So group admission has to be right in `config.rs` because there is no second gate downstream to catch it.")
2253/// @arch:see(.yah/docs/working/W338-workload-dependencies-and-appliance-composition.md)
2254/// @yah:depends_on(R860-T1)
2255/// @yah:handoff("ADMISSION NOW PLACES A GROUP, NOT A WORKLOAD. `admission_spec` (oss/yubaba/crates/cloud/src/config.rs:2009) takes `(ws, declared: &[WorkloadConfig])` and derives every axis from `placement_group(ws, declared)` (:2108) — the transitive closure of `local` requirement edges over `effective_requirements()`, traversing `Requirement::provides` where present and resolving by ident against `cfg.workloads` (.yah/infra/workloads/) otherwise, mesh-identity first and workload name second. Capacity is the SUM of the members' `memory_request_mb()` / `resources.cpu_millis` (saturating). Only `local` binds: `prefer-local` and `anywhere` (which every legacy `depends_on` folds into) are skipped, so a spec without local edges has a group of exactly itself and its axes are bit-identical to the pre-R860 derivation.")
2256/// @yah:handoff("REPEL BECAME A SET. `RequiredSpec::repel_archetype: Option<LifecycleArchetype>` is now `repel_archetypes: Vec<LifecycleArchetype>` (config.rs:3878), the union over group members; `matches` (:3971) rejects a node carrying `no-<taint_key()>` for ANY of them, `describe` emits one `not-tainted(...)` part per archetype, `is_unconstrained` tests `is_empty()`. The field is `#[serde(skip)]`, so no wire or schema drift, and grep over app/ crates/ oss/ xtask/ finds no other referent of the old name and no `RequiredSpec { .. }` literal outside config.rs — the rename is contained. `admit_workload` / `admit_workload_candidates` / `admit_workload_in_group` signatures are unchanged; all three now pass `&self.workloads`.")
2257/// @yah:handoff("CYCLE GUARD, AND THE BUG IT TOOK TO GET RIGHT. The walker keeps TWO visited lists: `in_group` (member mesh identities) and `expanded` (requirement idents already resolved). The first version used one list and was silently wrong in the common case — a requirement's ident IS its provider's mesh identity, so marking the ident before resolving made every provider look already-present and `placement_group` returned a group of one. Five of the new tests caught it. If you refactor this, keep the two questions separate.")
2258/// @yah:handoff("ELECT_NODE NEEDS NO CHANGE FOR THIS TICKET — read it (app/yah/cli/src/yubaba_client.rs:235-268). It calls `admit_workload_candidates`, so it now receives a pool already filtered to nodes that can host the WHOLE group, then probes for liveness within it. That is correct for T4 because only the requirer is deployed today. It becomes load-bearing at R860-T6: `supply = \"self\"` provisioning MUST reuse the node URL `elect_node` returned for the requirer and must not re-elect per member — the probe is liveness-sensitive, so a second election can legally return a different member of the same pool and split the group across two nodes.")
2259/// @yah:handoff("DECISIONS THE BRIEF DID NOT COVER, all recorded in doc comments at the site. (1) `mesh_tags` are UNIONED over the group — the axis is already a superset/AND check, so a node that cannot host one member cannot host the group; zero regression risk since nothing in the tree declares `requires` yet. (2) `nodes` (the R833-F8 operator pin) stays REQUIRER-ONLY: it is a membership list, so intersecting two members' pins can yield an empty vec, which the axis reads as no-constraint — the exact inverse of the conflict. (3) `requires_taint` is a single Option: the requirer's wins, else the first member declaring one. Two members demanding DIFFERENT taints is not representable and would be an unplaceable group; widening that axis to a set is a follow-up if a real case appears. (4) An unresolvable `local` ident is SKIPPED, not an error — admission is a pure function of the declared inventory and must not start refusing deploys over a provider a later ticket declares; the cost is that its request does not count toward the floor, which is the exposure `depends_on` has always had.")
2260/// @yah:handoff("DRAINABILITY: placement half landed, node half deliberately NOT touched. `group_is_drainable(members)` (config.rs:2160) is the set-valued predicate W338 §Placement consequences 2 asks for — false as soon as any member is an Appliance — and the `no-appliance` repulsion that follows from it is enforced through `repel_archetypes`. The node-side loop `drain_workloads` (oss/yubaba/crates/yubaba/src/lib.rs, the R572-F4 archetype_registry skip) still decides per workload and knows nothing about requirement edges, so a Server bound to an Appliance by a `local` edge would still be drained alone. Not fixed here for two reasons: that file has three sessions live in it (the brief named them), and the fix needs group edges plumbed to the node process, which is R860-T6's rail rather than a local edit. `yubaba` already depends on `cloud`, so the predicate is directly callable from there when that plumbing exists.")
2261/// @yah:verify("BASELINE recorded before editing, tree anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2: `cargo test -p yah-cloud --lib` (from oss/yubaba) = 1081 passed, 0 failed, 4 ignored, exit 0. AFTER: 1090 passed, 0 failed, 4 ignored, exit 0 — +9, exactly the nine tests added. `cargo check -p yah-cloud --all-targets` exit 0, and `cargo check -p yubaba --all-targets` exit 0 as well (yubaba consumes `cloud`, so it is where the `repel_archetypes` rename would have surfaced). Every exit code echoed explicitly, never inferred from an empty grep.")
2262/// @yah:verify("NEW TESTS (config.rs `mod tests`, R860-T4 section at the end): a_local_edge_binds_the_provider_into_the_placement_group; prefer_local_and_anywhere_edges_do_not_bind_the_group (covers a legacy `depends_on` too); the_group_is_the_transitive_closure_and_traverses_inline_provides; an_ident_cycle_closes_the_group_instead_of_looping_forever; an_unresolvable_local_ident_is_skipped_rather_than_refused; the_capacity_floor_is_the_sum_of_the_group_not_the_requirer_alone (a 300 MiB node refuses two 256 MiB members and the error names memory_mb>=512; a 512 MiB node admits); a_server_requiring_an_appliance_locally_is_repelled_by_no_appliance (same requirer alone still lands on the tainted Pi, so the repulsion provably comes from the edge); a_group_containing_an_appliance_is_not_drainable; a_spec_with_no_local_edges_admits_exactly_as_it_did_before.")
2263/// @yah:gotcha("The camp's `yah build run` rail killed three consecutive verification runs against the shared oss/yubaba/target dir: each ended with only `Blocking waiting for file lock on build directory` in the log and no exit code, after 121s / 720s. The green result above was obtained with `CARGO_TARGET_DIR=/tmp/r860t4-target`, which sidesteps the contended lock at the cost of one cold dep build. Worth reaching for directly when the yubaba target dir is busy rather than burning three cycles discovering it.")
2264/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
2265/// @yah:next("R860-T6 (supply = \"self\"): deploy the group's non-requirer members onto the node `elect_node` already returned for the requirer — do NOT re-elect per member, or a liveness probe can split the group across two nodes. `placement_group` (config.rs:2108) hands you the member specs in traversal order, requirer first.")
2266/// @yah:next("Node-side drain is still per-workload: teach `drain_workloads` (oss/yubaba/crates/yubaba/src/lib.rs) to consult `cloud::config::group_is_drainable` over the requesting workload's placement group once R860-T6 plumbs group membership to the node. Left untouched here on purpose — three sessions were live in that file.")
2267/// @yah:handoff("LEADER RE-VERIFIED (session:69b18855, independent of the courier's self-report). `cargo test -p yah-cloud --lib` from oss/yubaba: 1090 passed / 0 failed / 4 ignored, exit 0, against the courier's recorded 1081/0/4 baseline — +9 = exactly its new tests. Confirmed by content in config.rs: `placement_group` :2120 with the `req.locality != Locality::Local` guard at :2136 (so `prefer-local` and `anywhere` correctly do NOT bind), `group_is_drainable` :2172, and `RequiredSpec::repel_archetype: Option<_>` widened to `repel_archetypes: Vec<_>` at :3890 with the union built at :2039-2050 and enforced at :4004/:4044. The repel rename is `#[serde(skip)]`, so no wire or schema drift.")
2268/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
2269/// @yah:verify("cargo test -p yah-cloud --lib (from oss/yubaba): 1090 passed / 0 failed / 4 ignored, exit 0, vs a 1081/0/4 baseline. Exit codes echoed explicitly throughout rather than inferred from an empty grep — the trap that cost R860-T1 three misses.")
2270/// @yah:gotcha("CORRECTION FROM R860-T6, and the leader propagated the error so it is worth naming: this ticket's handoff asserted \\\"`yubaba` already depends on `cloud`, so the predicate is directly callable from there\\\". THAT IS WRONG. `cloud` is a DEV-dependency of yubaba only — oss/yubaba/crates/yubaba/Cargo.toml:150-152, under the comment \\\"Integration test harness\\\" — and cloud's own Cargo.toml records that the runtime yubaba→cloud edge was DELIBERATELY avoided from R374-F3 onward. The leader repeated the claim verbatim in R860-T6's dispatch brief; T6's courier checked it against the manifest instead of trusting it, which is the only reason it did not become a runtime dependency inversion. Resolution: `group_is_drainable`'s body moved down to `workload_spec::group_is_drainable` (workload-spec/src/lib.rs:2365), the shared home both crates already depend on, and `cloud::config::group_is_drainable` (config.rs:2240) now delegates to it keeping its signature. Verified after the move: yah-cloud still 1093/0/4, yah-workload-spec 171+98/0.")
2271/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
2272/// @yah:verify("RE-VERIFIED AT HEAD 00ee20d1 (session:aa5e882d, 2026-09-05). `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` = 1141 passed / 0 failed / 4 ignored, exit 0 (was 1093 at the first leader's check, 1110 at the second; the deltas are peers' tests). Group placement confirmed by content in oss/yubaba/crates/cloud/src/config.rs: `placement_group` derivation at :2073/:2108, `repel_archetypes: Vec<LifecycleArchetype>` at :3878. NOTE FOR ANYONE RE-RUNNING THIS: `cargo test -p yah-cloud --lib` from the repo root FAILS with \"package `yah-cloud` cannot be tested because it requires dev-dependencies and is not a member of the workspace\" — yah-cloud lives in the oss/yubaba workspace, so the invocation needs `--manifest-path oss/yubaba/Cargo.toml`.")
2273fn admission_spec(ws: &WorkloadSpec, declared: &[WorkloadConfig]) -> RequiredSpec {
2274 let group = placement_group(ws, declared);
2275
2276 // Capacity is the group's demand, not the requirer's (W338 §Placement
2277 // consequences 1). Saturating rather than wrapping: an absurd declared
2278 // request must read as "nothing is big enough", never as a small number.
2279 //
2280 // `memory_request_mb()` and NOT `resources.memory_mb`: the latter is a
2281 // cgroup ceiling, and reading a ceiling as a floor made `for_forge`'s
2282 // deliberately-roomy 32 GiB limit mean "only place me on a 32 GiB node".
2283 // That excluded every build-worker in the fleet but one. The accessor falls
2284 // back to `resources.memory_mb` when no request is declared, so specs that
2285 // never set one are admitted exactly as before.
2286 let mut memory_mb: u32 = 0;
2287 let mut cpu_millis: u32 = 0;
2288 // R572-F5 taint repulsion, unioned over the group (W338 §Placement
2289 // consequences 2): a group is non-drainable — and `no-appliance`-repelled —
2290 // if *any* member is an Appliance, even when the requirer is a Server.
2291 //
2292 // R876-B7 inverted the sense. Repulsion is now unconditional in `matches`,
2293 // so what this loop collects is still the group's archetype union, but it is
2294 // converted below into the complementary TOLERATION set. Same predicate,
2295 // stated from the other side.
2296 let mut group_archetypes: Vec<LifecycleArchetype> = Vec::new();
2297 // Mesh tags are already AND-ed (a machine must be a superset), so unioning
2298 // them over the group is the same predicate applied to every member: a node
2299 // that cannot host one member cannot host the group.
2300 let mut mesh_tags = node_selector_mesh_tags(ws);
2301
2302 for member in &group {
2303 memory_mb = memory_mb.saturating_add(member.memory_request_mb());
2304 cpu_millis = cpu_millis.saturating_add(member.resources.cpu_millis);
2305 let arch = member.effective_archetype();
2306 if !group_archetypes.contains(&arch) {
2307 group_archetypes.push(arch);
2308 }
2309 for tag in node_selector_mesh_tags(member) {
2310 if !mesh_tags.contains(&tag) {
2311 mesh_tags.push(tag);
2312 }
2313 }
2314 }
2315
2316 // R860-T5 / W338 §Placement consequences 3: per-node native-exec
2317 // capability. Computed over the group for the same reason every other axis
2318 // is — a `local` edge to a native provider makes the *requirer* unplaceable
2319 // on a node without the backend, even when the requirer is an ordinary
2320 // container workload. This is the `supply = "self"` precondition W338 names:
2321 // a self-supplied native provider has to be placeable where its requirer
2322 // lands, and until now nothing upstream could see whether it was.
2323 //
2324 // Appended to `mesh_tags` rather than given its own field: the axis is
2325 // already an AND-ed superset check against `machine.mesh_tags`, `describe`
2326 // already renders it, and `RequiredSpec` needs no new shape. See
2327 // [`NATIVE_EXEC_MESH_TAG`] for why a tag and not a taint.
2328 if group.iter().any(WorkloadSpec::wants_native_exec)
2329 && !mesh_tags.iter().any(|t| t == NATIVE_EXEC_MESH_TAG)
2330 {
2331 mesh_tags.push(NATIVE_EXEC_MESH_TAG.to_string());
2332 }
2333
2334 RequiredSpec {
2335 mesh_tags,
2336 // R833-F8: imperative node pin. Derived here alongside the inferred
2337 // mesh tags rather than short-circuiting the resolver, so a pinned
2338 // workload is still checked against capacity and taints.
2339 //
2340 // Requirer-only on purpose: the pin is what the operator typed on
2341 // *this* deploy, and `nodes` is a membership list, so intersecting two
2342 // members' pins could yield an empty vec — which this axis reads as "no
2343 // constraint", i.e. the exact opposite of the conflict it represents.
2344 nodes: node_selector_node(ws).into_iter().collect(),
2345 memory_mb,
2346 cpu_millis,
2347 // R876-B7: the archetype union, restated as tolerations — every
2348 // repelling key that is NOT this group's own class. A Server group
2349 // tolerates `no-appliance` and `no-job` and is still blocked by
2350 // `no-server`, which is precisely what the pre-B7 `repel_archetypes`
2351 // axis computed. That equivalence is the migration: the `admit_workload`
2352 // path's placement answers are unchanged for every fleet machine, while
2353 // the mirror-declared path — which could never populate an archetype set
2354 // and so read no taints at all — becomes repel-by-default.
2355 //
2356 // Derived from `LifecycleArchetype::ALL` rather than a literal list, so
2357 // a fourth archetype is tolerated by unrelated groups automatically,
2358 // exactly as `taint_effect` already derives the repulsion half.
2359 tolerates: LifecycleArchetype::ALL
2360 .into_iter()
2361 .filter(|a| !group_archetypes.contains(a))
2362 .map(|a| format!("no-{}", a.taint_key()))
2363 .collect(),
2364 // R572-F5: taint affinity from the requires-taint annotation. The
2365 // requirer's wins; otherwise the first member that declares one, since
2366 // the group shares a node and this axis holds a single key. Two members
2367 // demanding *different* taints is not representable here and would be
2368 // an unplaceable group anyway — see the R860-T4 handoff.
2369 requires_taint: group
2370 .iter()
2371 .find_map(|m| m.requires_taint().map(str::to_owned)),
2372 ..Default::default()
2373 }
2374}
2375
2376/// The workloads that must be placed together with `ws`: the transitive closure
2377/// of `local` requirement edges over [`WorkloadSpec::effective_requirements`],
2378/// starting at the requirer (R860-T4 / W338 §"Each member keeps its own mesh
2379/// identity").
2380///
2381/// **Only `local` binds.** `prefer-local` explicitly "never blocks placement"
2382/// (W338's locality table) and `anywhere` is an ordinary service dependency —
2383/// treating either as a co-scheduling constraint would turn every `depends_on`
2384/// in the tree into one, since the legacy field folds in as `anywhere` + `wait`.
2385///
2386/// A group is **not** a new addressable object: every member keeps its own mesh
2387/// identity, spec and healthcheck (W338). This function returns the members'
2388/// specs so admission can take the sum / union over them, and nothing here
2389/// deploys, provisions or tears anything down — `supply = "self"` provisioning
2390/// is R860-T6 and per-node native-exec capability is R860-T5.
2391///
2392/// Two ways a member is reached, in this order:
2393/// - [`Requirement::provides`], the inline spec a `supply = "self"` requirement
2394/// carries;
2395/// - otherwise an ident lookup against `declared` (`.yah/infra/workloads/`),
2396/// matched on mesh identity first and on workload name second, because those
2397/// coincide for every spec in the tree today but the requirement is written in
2398/// the mesh-identity currency.
2399///
2400/// An ident that resolves to neither is **skipped**, not an error: admission is
2401/// a pure function of the declared inventory and must not start failing deploys
2402/// over a provider that a not-yet-written ticket will declare. The cost is that
2403/// its request does not count toward the floor, which is the same exposure
2404/// `depends_on` has always had.
2405///
2406/// **Cycle-guarded.** `validate::check_requires` bounds `provides` *nesting* to
2407/// depth 1 but nothing stops two separately-declared specs from requiring each
2408/// other, and this closure would otherwise not terminate. Each requirement ident
2409/// is resolved at most once and each member joins the group at most once, so a
2410/// cycle simply closes the group.
2411pub fn placement_group(ws: &WorkloadSpec, declared: &[WorkloadConfig]) -> Vec<WorkloadSpec> {
2412 let mut members = vec![ws.clone()];
2413 // Two separate visited sets, because the two questions differ: `in_group`
2414 // stops a workload being added twice, `expanded` stops an ident being
2415 // resolved twice. Folding them into one list makes the ident of a member
2416 // already in the group indistinguishable from the member itself — and since
2417 // a requirement's ident *is* its provider's mesh identity, that reads every
2418 // provider as already-present and silently returns a group of one.
2419 let mut in_group: Vec<String> = vec![group_key(ws)];
2420 let mut expanded: Vec<String> = Vec::new();
2421 let mut next = 0;
2422
2423 while next < members.len() {
2424 let requirements = members[next].effective_requirements();
2425 next += 1;
2426 for req in requirements {
2427 if req.locality != Locality::Local {
2428 continue;
2429 }
2430 if expanded.contains(&req.ident.0) {
2431 continue;
2432 }
2433 expanded.push(req.ident.0.clone());
2434
2435 let provider = match req.provides.as_deref() {
2436 Some(spec) => spec.clone(),
2437 None => match resolve_requirement_ident(&req.ident, declared) {
2438 Some(spec) => spec,
2439 None => continue,
2440 },
2441 };
2442 let key = group_key(&provider);
2443 if in_group.contains(&key) {
2444 continue;
2445 }
2446 in_group.push(key);
2447 members.push(provider);
2448 }
2449 }
2450
2451 members
2452}
2453
2454/// Whether a placement group may be drained off its node (W338 §Placement
2455/// consequences 2): false as soon as **any** member is an Appliance.
2456///
2457/// The set-valued form of the per-workload check the node itself makes in
2458/// `drain_workloads` (`oss/yubaba/crates/yubaba/src/lib.rs`), which skips an
2459/// Appliance by its own archetype and knows nothing about requirement edges. A
2460/// `Server` bound to an Appliance by a `local` edge has to move with it or not
2461/// at all, so draining it alone breaks the group the same way placing it alone
2462/// would.
2463///
2464/// R860-T6 moved the body to [`workload_spec::group_is_drainable`] and left this
2465/// signature untouched. The node's `drain_workloads` needs the identical
2466/// predicate, and yubaba has no runtime dependency on this crate by design
2467/// (R374-F3) — so the one implementation now lives in the crate both sides
2468/// already depend on, rather than being copied into the second caller.
2469pub fn group_is_drainable(members: &[WorkloadSpec]) -> bool {
2470 workload_spec::group_is_drainable(members)
2471}
2472
2473/// Identity a placement-group member is deduplicated by — its mesh identity,
2474/// which is the currency [`Requirement::ident`] is written in.
2475fn group_key(ws: &WorkloadSpec) -> String {
2476 ws.expose.mesh.identity.0.clone()
2477}
2478
2479/// Resolve a requirement's ident to a separately-declared spec: mesh identity
2480/// first, workload file name second.
2481fn resolve_requirement_ident(
2482 ident: &workload_spec::MeshIdent,
2483 declared: &[WorkloadConfig],
2484) -> Option<WorkloadSpec> {
2485 declared
2486 .iter()
2487 .find(|w| w.spec.expose.mesh.identity == *ident)
2488 .or_else(|| declared.iter().find(|w| w.spec.name == ident.0))
2489 .map(|w| w.spec.clone())
2490}
2491
2492/// The mesh tag a node declares to advertise that its kamaji can run **native**
2493/// (fork+exec) workloads — R860-T5 / W338 §"Placement consequences" 3.
2494///
2495/// A workload marked `yah.exec = native` ([`WorkloadSpec::wants_native_exec`])
2496/// is not containerized: kamaji fork+execs it on the node's own userland. That
2497/// backend only exists when the node's kamaji was **built** with the
2498/// `native-exec` cargo feature and **started** with `--native-exec-dir`
2499/// (`oss/kamaji/crates/kamaji-bin/src/main.rs`). Both are node-local startup
2500/// decisions, invisible to everything upstream — so before this tag, placement
2501/// happily elected a node whose kamaji then refused the deploy with
2502/// `BackendRefused: ... no native backend is available (native backend not
2503/// configured — start kamaji with --native-exec-dir)`. That is exactly how the
2504/// mesh lost its coordination server for 25 hours on 2026-09-03 (R858: raft
2505/// leadership moved headscale, a native workload, to `us-south-001`, which has
2506/// no such kamaji). [`admission_spec`] now requires this tag whenever any
2507/// placement-group member is native, which turns that dispatch-time surprise
2508/// into a placement precondition.
2509///
2510/// # Why a mesh tag and not a taint
2511///
2512/// The two vocabularies on [`MachineConfig`] mean opposite things. `mesh_tags`
2513/// are **positive capability** matched as a superset — "this node CAN" — which
2514/// is precisely the claim being made, and an extra tag on a machine can only
2515/// ever make it match *more* requirement sets, so declaring it is regression-
2516/// free. `taints` are **repulsion** — "keep this class off" — and would have to
2517/// be inverted (`no-native-exec` on every node lacking the backend, i.e. the
2518/// declaration burden falls on the majority) *and* taught to
2519/// [`taint_effect`], or [`crate::validate::check_inert_taints`] would correctly
2520/// lint the key dead.
2521///
2522/// # The `cap:` namespace
2523///
2524/// New here. The live prefixes are `tag:` (role — `tag:build-worker`,
2525/// `tag:qed`, `tag:cloud-runner`), `arch:` and `os:` (facts about the silicon
2526/// and userland), and `tier:` is reserved for the environment axis (R763, see
2527/// [`crate::validate::check_retired_arch_tags`]). A *capability the daemon was
2528/// configured with* is none of those: it is not a role an operator assigns and
2529/// not a property of the hardware, it is a fact about how kamaji was started,
2530/// and it changes when the node is rolled. Nothing validates tag prefixes, so
2531/// this costs no wiring.
2532///
2533/// # Fails closed
2534///
2535/// A node that does not declare it is not a candidate. An undeclared fleet
2536/// therefore reports "no node admits" at election time rather than dispatching
2537/// to a node that will refuse — the refusal moves earlier and names the
2538/// constraint, which is the whole point. Declared today (from readings recorded
2539/// in-repo, not inferred) on `us-west-001` and `us-west-003`; see those
2540/// machines' TOMLs for the evidence and the date.
2541pub const NATIVE_EXEC_MESH_TAG: &str = "cap:native-exec";
2542
2543/// Parse the R594 mesh-tag node-selector off a workload's annotations into the
2544/// requested tag set. Absent annotation or empty value ⇒ empty vec ("no
2545/// constraint"). Whitespace around each comma-separated tag is trimmed and
2546/// empty segments are dropped, so `"tag:build-worker, arch:x86"` and
2547/// `"tag:build-worker,arch:x86"` parse identically.
2548pub fn node_selector_mesh_tags(ws: &WorkloadSpec) -> Vec<String> {
2549 ws.annotations
2550 .get(velveteen_exec::remote::NODE_SELECTOR_MESH_TAGS_ANNOTATION)
2551 .map(|v| {
2552 v.split(',')
2553 .map(str::trim)
2554 .filter(|s| !s.is_empty())
2555 .map(String::from)
2556 .collect()
2557 })
2558 .unwrap_or_default()
2559}
2560
2561/// Parse the R833-F8 imperative node-selector off a workload's annotations —
2562/// the single machine `name` the operator pinned the run to
2563/// (`--where=node:us-west-003`). Absent or blank ⇒ `None` ("no constraint"),
2564/// which is every workload built before this axis existed.
2565///
2566/// One node, not a list: the annotation exists to express "run it *there*", and
2567/// a comma-joined set would be a worse spelling of the mesh-tag selector that
2568/// already handles "any of these".
2569pub fn node_selector_node(ws: &WorkloadSpec) -> Option<String> {
2570 ws.annotations
2571 .get(velveteen_exec::remote::NODE_SELECTOR_NODE_ANNOTATION)
2572 .map(|v| v.trim())
2573 .filter(|v| !v.is_empty())
2574 .map(String::from)
2575}
2576
2577/// Load every `.yah/infra/providers/*.toml` into a [`ProviderConfig`] list.
2578/// Missing directory → empty list.
2579fn load_providers(dir: &Path) -> Result<Vec<ProviderConfig>> {
2580 if !dir.exists() {
2581 return Ok(vec![]);
2582 }
2583 let mut items = vec![];
2584 let mut entries: Vec<_> = std::fs::read_dir(dir)
2585 .with_context(|| format!("reading {}", dir.display()))?
2586 .filter_map(|e| e.ok())
2587 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
2588 .collect();
2589 entries.sort_by_key(|e| e.file_name());
2590 for entry in entries {
2591 items.push(ProviderConfig::load(&entry.path())?);
2592 }
2593 Ok(items)
2594}
2595
2596/// Map legacy mirror file stems to their canonical tier names.
2597///
2598/// Canonical tiers: `dev` / `pond` / `cloud` / `ha`.
2599/// Legacy stems pre-R362: `local` (dev tier), `local-sim` / `sim` (pond tier), `prod` (cloud tier).
2600/// Both forms are accepted; canonical names are preferred for new files.
2601pub fn canonical_tier(stem: &str) -> &str {
2602 match stem {
2603 "local" => "dev",
2604 "local-sim" | "sim" => "pond",
2605 "prod" => "cloud",
2606 other => other,
2607 }
2608}
2609
2610/// Walk `.yah/services/<svc>/` for every service and its mirrors.
2611/// Missing directory → empty map. Mirror file stems are normalized to canonical
2612/// tier names via [`canonical_tier`] so callers always see `dev/pond/cloud/ha`.
2613fn load_services(
2614 dir: &Path,
2615 workspace_root: &Path,
2616) -> Result<BTreeMap<String, ServiceWithMirrors>> {
2617 if !dir.exists() {
2618 return Ok(BTreeMap::new());
2619 }
2620 let mut out = BTreeMap::new();
2621 let mut entries: Vec<_> = std::fs::read_dir(dir)
2622 .with_context(|| format!("reading {}", dir.display()))?
2623 .filter_map(|e| e.ok())
2624 .filter(|e| e.path().is_dir())
2625 .collect();
2626 entries.sort_by_key(|e| e.file_name());
2627
2628 for entry in entries {
2629 let svc_dir = entry.path();
2630 let service_toml = svc_dir.join("service.toml");
2631 if !service_toml.exists() {
2632 // Skip directories without a service.toml — leaves room for
2633 // future siblings (e.g. `secrets/`, `README.md`) without
2634 // triggering false-positive parse errors.
2635 continue;
2636 }
2637 let service = ServiceConfig::load(&service_toml)?;
2638 let mut mirrors = BTreeMap::new();
2639 let mirrors_dir = svc_dir.join("mirrors");
2640 if mirrors_dir.exists() {
2641 let mut menv: Vec<_> = std::fs::read_dir(&mirrors_dir)
2642 .with_context(|| format!("reading {}", mirrors_dir.display()))?
2643 .filter_map(|e| e.ok())
2644 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
2645 .collect();
2646 menv.sort_by_key(|e| e.file_name());
2647 for m in menv {
2648 let path = m.path();
2649 let stem = path
2650 .file_stem()
2651 .and_then(|s| s.to_str())
2652 .unwrap_or("")
2653 .to_string();
2654 let tier = canonical_tier(&stem).to_string();
2655 // Last-write wins if both legacy and canonical forms coexist
2656 // (e.g. local-sim.toml + pond.toml). Sort order ensures the
2657 // canonical file (pond.toml) wins because 'p' > 'l'.
2658 mirrors.insert(tier, MirrorConfig::load(&path)?);
2659 }
2660 }
2661 let mut component_transform_recipes = BTreeMap::new();
2662 for component in &service.components {
2663 if component.kind == "static-asset" {
2664 if let Some(recipe) =
2665 read_component_transform_recipe(workspace_root, &component.path)
2666 {
2667 component_transform_recipes.insert(component.id.clone(), recipe);
2668 }
2669 }
2670 }
2671 let passway_machines = mirrors
2672 .iter()
2673 .filter_map(|(env, m)| m.passway_machines().map(|ms| (env.clone(), ms)))
2674 .collect();
2675 out.insert(
2676 service.name.clone(),
2677 ServiceWithMirrors {
2678 service,
2679 mirrors,
2680 component_transform_recipes,
2681 passway_machines,
2682 },
2683 );
2684 }
2685 Ok(out)
2686}
2687
2688/// Read the first transform recipe name from a component's `workload.toml`.
2689/// Returns `None` when the file is absent or has no `[asset.derive.transform]`
2690/// section. Best-effort — parse failures are silently ignored so a malformed
2691/// workload.toml doesn't abort the entire service catalog load.
2692fn read_component_transform_recipe(workspace_root: &Path, component_path: &str) -> Option<String> {
2693 let workload_path = workspace_root.join(component_path).join("workload.toml");
2694 let text = std::fs::read_to_string(&workload_path).ok()?;
2695 let value: toml::Value = toml::from_str(&text).ok()?;
2696 let assets = value.get("asset")?.as_array()?;
2697 for asset in assets {
2698 if let Some(recipe) = asset
2699 .get("derive")
2700 .and_then(|d| d.get("transform"))
2701 .and_then(|t| t.get("recipe"))
2702 .and_then(|r| r.as_str())
2703 {
2704 return Some(recipe.to_string());
2705 }
2706 }
2707 None
2708}
2709
2710/// Load every `.yah/domains/*.toml` into a [`DomainConfig`] map keyed by
2711/// file stem. Missing directory → empty map.
2712fn load_domains(dir: &Path) -> Result<BTreeMap<String, DomainConfig>> {
2713 if !dir.exists() {
2714 return Ok(BTreeMap::new());
2715 }
2716 let mut out = BTreeMap::new();
2717 let mut entries: Vec<_> = std::fs::read_dir(dir)
2718 .with_context(|| format!("reading {}", dir.display()))?
2719 .filter_map(|e| e.ok())
2720 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
2721 .collect();
2722 entries.sort_by_key(|e| e.file_name());
2723 for entry in entries {
2724 let path = entry.path();
2725 let stem = path
2726 .file_stem()
2727 .and_then(|s| s.to_str())
2728 .unwrap_or("")
2729 .to_string();
2730 let dom = DomainConfig::load(&path)?;
2731 if dom.name != stem {
2732 anyhow::bail!(
2733 "domains/{}.toml: name = \"{}\" must match the file stem",
2734 stem,
2735 dom.name
2736 );
2737 }
2738 out.insert(dom.name.clone(), dom);
2739 }
2740 Ok(out)
2741}
2742
2743/// Load and shape-validate all `*.toml` files in `dir` as [`WorkloadConfig`].
2744fn load_workloads(dir: std::path::PathBuf) -> Result<Vec<WorkloadConfig>> {
2745 if !dir.exists() {
2746 return Ok(vec![]);
2747 }
2748 let mut items = vec![];
2749 let mut entries: Vec<_> = std::fs::read_dir(&dir)
2750 .with_context(|| format!("reading {}", dir.display()))?
2751 .filter_map(|e| e.ok())
2752 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
2753 .collect();
2754 entries.sort_by_key(|e| e.file_name());
2755
2756 for entry in entries {
2757 let path = entry.path();
2758 let path_str = path.display().to_string();
2759 let src =
2760 std::fs::read_to_string(&path).with_context(|| format!("reading {}", path_str))?;
2761 let spec: WorkloadSpec =
2762 toml::from_str(&src).with_context(|| format!("parsing {}", path_str))?;
2763
2764 // Shape-validate before accepting into the loaded config.
2765 validate::shape(&spec)
2766 .map_err(|e| anyhow::anyhow!("workload {} failed shape validation: {e}", path_str))?;
2767
2768 items.push(WorkloadConfig { spec });
2769 }
2770 Ok(items)
2771}
2772
2773/// Load all mirror configs from the `mirrors/` directory.
2774///
2775/// Handles two layouts that may coexist:
2776/// - **Folder**: `mirrors/<id>/mirror.toml` — preferred; allows secrets and
2777/// per-mirror overrides to live next to the config file.
2778/// - **Flat**: `mirrors/<id>.toml` — legacy; still supported.
2779///
2780/// Each file is parsed as [`LegacyMirrorConfig`]. A malformed file returns an error
2781/// that includes the file path and the TOML field path + line/column, so the
2782/// caller can surface it to the user directly.
2783fn load_mirrors(dir: std::path::PathBuf) -> Result<Vec<LegacyMirrorConfig>> {
2784 if !dir.exists() {
2785 return Ok(vec![]);
2786 }
2787 let mut mirrors = vec![];
2788 let mut entries: Vec<_> = std::fs::read_dir(&dir)
2789 .with_context(|| format!("reading {}", dir.display()))?
2790 .filter_map(|e| e.ok())
2791 .collect();
2792 entries.sort_by_key(|e| e.file_name());
2793
2794 for entry in entries {
2795 let path = entry.path();
2796 if path.is_dir() {
2797 // Folder layout: mirrors/<id>/mirror.toml
2798 let mirror_toml = path.join("mirror.toml");
2799 if mirror_toml.exists() {
2800 let src = std::fs::read_to_string(&mirror_toml)
2801 .with_context(|| format!("reading {}", mirror_toml.display()))?;
2802 let cfg: LegacyMirrorConfig = toml::from_str(&src)
2803 .with_context(|| format!("parsing {}", mirror_toml.display()))?;
2804 mirrors.push(cfg);
2805 }
2806 } else if path.extension().map_or(false, |e| e == "toml") {
2807 // Flat layout: mirrors/<id>.toml
2808 let src = std::fs::read_to_string(&path)
2809 .with_context(|| format!("reading {}", path.display()))?;
2810 let cfg: LegacyMirrorConfig =
2811 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
2812 mirrors.push(cfg);
2813 }
2814 }
2815 Ok(mirrors)
2816}
2817
2818/// Load `topology.toml` if it exists; return a default (empty) topology otherwise.
2819fn load_topology(path: std::path::PathBuf) -> Result<TopologyConfig> {
2820 if !path.exists() {
2821 return Ok(TopologyConfig::default());
2822 }
2823 let src =
2824 std::fs::read_to_string(&path).with_context(|| format!("reading {}", path.display()))?;
2825 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
2826}
2827
2828/// R555-S1: entries are sorted by file name before parsing, so "declaration
2829/// order in `.yah/infra/machines/` breaks ties" — the contract
2830/// [`CloudConfig::admit_workload`] documents — is actually true. `read_dir`
2831/// yields filesystem order, which is unspecified and differs between APFS and
2832/// a hashed-dir ext4; without the sort, *which* of two equally-matching nodes a
2833/// workload admits to could change when an unrelated file is added to the
2834/// directory. That was latent while each tag set had one match and became
2835/// observable the day us-west-003 joined us-west-002 on
2836/// `[tag:build-worker, arch:x86, os:linux]`. Same sort `load_providers` has
2837/// always done.
2838fn load_dir<T: for<'de> Deserialize<'de>>(dir: std::path::PathBuf) -> Result<Vec<T>> {
2839 if !dir.exists() {
2840 return Ok(vec![]);
2841 }
2842 let mut entries: Vec<_> = std::fs::read_dir(&dir)
2843 .with_context(|| format!("reading {}", dir.display()))?
2844 .collect::<std::io::Result<Vec<_>>>()
2845 .with_context(|| format!("reading {}", dir.display()))?;
2846 entries.sort_by_key(|e| e.file_name());
2847
2848 let mut items = vec![];
2849 for entry in entries {
2850 let path = entry.path();
2851 if path.extension().map_or(false, |e| e == "toml") {
2852 let src = std::fs::read_to_string(&path)
2853 .with_context(|| format!("reading {}", path.display()))?;
2854 let item: T =
2855 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
2856 items.push(item);
2857 }
2858 }
2859 Ok(items)
2860}
2861
2862// ─── New manifest shapes (R222 B2) ───────────────────────────────────────────
2863//
2864// The post-R215 layout splits substrate from service declarations:
2865//
2866// .yah/infra/providers/<id>.toml → ProviderConfig
2867// .yah/services/<svc>/service.toml → ServiceConfig
2868// .yah/services/<svc>/mirrors/<env>.toml → MirrorConfig
2869//
2870// CloudConfig::load still reads the legacy layout — B3 swaps in these types
2871// and removes the Legacy* shapes plus TopologyConfig.
2872
2873/// Tag for the infrastructure provider kind. Drives which fields are valid in
2874/// a [`ProviderConfig`] body or a [`MirrorProviderSlot::Inline`] block.
2875///
2876/// Two flavors:
2877/// - **Account/runtime providers** (`cloudflare`, `hetzner`, `local-container`)
2878/// live as files under `.yah/infra/providers/<id>.toml` and are referenced
2879/// from a mirror via `use = "<id>"`.
2880/// - **Inline-only providers** (`local-static`, `miniflare-container`,
2881/// `minio-container`) declare an operator-local stand-in directly inside a
2882/// mirror via `kind = "..."`. They carry no credentials and have no provider
2883/// file. The container-backed kinds ride on top of whichever
2884/// `local-container` runtime is declared in infra (orbstack/colima/docker);
2885/// the reconciler resolves the runtime at up-time.
2886#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
2887#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
2888#[serde(rename_all = "kebab-case")]
2889pub enum Provider {
2890 /// Cloudflare account: R2 buckets, DNS, Workers, Tunnels.
2891 Cloudflare,
2892 /// Hetzner Cloud + Object Storage account.
2893 Hetzner,
2894 /// Vultr cloud VPS — auto-provisioned via the `cloud.vps.*` Envoy
2895 /// (`VultrEnvoy`), the burst/scaling counterpart to Hetzner. Driver-backed.
2896 Vultr,
2897 /// BYO bare/static node (OVH, on-prem, anything we did NOT provision via a
2898 /// cloud API). Brought up over SSH (`stand-up-yubaba.sh` / `yah cloud
2899 /// machine bootstrap`); reach is declared in the machine's `[connect]`
2900 /// block. No create/destroy driver — placement-only.
2901 Static,
2902 /// Built-in static-file server bound to localhost. Inline-only; never
2903 /// declared as a standalone provider file because it carries no creds.
2904 LocalStatic,
2905 /// Local container runtime (orbstack/colima/docker). Configured by a
2906 /// provider file under `.yah/infra/providers/` so the discovery hints +
2907 /// runtime override sit in one place.
2908 LocalContainer,
2909 /// Dev-tier compute: the component runs as a kamaji-supervised host
2910 /// process against the operator's real workspace, no container and no
2911 /// build step per edit. Inline-only — it carries no credentials, and
2912 /// "the machine you are sitting at" is not an account to point at.
2913 /// See `reconciler::local_process`.
2914 LocalProcess,
2915 /// Containerized miniflare (workerd subprocess) fronting MinIO — the
2916 /// pond-tier stand-in for a CF Worker + R2 static surface. Inline-only;
2917 /// the reconciler spawns miniflare via the JS runtime and starts a MinIO
2918 /// container on the local-container runtime.
2919 MiniflareContainer,
2920 /// Containerized MinIO providing an S3-compatible API — the pond-tier
2921 /// stand-in for Cloudflare R2. Inline-only; the reconciler spins up the
2922 /// container on the local-container runtime and auto-creates the declared
2923 /// bucket on first up.
2924 MinioContainer,
2925 /// Dev-tier PostgreSQL — a real server speaking real pgwire on loopback,
2926 /// supervised by kamaji as the `yah-pg-dev` workload (W265, R584-F1). No
2927 /// docker daemon: the driver fetches a per-arch PostgreSQL tarball on first
2928 /// run and `initdb`s a cluster under `.yah/infra/state/dev/pg/`.
2929 ///
2930 /// Inline-only — it carries no credentials worth a provider file (the
2931 /// cluster is loopback-bound with a fixed dev password). Declared under
2932 /// [`MirrorConfig::drivers`], not `providers`:
2933 ///
2934 /// ```toml
2935 /// [drivers.pg]
2936 /// kind = "local-pg-dev"
2937 /// ```
2938 LocalPgDev,
2939}
2940
2941/// A provider account/runtime binding from `.yah/infra/providers/<id>.toml`.
2942///
2943/// The `kind` discriminator picks the schema for the remaining fields. Strict
2944/// on `kind` (unknown values are a parse error); permissive on per-kind fields
2945/// (carried as a free-form map so this loader stays stable as new fields land).
2946/// B3/B4 will tighten by introducing typed variants alongside JSON Schema.
2947#[derive(Debug, Clone, Serialize, Deserialize)]
2948#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
2949pub struct ProviderConfig {
2950 pub schema_version: u32,
2951 pub id: String,
2952 pub kind: Provider,
2953 /// Reference into the OS keystore for live credentials (e.g.
2954 /// `"keystore://cloudflare/yah"`). `None` for providers that don't need
2955 /// creds (local-static, optionally local-container).
2956 #[serde(default, skip_serializing_if = "Option::is_none")]
2957 pub credentials: Option<String>,
2958 /// Kind-specific fields. Examples:
2959 /// - cloudflare: `default_zone`
2960 /// - hetzner: `default_location`, `default_server_type`, `ssh_keys`
2961 /// - local-container: `runtime`, `discovery`
2962 #[serde(flatten)]
2963 #[cfg_attr(
2964 feature = "json-schema",
2965 schemars(with = "std::collections::BTreeMap<String, serde_json::Value>")
2966 )]
2967 pub fields: BTreeMap<String, toml::Value>,
2968}
2969
2970impl ProviderConfig {
2971 /// Parse a single `providers/<id>.toml` file.
2972 pub fn load(path: &Path) -> Result<Self> {
2973 let src =
2974 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
2975 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
2976 }
2977}
2978
2979/// An operator-facing service declaration from
2980/// `.yah/services/<svc>/service.toml`.
2981///
2982/// A service groups one or more components (a static surface, a containerized
2983/// API, an almanac…) under a single domain. Mirrors project the service onto
2984/// concrete infra; see [`MirrorConfig`].
2985#[derive(Debug, Clone, Serialize, Deserialize)]
2986#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
2987pub struct ServiceConfig {
2988 pub schema_version: u32,
2989 pub name: String,
2990 pub domain: String,
2991 #[serde(default, skip_serializing_if = "Vec::is_empty")]
2992 pub components: Vec<ServiceComponent>,
2993 /// Databases this service exposes, grouped by environment (W241). Every
2994 /// entry becomes a data-workbench / `sql_*` catalog id of the shape
2995 /// `<env>:<service>:<name>` (e.g. `pond:scrabcake:main`). Optional and
2996 /// default-empty — services without databases omit the `[db]` table
2997 /// entirely.
2998 #[serde(default, skip_serializing_if = "DbCatalog::is_empty")]
2999 pub db: DbCatalog,
3000}
3001
3002impl ServiceConfig {
3003 /// Parse a single `services/<svc>/service.toml` file.
3004 pub fn load(path: &Path) -> Result<Self> {
3005 let src =
3006 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
3007 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
3008 }
3009
3010 /// Persist to `.yah/services/<name>/service.toml`, creating the service
3011 /// directory if needed. Create-or-overwrite — the canonical replacement
3012 /// for the legacy `sites.json` write path. `workspace_root` is the camp
3013 /// dir (the parent of `.yah/`).
3014 pub fn save(&self, workspace_root: &Path) -> Result<()> {
3015 let dir = crate::paths::service_dir(workspace_root, &self.name);
3016 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
3017 let path = crate::paths::service_toml(workspace_root, &self.name);
3018 let s = toml::to_string_pretty(self)
3019 .with_context(|| format!("serializing service {}", self.name))?;
3020 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
3021 }
3022
3023 /// Remove `.yah/services/<name>/` and everything under it (service.toml
3024 /// plus its `mirrors/`). Returns `false` when the directory was already
3025 /// absent, so callers can distinguish "deleted" from "no-op".
3026 pub fn delete(workspace_root: &Path, name: &str) -> Result<bool> {
3027 let dir = crate::paths::service_dir(workspace_root, name);
3028 if !dir.exists() {
3029 return Ok(false);
3030 }
3031 std::fs::remove_dir_all(&dir).with_context(|| format!("removing {}", dir.display()))?;
3032 Ok(true)
3033 }
3034}
3035
3036/// A git source for a component (R561-F1, "BYO git").
3037///
3038/// When a [`ServiceComponent`] sets `git`, the component's code is NOT in this
3039/// workspace — it lives in an external repo that the reconciler shallow-clones
3040/// into a source cache before build (approach A: clone-at-reconcile, so config
3041/// load + validation stay offline). The component's `path` is then interpreted
3042/// relative to `<checkout>/<subdir>` instead of the workspace root.
3043#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3044#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3045pub struct GitSource {
3046 /// Clone URL (https or ssh) of the tenant repo.
3047 pub repo: String,
3048 /// Branch, tag, or commit SHA to check out. Defaults to `"main"`.
3049 #[serde(default = "default_git_ref")]
3050 pub r#ref: String,
3051 /// Optional sub-directory within the repo that the workspace is rooted at
3052 /// (e.g. a monorepo's `site/`). `path` is resolved relative to this.
3053 #[serde(default, skip_serializing_if = "Option::is_none")]
3054 pub subdir: Option<String>,
3055}
3056
3057fn default_git_ref() -> String {
3058 "main".to_string()
3059}
3060
3061/// How to reach an external infra root (R615-F1 / W274, "linked infra
3062/// sources"): a filesystem link to a sibling camp's live tree, or a git
3063/// checkout of an extracted infra repo.
3064#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3065#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3066#[serde(tag = "kind", rename_all = "kebab-case")]
3067pub enum InfraSourceKind {
3068 /// Filesystem link — reads the owner's live tree. The dev-loop shortcut,
3069 /// and the whole story until W274's "infra as its own repo" end-state.
3070 /// `path` is relative to *this* camp's root; infra is read from
3071 /// `<path>/.yah/infra/`.
3072 Path {
3073 path: String,
3074 },
3075 /// Git link — reused verbatim from [`GitSource`] (R561, "BYO git"),
3076 /// lifted here from "a component's code" to "a camp's infra registry."
3077 /// Loading stays offline (W274 §3): `yah infra sync` (R615-T3) is what
3078 /// clones/pulls this into `.yah/cache/infra/<owner>/`; `CloudConfig::load`
3079 /// only ever reads that cache, never the network.
3080 Git(GitSource),
3081}
3082
3083/// Write-gate for a linked [`InfraSource`] (R615-F1 / W274).
3084///
3085/// An enum, not a bool: the two states today are "borrower renders/plans but
3086/// cannot reconcile" and "this camp genuinely co-administers the shared
3087/// root," and a future read-write-with-approval tier is a third variant, not
3088/// a renamed boolean.
3089#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq, Eq)]
3090#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3091#[serde(rename_all = "kebab-case")]
3092pub enum SourceMode {
3093 /// Borrower can render and plan against the linked entries but cannot
3094 /// reconcile/mutate them — the owner remains the single manager. Default:
3095 /// a borrower is opt-in to write access, never opt-out of the safe state.
3096 #[default]
3097 ReadOnly,
3098 /// Escape hatch for a camp that genuinely co-administers a shared root.
3099 Manage,
3100}
3101
3102/// One `[[source]]` entry in `.yah/infra/sources.toml` (R615-F1 / W274) — an
3103/// external infra root this camp borrows machines/providers from.
3104#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3105#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3106pub struct InfraSource {
3107 /// Logical owner name, badged in the Infra tab (e.g. `"yah"`). Distinct
3108 /// from any camp/repo name the `kind` resolves through — this is what an
3109 /// operator sees on a borrowed row, not a path.
3110 pub owner: String,
3111 #[serde(flatten)]
3112 pub kind: InfraSourceKind,
3113 #[serde(default)]
3114 pub mode: SourceMode,
3115 /// Optional filter — name globs or mesh-tag selectors — to borrow a
3116 /// subset of the source root rather than everything it declares. Empty
3117 /// (the default) borrows everything.
3118 #[serde(default)]
3119 pub select: Vec<String>,
3120}
3121
3122fn default_sources_schema_version() -> u32 {
3123 1
3124}
3125
3126/// `.yah/infra/sources.toml` — the ordered list of external infra roots this
3127/// camp borrows from (R615-F1 / W274).
3128#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3129#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3130pub struct SourcesConfig {
3131 #[serde(default = "default_sources_schema_version")]
3132 pub schema_version: u32,
3133 /// `[[source]]` entries, in declaration order — overlay order matters
3134 /// when two linked sources both name the same machine (R615-F2).
3135 #[serde(default, rename = "source")]
3136 pub source: Vec<InfraSource>,
3137}
3138
3139impl Default for SourcesConfig {
3140 fn default() -> Self {
3141 Self {
3142 schema_version: default_sources_schema_version(),
3143 source: Vec::new(),
3144 }
3145 }
3146}
3147
3148impl SourcesConfig {
3149 /// Load `<infra_dir>/sources.toml`. A missing file is not an error —
3150 /// every camp without linked infra has none, which today is every camp —
3151 /// and yields an empty source list rather than `Err`.
3152 pub fn load(infra_dir: &Path) -> Result<Self> {
3153 let path = infra_dir.join("sources.toml");
3154 if !path.exists() {
3155 return Ok(Self::default());
3156 }
3157 let src =
3158 std::fs::read_to_string(&path).with_context(|| format!("reading {}", path.display()))?;
3159 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
3160 }
3161}
3162
3163impl InfraSource {
3164 /// Human-readable descriptor of *which* source this is, for
3165 /// [`InfraOrigin::source`] — distinguishes two linked sources from the
3166 /// same owner. Never includes credentials: `GitSource.repo` is a clone
3167 /// URL (https/ssh), the same thing R561 already treats as safe to log,
3168 /// with any real secret resolved separately via `keystore://` (W274's
3169 /// own precedent).
3170 fn describe(&self) -> String {
3171 match &self.kind {
3172 InfraSourceKind::Path { path } => format!("path:{path}"),
3173 InfraSourceKind::Git(g) => format!("git:{}@{}", g.repo, g.r#ref),
3174 }
3175 }
3176
3177 /// Resolve this source to an infra root directory (R615-F2 / W274 §3).
3178 /// Does no I/O and touches no network: `path` sources read the owner's
3179 /// live tree directly; `git` sources read wherever `yah infra sync`
3180 /// (R615-T3) last synced to, which may not exist yet (an unsynced git
3181 /// source overlays nothing, not an error — see [`load_dir_tolerant`]).
3182 ///
3183 /// `git.subdir` (reused verbatim from [`GitSource`]/R561) is honoured
3184 /// exactly like the component case: the checkout root when unset, or
3185 /// `<checkout>/<subdir>` when set — e.g. `subdir = "infra"` for a
3186 /// monorepo whose infra registry lives under `infra/` rather than at the
3187 /// clone's root. `yah infra sync` (R615-T3) clones into the *checkout*
3188 /// root ([`crate::paths::infra_source_cache_dir`]), never into a
3189 /// subdir-suffixed path, so this is the one place that appends `subdir`.
3190 fn infra_root(&self, workspace_root: &Path) -> std::path::PathBuf {
3191 match &self.kind {
3192 InfraSourceKind::Path { path } => workspace_root.join(path).join(".yah").join("infra"),
3193 InfraSourceKind::Git(g) => {
3194 let checkout = crate::paths::infra_source_cache_dir(workspace_root, &self.owner);
3195 match g.subdir.as_deref() {
3196 Some(subdir) => checkout.join(subdir),
3197 None => checkout,
3198 }
3199 }
3200 }
3201 }
3202}
3203
3204/// Provenance for a [`MachineConfig`] or [`ProviderConfig`] pulled in from a
3205/// linked `.yah/infra/sources.toml` entry, rather than declared in this
3206/// camp's own `.yah/infra/` (R615-F2 / W274).
3207///
3208/// Lives in [`CloudConfig::machine_origins`] / `provider_origins`, keyed by
3209/// name/id, rather than as a field on `MachineConfig`/`ProviderConfig`
3210/// themselves: those two types are constructed by struct literal in test
3211/// helpers across several crates (including ones this ticket has no reason to
3212/// touch), so widening either shape would ripple out past this crate for no
3213/// semantic gain — origin is a property of *this load*, not an inherent
3214/// property of the machine/provider. A name absent from the map is
3215/// camp-local; present means borrowed, and the Infra tab (R615-F4) / reconcile
3216/// gating (`InfraSource::mode`, copied onto `mode` below) read it from here.
3217#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3218#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3219pub struct InfraOrigin {
3220 /// The [`InfraSource::owner`] that supplied this entry, e.g. `"yah"`.
3221 pub owner: String,
3222 /// Which source, rendered — see [`InfraSource::describe`].
3223 pub source: String,
3224 /// The write-gate that applied when this entry was overlaid — copied
3225 /// from [`InfraSource::mode`] so a caller holding just the machine/
3226 /// provider doesn't need the source list in hand to know it's borrowed
3227 /// read-only.
3228 pub mode: SourceMode,
3229}
3230
3231/// Like [`load_dir`], but tolerant **per file**: a foreign infra root (an
3232/// owner's live tree, or a synced git checkout) can carry entries this
3233/// binary's `T` predates — noisetable's pre-migration machines used an older
3234/// schema than yah's, and the reverse will happen too as each side evolves
3235/// independently. One unparseable file on a source this camp doesn't own must
3236/// never sink every other entry in the same directory, let alone this camp's
3237/// own load (R615-F2 gotcha). Contrast [`load_dir`], which stays strict for
3238/// camp-local files, where a malformed TOML genuinely should be a hard error.
3239///
3240/// Returns the entries that parsed, plus `(path, error)` for every file that
3241/// didn't — the caller logs those, it doesn't drop them silently. A missing
3242/// or unreadable directory yields `(vec![], vec![])`, same "no entries" as
3243/// `load_dir`'s `!dir.exists()` case (an unsynced git source, or a source
3244/// root with no `providers/` at all, are both normal, not warnings).
3245fn load_dir_tolerant<T: for<'de> Deserialize<'de>>(
3246 dir: &Path,
3247) -> (Vec<T>, Vec<(std::path::PathBuf, anyhow::Error)>) {
3248 let Ok(read_dir) = std::fs::read_dir(dir) else {
3249 return (Vec::new(), Vec::new());
3250 };
3251 let mut entries: Vec<_> = read_dir.filter_map(|e| e.ok()).collect();
3252 entries.sort_by_key(|e| e.file_name());
3253
3254 let mut items = Vec::new();
3255 let mut skipped = Vec::new();
3256 for entry in entries {
3257 let path = entry.path();
3258 if path.extension().map_or(true, |e| e != "toml") {
3259 continue;
3260 }
3261 let parsed = std::fs::read_to_string(&path)
3262 .with_context(|| format!("reading {}", path.display()))
3263 .and_then(|src| {
3264 toml::from_str::<T>(&src).with_context(|| format!("parsing {}", path.display()))
3265 });
3266 match parsed {
3267 Ok(item) => items.push(item),
3268 Err(e) => skipped.push((path, e)),
3269 }
3270 }
3271 (items, skipped)
3272}
3273
3274/// Whether a borrowed machine passes an [`InfraSource::select`] filter
3275/// (R615-F2 / W274). Empty `select` borrows everything. A non-empty `select`
3276/// entry matches either the machine's exact `name` or literal membership in
3277/// its `mesh_tags` — the one shape W274's own example uses
3278/// (`select = ["tag:cloud-runner"]`). Not a glob engine: mesh tags are
3279/// already flat strings compared for exact equality everywhere else in this
3280/// crate (see `resolve_machine_by_mesh_tags`), so a select entry is that same
3281/// comparison, not a new pattern language.
3282fn machine_matches_select(machine: &MachineConfig, select: &[String]) -> bool {
3283 select.is_empty()
3284 || select
3285 .iter()
3286 .any(|s| *s == machine.name || machine.mesh_tags.contains(s))
3287}
3288
3289/// What one `[[source]]` in `.yah/infra/sources.toml` actually contributed to
3290/// [`FleetInventory`] on this load (R870-B13).
3291///
3292/// Recorded because a link that resolves to *nothing* is indistinguishable, at
3293/// every downstream use site, from a camp that declared no link at all — and
3294/// that is precisely the failure this ticket exists to fix. A source that
3295/// contributes zero machines is not an error here (an unsynced `kind = "git"`
3296/// source is legitimately empty, and `load()` must stay offline), so instead
3297/// the fact is *carried* to whoever fails for want of a machine. See
3298/// [`FleetInventory::describe_sources`].
3299#[derive(Debug, Clone, PartialEq, Eq)]
3300pub struct SourceContribution {
3301 /// [`InfraSource::owner`] — the name this camp knows the fleet by.
3302 pub owner: String,
3303 /// The source, rendered — see [`InfraSource::describe`].
3304 pub source: String,
3305 /// Where the link resolved to, i.e. the foreign `.yah/infra/`.
3306 pub root: std::path::PathBuf,
3307 /// Whether `root` exists on disk. `false` for a `kind = "path"` link
3308 /// aimed at a directory that is not a camp, and for a `kind = "git"`
3309 /// source that `yah infra sync` has never fetched.
3310 pub root_exists: bool,
3311 /// How many machines this source actually added to the inventory — after
3312 /// [`InfraSource::select`] filtering and after losing every name a
3313 /// camp-local entry or an earlier source already claimed.
3314 pub machines: usize,
3315}
3316
3317/// A camp's resolved machine inventory: **the** answer to "which machines does
3318/// this camp have", with exactly one implementation
3319/// ([`resolve_fleet_inventory`]) behind it (R870-B13).
3320///
3321/// A borrowing camp — one whose own `.yah/infra/machines/` is empty and which
3322/// declares `[[source]]` links to another camp's fleet in
3323/// `.yah/infra/sources.toml` — is the case this type exists for. Before it,
3324/// the overlay was applied inline inside [`CloudConfig::load`], so the two
3325/// callers that resolve a *machine name to a machine* (ingress collation and
3326/// the sovereign apex render) read a camp-local-only loader and saw an empty
3327/// fleet. There was no bug in either of them; the inventory simply had two
3328/// readers that disagreed about what the inventory was.
3329#[derive(Debug)]
3330pub struct FleetInventory {
3331 /// Camp-local machines first, then each source's contribution in
3332 /// declaration order. Camp-local wins any name collision; among sources,
3333 /// the earlier-declared one wins.
3334 pub machines: Vec<MachineConfig>,
3335 /// Provenance for the borrowed entries, keyed by [`MachineConfig::name`].
3336 /// A name absent here is camp-local. Same shape and meaning as
3337 /// [`CloudConfig::machine_origins`], which is populated from this.
3338 pub origins: BTreeMap<String, InfraOrigin>,
3339 /// The parsed `.yah/infra/sources.toml`, kept so a caller that already has
3340 /// an inventory in hand does not re-read it (`CloudConfig::load` overlays
3341 /// providers from the same list).
3342 pub sources: SourcesConfig,
3343 /// Per-source accounting — see [`SourceContribution`].
3344 pub contributions: Vec<SourceContribution>,
3345}
3346
3347impl FleetInventory {
3348 /// One line per declared `[[source]]`, for attaching to the error a caller
3349 /// raises when a machine name does not resolve (R870-B13).
3350 ///
3351 /// The failure being diagnosed is always "I was told about machine X and
3352 /// cannot find it", and the three ways a borrowing camp gets there — no
3353 /// link declared, a link pointing somewhere that is not a camp, a link
3354 /// whose `select` filtered X out — are indistinguishable from the name
3355 /// alone. Empty string when the camp declares no sources, so the caller
3356 /// can append it unconditionally without emitting a dangling header.
3357 pub fn describe_sources(&self) -> String {
3358 if self.contributions.is_empty() {
3359 return String::new();
3360 }
3361 let mut out = String::from("linked infra sources consulted:");
3362 for c in &self.contributions {
3363 out.push_str(&format!(
3364 "\n {} ({}) -> {}{} — contributed {} machine(s)",
3365 c.owner,
3366 c.source,
3367 c.root.display(),
3368 if c.root_exists {
3369 ""
3370 } else {
3371 " [ABSENT: not a camp, or an unsynced git source]"
3372 },
3373 c.machines,
3374 ));
3375 }
3376 out
3377 }
3378}
3379
3380/// Resolve a camp's machine inventory: camp-local `.yah/infra/machines/`, the
3381/// pre-R215 `.yah/cloud/machines/` tree, then every machine borrowed through
3382/// `.yah/infra/sources.toml` (R870-B13, on R615-F2's mechanism).
3383///
3384/// **How a camp names another camp's fleet**, decided here rather than
3385/// invented: through the `[[source]]` entry R615-F1 already defines — `owner`
3386/// is the logical name an operator sees, `kind = "path"` resolves against the
3387/// borrowing camp's own root and `kind = "git"` against `yah infra sync`'s
3388/// cache. There is deliberately no second naming scheme: a camp that could
3389/// name a foreign fleet two ways would be a camp whose inventory can drift
3390/// from itself, which is the thing this ticket rejected.
3391///
3392/// **There is exactly one copy.** A `kind = "path"` source reads the owner's
3393/// live tree at `<path>/.yah/infra/` on every load — the borrowing camp
3394/// persists nothing, so the two can never disagree. `kind = "git"` reads a
3395/// synced checkout, which *is* a copy, but an explicit one with a named
3396/// refresh verb (`yah infra sync`) and a pinned `ref`; that is the cache with
3397/// an invalidation story, as against a hand-maintained second inventory.
3398///
3399/// Camp-local files are strict (a malformed TOML this camp owns is a hard
3400/// error) and foreign files are tolerant per-file (R615-F2: a foreign entry
3401/// whose schema this binary predates must not sink the load). A foreign
3402/// machine skipped that way is not silently lost — it fails loudly at the
3403/// point some caller needs it, with [`FleetInventory::describe_sources`]
3404/// naming the link it should have come from.
3405///
3406/// Deliberately *without* [`CloudConfig::load`]'s R844-B7 wrong-root guard: a
3407/// missing `.yah/infra/machines/` is an empty inventory here, because the
3408/// callers that resolve against it (ingress collation, apex render) are handed
3409/// a root that a `CloudConfig::load` already accepted.
3410pub fn resolve_fleet_inventory(workspace_root: &Path) -> Result<FleetInventory> {
3411 let mut machines = load_dir::<MachineConfig>(crate::paths::machines_dir(workspace_root))?;
3412
3413 // Pre-R215 `.yah/cloud/machines/`. Shouldn't have anything since R215-B1
3414 // moved them, but if it does we dedupe by name — R215+ wins.
3415 let cloud_dir = crate::paths::legacy_cloud_dir(workspace_root);
3416 if cloud_dir.exists() {
3417 let names: std::collections::HashSet<String> =
3418 machines.iter().map(|m| m.name.clone()).collect();
3419 for m in load_dir::<MachineConfig>(cloud_dir.join("machines"))? {
3420 if !names.contains(&m.name) {
3421 machines.push(m);
3422 }
3423 }
3424 }
3425
3426 // `SourcesConfig::load` never touches the network — git sources are read
3427 // from `yah infra sync`'s cache (R615-T3) — so this keeps the whole
3428 // offline contract `CloudConfig::load` has always had.
3429 let sources = SourcesConfig::load(&crate::paths::infra_dir(workspace_root))?;
3430 let mut origins = BTreeMap::new();
3431 let contributions = overlay_source_machines(workspace_root, &sources, &mut machines, &mut origins);
3432
3433 Ok(FleetInventory {
3434 machines,
3435 origins,
3436 sources,
3437 contributions,
3438 })
3439}
3440
3441/// Overlay every linked `.yah/infra/sources.toml` source's machines into
3442/// `machines`, recording provenance into `machine_origins` (R615-F2 / W274).
3443/// Must be called AFTER camp-local entries are already in the vector:
3444/// collision resolution is "first writer wins," so seeding with camp-local
3445/// first is what makes camp-local win over every source, and an earlier source
3446/// win over a later one.
3447///
3448/// `select` filters which machines a source contributes. Returns one
3449/// [`SourceContribution`] per declared source, in declaration order.
3450fn overlay_source_machines(
3451 workspace_root: &Path,
3452 sources: &SourcesConfig,
3453 machines: &mut Vec<MachineConfig>,
3454 machine_origins: &mut BTreeMap<String, InfraOrigin>,
3455) -> Vec<SourceContribution> {
3456 let mut seen_machine_names: std::collections::HashSet<String> =
3457 machines.iter().map(|m| m.name.clone()).collect();
3458 let mut contributions = Vec::with_capacity(sources.source.len());
3459
3460 for source in &sources.source {
3461 let root = source.infra_root(workspace_root);
3462 let origin = InfraOrigin {
3463 owner: source.owner.clone(),
3464 source: source.describe(),
3465 mode: source.mode,
3466 };
3467
3468 let (foreign_machines, skipped) = load_dir_tolerant::<MachineConfig>(&root.join("machines"));
3469 for (path, e) in skipped {
3470 tracing::warn!(
3471 "infra source {:?} ({}): skipping unparseable machine {}: {e:#}",
3472 source.owner,
3473 root.display(),
3474 path.display()
3475 );
3476 }
3477 let mut added = 0usize;
3478 for m in foreign_machines {
3479 if seen_machine_names.contains(&m.name) {
3480 continue; // camp-local, or an earlier source, already claimed this name
3481 }
3482 if !machine_matches_select(&m, &source.select) {
3483 continue;
3484 }
3485 seen_machine_names.insert(m.name.clone());
3486 machine_origins.insert(m.name.clone(), origin.clone());
3487 machines.push(m);
3488 added += 1;
3489 }
3490
3491 contributions.push(SourceContribution {
3492 owner: source.owner.clone(),
3493 source: source.describe(),
3494 root_exists: root.is_dir(),
3495 root,
3496 machines: added,
3497 });
3498 }
3499
3500 contributions
3501}
3502
3503/// Overlay every linked source's providers into `providers`, recording
3504/// provenance into `provider_origins` (R615-F2 / W274). Same first-writer-wins
3505/// rule as [`overlay_source_machines`], and the same requirement that
3506/// camp-local entries already be in the vector.
3507///
3508/// [`InfraSource::select`] deliberately does not apply: nothing in W274 or
3509/// R615-F1 describes a provider-scoped filter — every provider a source
3510/// declares either overlays whole or, on an id collision, doesn't.
3511fn overlay_source_providers(
3512 workspace_root: &Path,
3513 sources: &SourcesConfig,
3514 providers: &mut Vec<ProviderConfig>,
3515 provider_origins: &mut BTreeMap<String, InfraOrigin>,
3516) {
3517 let mut seen_provider_ids: std::collections::HashSet<String> =
3518 providers.iter().map(|p| p.id.clone()).collect();
3519
3520 for source in &sources.source {
3521 let root = source.infra_root(workspace_root);
3522 let origin = InfraOrigin {
3523 owner: source.owner.clone(),
3524 source: source.describe(),
3525 mode: source.mode,
3526 };
3527
3528 let (foreign_providers, skipped) =
3529 load_dir_tolerant::<ProviderConfig>(&root.join("providers"));
3530 for (path, e) in skipped {
3531 tracing::warn!(
3532 "infra source {:?} ({}): skipping unparseable provider {}: {e:#}",
3533 source.owner,
3534 root.display(),
3535 path.display()
3536 );
3537 }
3538 for p in foreign_providers {
3539 if seen_provider_ids.contains(&p.id) {
3540 continue;
3541 }
3542 seen_provider_ids.insert(p.id.clone());
3543 provider_origins.insert(p.id.clone(), origin.clone());
3544 providers.push(p);
3545 }
3546 }
3547}
3548
3549/// One component of a [`ServiceConfig`]. The `kind` (e.g. `"mesofact-static"`,
3550/// `"almanac"`, `"container"`) selects which reconciler runs against the
3551/// pointed-at workload manifest.
3552#[derive(Debug, Clone, Serialize, Deserialize)]
3553#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3554pub struct ServiceComponent {
3555 pub id: String,
3556 pub kind: String,
3557 /// Path of the directory holding this component's `workload.toml`. Relative
3558 /// to the workspace root for in-tree components, or to the materialized
3559 /// `<checkout>/<subdir>` when [`git`](Self::git) is set.
3560 pub path: String,
3561 /// Optional external git source (R561-F1). When set, the component's code
3562 /// is materialized by shallow-clone before build; see [`GitSource`].
3563 #[serde(default, skip_serializing_if = "Option::is_none")]
3564 pub git: Option<GitSource>,
3565 /// Operator-facing role label, e.g. `"static"`, `"dynamic"`, `"compute"`.
3566 pub role: String,
3567 /// Optional artifact kind this component publishes (`"static"`,
3568 /// `"container-image"`, …). Drives mirror provider-slot routing.
3569 #[serde(default, skip_serializing_if = "Option::is_none")]
3570 pub publishes: Option<String>,
3571 /// URL sub-path a static component's build output is published under,
3572 /// relative to the service's publish prefix (R746). `None` = the service
3573 /// root, which is what every pre-R746 component means.
3574 ///
3575 /// Static publishers lay a component's `out_dir` down at
3576 /// `<bucket>/<service>/<env>/…` and the front door fetches
3577 /// `${ASSET_ORIGIN}/<request path>` — the request path *is* the key. So a
3578 /// service with two static components had them overwrite each other at
3579 /// one prefix, and there was no way to say "this bundle serves under
3580 /// /app". `mount` is that: it appends to the publish prefix, which makes
3581 /// the URL sub-path and the storage sub-path the same string by
3582 /// construction rather than by two manifests agreeing.
3583 ///
3584 /// Cross-checked against the domain route that names the component
3585 /// ([`CloudConfig::cross_ref_validate`]): a component mounted at `/app`
3586 /// must be routed at `/app` or `/app/*`, because a disagreement means
3587 /// requests land on a prefix nothing published to — a 404 whose cause is
3588 /// two files apart.
3589 #[serde(default, skip_serializing_if = "Option::is_none")]
3590 pub mount: Option<String>,
3591 /// Sync-wave index (0-based). Components in wave 0 roll out in parallel
3592 /// first; the reconciler waits for all wave-N components to become healthy
3593 /// before starting wave N+1. Defaults to 0 (all components in one wave).
3594 #[serde(default, skip_serializing_if = "is_zero_u32")]
3595 pub wave: u32,
3596
3597 /// Whether this component ships inside the service's one assembled bundle
3598 /// or as a deployed unit of its own (R870-F23).
3599 ///
3600 /// This is the vocabulary R870-F15's design needed and the config did not
3601 /// have. `[providers.bundle]` is a per-**mirror** slot, so before this
3602 /// there was no way to say "give this one component its own workload" at
3603 /// all — the whole service was one bundle or it was nothing, and a service
3604 /// whose components genuinely release on different cadences had no shape
3605 /// to declare.
3606 ///
3607 /// It is one field rather than a pair of flags on purpose: a component
3608 /// being both bundle-staged and its own workload is the second admission
3609 /// rule R870-F23 was asked to enforce, and an enum makes it unrepresentable
3610 /// instead of merely refused.
3611 #[serde(default, skip_serializing_if = "DeployTier::is_default")]
3612 pub deploy: DeployTier,
3613}
3614
3615/// How one [`ServiceComponent`] reaches a node — see
3616/// [`ServiceComponent::deploy`].
3617#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
3618#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3619#[serde(rename_all = "kebab-case")]
3620pub enum DeployTier {
3621 /// Staged into the service's single assembled W272 bundle under
3622 /// `app/dist/<mount>/` and served by the one bundle workload (R870-B11).
3623 /// The default, and what every component in the tree means today.
3624 #[default]
3625 Bundle,
3626 /// Deployed as its own workload, with its own release cadence, its own
3627 /// address, and its own place in the inner door's mount table.
3628 Workload,
3629}
3630
3631impl DeployTier {
3632 /// Skip serializing the default so existing `service.toml` files
3633 /// round-trip byte-identically.
3634 fn is_default(&self) -> bool {
3635 matches!(self, DeployTier::Bundle)
3636 }
3637}
3638
3639#[inline]
3640fn is_zero_u32(n: &u32) -> bool {
3641 *n == 0
3642}
3643
3644/// A service's declared databases, grouped by environment (W241 §Sections).
3645/// Parsed from the `[db]` table of `service.toml`; each `[[db.<env>]]` array
3646/// entry names one database. The environment tag drives backend selection at
3647/// query time (see the data-workbench's `db.query` / the `sql_*` MCP tools):
3648/// `dev` = local file, `pond` = a DB inside the running pond container stack
3649/// (reached on a declared localhost port), `cloud` = a remote libSQL/Turso or
3650/// Postgres endpoint whose auth comes from an env var (never stored in TOML).
3651#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq, Eq)]
3652#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3653pub struct DbCatalog {
3654 /// Local-file SQLite databases used in dev mode.
3655 #[serde(default, skip_serializing_if = "Vec::is_empty")]
3656 pub dev: Vec<DevDb>,
3657 /// Databases running inside the pond container stack.
3658 #[serde(default, skip_serializing_if = "Vec::is_empty")]
3659 pub pond: Vec<PondDb>,
3660 /// Remote cloud databases (Turso, Postgres).
3661 #[serde(default, skip_serializing_if = "Vec::is_empty")]
3662 pub cloud: Vec<CloudDb>,
3663}
3664
3665impl DbCatalog {
3666 /// True when no database is declared in any environment. Lets
3667 /// [`ServiceConfig`] skip serializing an empty `[db]` table.
3668 pub fn is_empty(&self) -> bool {
3669 self.dev.is_empty() && self.pond.is_empty() && self.cloud.is_empty()
3670 }
3671}
3672
3673/// A dev-mode local SQLite database (`[[db.dev]]`). `path` is resolved
3674/// relative to the workspace root and opened as a local file — read/write, no
3675/// network, no auth.
3676#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3677#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3678pub struct DevDb {
3679 /// Logical name, unique within the service's `dev` list. Forms the `name`
3680 /// segment of the catalog id `dev:<service>:<name>`.
3681 pub name: String,
3682 /// On-disk SQLite path, relative to the workspace root (or absolute).
3683 pub path: String,
3684}
3685
3686/// A database running inside the pond container stack (`[[db.pond]]`). The
3687/// pond publishes the DB on a localhost TCP port; the hub connects to
3688/// `127.0.0.1:<port>` when the pond is up and returns a clear error when it is
3689/// not. Either `port` (defaulting to a libSQL/`sqld` HTTP endpoint) or a full
3690/// `url` must be given.
3691#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3692#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3693pub struct PondDb {
3694 /// Logical name, unique within the service's `pond` list.
3695 pub name: String,
3696 /// Localhost TCP port the pond publishes the DB on. Interpreted per
3697 /// [`kind`](Self::kind). Mutually complete with `url` (provide one).
3698 #[serde(default, skip_serializing_if = "Option::is_none")]
3699 pub port: Option<u16>,
3700 /// Full connection URL, overriding `port` when set (e.g. a non-localhost
3701 /// host or an explicit scheme).
3702 #[serde(default, skip_serializing_if = "Option::is_none")]
3703 pub url: Option<String>,
3704 /// Wire protocol the pond DB speaks. Selects how a bare `port` becomes a
3705 /// URL: `turso` → `http://127.0.0.1:<port>` (libSQL/`sqld` over Hrana),
3706 /// `postgres` → `postgres://127.0.0.1:<port>`.
3707 #[serde(default)]
3708 pub kind: PondDbKind,
3709}
3710
3711/// Wire protocol of a [`PondDb`].
3712#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq, Eq)]
3713#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3714#[serde(rename_all = "kebab-case")]
3715pub enum PondDbKind {
3716 /// libSQL / `sqld` over Hrana HTTP — the default.
3717 #[default]
3718 Turso,
3719 /// PostgreSQL wire protocol.
3720 Postgres,
3721}
3722
3723/// A remote cloud database (`[[db.cloud]]`). The connection `url` is stored in
3724/// TOML but the credential never is — `auth_token_env` names an environment
3725/// variable the daemon reads at connect time, so the same declaration works
3726/// whether the token is provisioned service-locally or camp-shared (W241;
3727/// operator confirmed both scopes are needed). A camp-wide cloud DB not owned
3728/// by any single service is declared identically in `.yah/db/cloud.toml`.
3729#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3730#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3731pub struct CloudDb {
3732 /// Logical name, unique within its `cloud` list.
3733 pub name: String,
3734 /// Connection URL: `libsql://…` / `http(s)://…` (Turso, `sqld`) or
3735 /// `postgres://…`.
3736 pub url: String,
3737 /// Name of the environment variable holding the auth token. Resolved in
3738 /// the daemon at connect time (value never stored on disk). For a libSQL
3739 /// URL the token is threaded as `?auth_token=…`.
3740 #[serde(default, skip_serializing_if = "Option::is_none")]
3741 pub auth_token_env: Option<String>,
3742}
3743
3744/// A camp-shared cloud database catalog, parsed from `.yah/db/cloud.toml`.
3745/// These are cloud DBs not owned by any single service — declared once at camp
3746/// scope and addressed as `cloud:<name>` (two-segment id), distinct from a
3747/// service-local `cloud:<service>:<name>`.
3748#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq, Eq)]
3749#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3750pub struct CampCloudDbs {
3751 #[serde(default, rename = "cloud", skip_serializing_if = "Vec::is_empty")]
3752 pub cloud: Vec<CloudDb>,
3753}
3754
3755impl CampCloudDbs {
3756 /// Load `<camp_root>/.yah/db/cloud.toml`, or an empty catalog if the file
3757 /// is absent (the common case — most camps declare no shared cloud DBs).
3758 pub fn load(camp_root: &Path) -> Result<Self> {
3759 let path = camp_root.join(".yah/db/cloud.toml");
3760 if !path.exists() {
3761 return Ok(Self::default());
3762 }
3763 let src = std::fs::read_to_string(&path)
3764 .with_context(|| format!("reading {}", path.display()))?;
3765 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
3766 }
3767}
3768
3769/// Topological shape of a mirror — how its providers sit relative to each other.
3770#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
3771#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3772#[serde(rename_all = "kebab-case")]
3773pub enum MirrorShape {
3774 /// Single machine hosts compute (and any non-Cloudflare-fronted static).
3775 SingleMachine,
3776 /// Operator-local dev mirror — static via built-in file server, compute
3777 /// via the local container runtime.
3778 Local,
3779 /// Multi-machine deployment (machines listed per provider slot).
3780 MultiMachine,
3781}
3782
3783/// Which public-ingress provider fronts this mirror's compute (W267, R594-F11).
3784///
3785/// Both arms answer exactly one question — *given these local workload ports,
3786/// make them publicly reachable at these hostnames* — and they differ only in
3787/// where the ingress rules live and who supervises the front door:
3788///
3789/// | | [`CloudflareTunnel`](Self::CloudflareTunnel) | [`Passway`](Self::Passway) |
3790/// |---|---|---|
3791/// | Ingress rules live | Cloudflare's API (token-form tunnels are remotely-managed) | the pingora `Backends` set in the proxy process |
3792/// | How they get there | an API call per deployed workload | passway polls `GET /service-records?ready=true` |
3793/// | Front door lifecycle | a kamaji-supervised `cloudflared` appliance | a kamaji-supervised passway appliance |
3794///
3795/// Flipping this field is the whole tier ladder: rented edge → sovereign edge
3796/// is a one-line mirror edit, not a rewrite. The provider owns **addressing**
3797/// and never **rendering** — the W173 render cube stays in mesofact's manifest.
3798#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
3799#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3800#[serde(rename_all = "kebab-case")]
3801pub enum IngressProvider {
3802 /// No public front door for this mirror. The default: a mirror that
3803 /// publishes to R2 behind a Worker, or a mesh-only compute tier, has no
3804 /// ingress provider to reconcile.
3805 #[default]
3806 None,
3807 /// Rented edge — `cloudflared` dials *out* from the node to Cloudflare's
3808 /// edge. Zero inbound ports, no TLS to manage on the box, hostname rules
3809 /// held in Cloudflare's API.
3810 CloudflareTunnel,
3811 /// Sovereign edge — passway terminates TLS on the node and load-balances
3812 /// an upstream set discovered from yubaba's service records.
3813 Passway,
3814}
3815
3816impl IngressProvider {
3817 /// `true` when this mirror declares a front door that has to be reconciled.
3818 pub fn is_declared(self) -> bool {
3819 !matches!(self, Self::None)
3820 }
3821
3822 /// Kebab-case wire name, as it appears in `mirrors/<env>.toml`.
3823 pub fn as_str(self) -> &'static str {
3824 match self {
3825 Self::None => "none",
3826 Self::CloudflareTunnel => "cloudflare-tunnel",
3827 Self::Passway => "passway",
3828 }
3829 }
3830}
3831
3832/// One declared **edge**: a front door, the slots it fronts, and the nodes it
3833/// is placed on (W305 F2).
3834///
3835/// A mirror declares a *list* of these, which is what lets one service mix
3836/// front doors — cloudflare for the public web tier, passway for an internal or
3837/// high-throughput one. Before this, [`MirrorConfig::ingress`] was a single
3838/// [`IngressProvider`], so a mirror could **swap** front doors but never mix
3839/// them.
3840///
3841/// ```toml
3842/// [[ingress]]
3843/// provider = "passway"
3844/// machines = ["us-east-001", "us-south-001"]
3845/// slots = ["bundle"]
3846///
3847/// [[ingress]]
3848/// provider = "cloudflare-tunnel"
3849/// hostnames = ["issues.yah.dev"]
3850/// ```
3851///
3852/// **The per-node appliance is derived from this, never declared beside it.**
3853/// An edge does invoke a cloudflared or passway process on a box, but that is a
3854/// *consequence* of the service's declaration:
3855/// [`collate_front_doors`](crate::reconciler::collate_front_doors) walks every
3856/// service and derives what each node must run. Declaring it node-side too is
3857/// what produces two sources of truth for one fact.
3858#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
3859#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3860pub struct IngressEdge {
3861 /// Which front door this edge is. [`IngressProvider::None`] is rejected at
3862 /// plan time — an edge that fronts with nothing is always a typo, never an
3863 /// intent (write no edge instead).
3864 pub provider: IngressProvider,
3865 /// Nodes this front door is placed on — **independent of where the fronted
3866 /// workload runs** (R330-F37).
3867 ///
3868 /// Empty falls back to the fronted slot's own `machine` / `machines`, which
3869 /// is the co-located shape every mirror had before front-door placement was
3870 /// expressible. Listing several is what lets the ingress tier and the
3871 /// service tier scale independently: **N front doors over ONE deployment**,
3872 /// one rendered copy, so no cache coherence to settle.
3873 #[serde(default, skip_serializing_if = "Vec::is_empty")]
3874 pub machines: Vec<String>,
3875 /// Provider slot roles this edge fronts (`"bundle"`, `"compute"`, …).
3876 ///
3877 /// One of the two selectors. With a single edge both may be empty, meaning
3878 /// "every fronted slot" — the legacy shape. With **several** edges a
3879 /// selector is mandatory on each, and the partition must be total and
3880 /// disjoint: a slot claimed by no edge, or by two, is an error naming it.
3881 /// An implicit catch-all across mixed front doors would silently publish a
3882 /// service through the wrong one.
3883 #[serde(default, skip_serializing_if = "Vec::is_empty")]
3884 pub slots: Vec<String>,
3885 /// Public hostnames this edge fronts — the other selector, for partitioning
3886 /// by what the world dials rather than by which slot serves it.
3887 #[serde(default, skip_serializing_if = "Vec::is_empty")]
3888 pub hostnames: Vec<String>,
3889 /// Cloudflare Tunnel id this edge publishes through, overriding the
3890 /// fronting machine's [`MachineConfig::cloudflared`].
3891 ///
3892 /// This is W267 Gap 3's real fix, and it is the *service* side of it: a node
3893 /// can join two cohorts' orange networks, and since §Granularity argues the
3894 /// tunnel credential **is** the isolation boundary, which cohort a given
3895 /// service fronts through is a property of the service, not of the box.
3896 /// `MachineConfig.cloudflared` stays as the per-node default (one tunnel is
3897 /// the common case, and the credential does live on the node), but it is no
3898 /// longer the only way to say it — so the node never has to enumerate
3899 /// cohorts.
3900 #[serde(default, skip_serializing_if = "Option::is_none")]
3901 pub tunnel_id: Option<String>,
3902 /// Infra provider id whose credentials this edge's front door authenticates
3903 /// with — `use = "cloudflare"`, resolved through
3904 /// `.yah/infra/providers/<id>.toml` exactly as a slot's `use` is.
3905 ///
3906 /// Same split as [`tunnel_id`](Self::tunnel_id), one field over: whose
3907 /// Cloudflare account holds the tunnel is a property of the **front door**,
3908 /// not of the box that runs the compute. Without this the account was read
3909 /// off the fronted slot's own `use`, which conflates two unrelated facts —
3910 /// and is unwritable for a slot whose compute provider is `kind = "static"`
3911 /// (a borrowed bare box: placement only, no credentials). Such a mirror had
3912 /// no way to name a Cloudflare account at all, short of writing
3913 /// `use = "cloudflare"` on the compute slot and lying about what runs it
3914 /// (R845).
3915 ///
3916 /// `None` falls back to the fronted slot's `use`, which is what every
3917 /// mirror written before this field meant.
3918 #[serde(default, rename = "use", skip_serializing_if = "Option::is_none")]
3919 pub provider_id: Option<String>,
3920 /// Digest-pinned image reference for this edge's front-door appliance,
3921 /// e.g. `localhost/passway:tag@sha256:<hex>` (R870-F16).
3922 ///
3923 /// `None` is the state of every mirror on disk today: the passway arm of
3924 /// `yah cloud apply` cannot deploy an appliance the mirror doesn't name an
3925 /// image for, so it renders the manual `yah cloud ingress deploy …
3926 /// --image <passway-ref>` step instead of running it. Declaring this field
3927 /// is what makes the arm self-sufficient, matching the CloudflareTunnel
3928 /// arm's real-API-call shape rather than only printing for an operator to
3929 /// copy by hand.
3930 #[serde(default, skip_serializing_if = "Option::is_none")]
3931 pub image: Option<String>,
3932}
3933
3934impl IngressEdge {
3935 /// An edge with no selector — fronts every fronted slot, legal only when it
3936 /// is the mirror's only edge.
3937 pub fn all_slots(provider: IngressProvider, machines: Vec<String>) -> Self {
3938 Self {
3939 provider,
3940 machines,
3941 slots: Vec::new(),
3942 hostnames: Vec::new(),
3943 tunnel_id: None,
3944 provider_id: None,
3945 image: None,
3946 }
3947 }
3948
3949 /// `true` when this edge names which slots/hostnames it fronts.
3950 pub fn has_selector(&self) -> bool {
3951 !self.slots.is_empty() || !self.hostnames.is_empty()
3952 }
3953
3954 /// Does this edge claim the rule derived from `slot` publishing `hostname`?
3955 ///
3956 /// A selectorless edge claims everything; that is checked to be
3957 /// unambiguous (one edge only) before this is consulted.
3958 pub fn claims(&self, slot: &str, hostname: &str) -> bool {
3959 if !self.has_selector() {
3960 return true;
3961 }
3962 self.slots.iter().any(|s| s == slot) || self.hostnames.iter().any(|h| h == hostname)
3963 }
3964
3965 /// Human-readable identity for an error message — the provider plus
3966 /// whichever selector was written.
3967 pub fn label(&self) -> String {
3968 let sel = match (self.slots.is_empty(), self.hostnames.is_empty()) {
3969 (true, true) => "no selector".to_string(),
3970 (false, true) => format!("slots = {:?}", self.slots),
3971 (true, false) => format!("hostnames = {:?}", self.hostnames),
3972 (false, false) => format!("slots = {:?} + hostnames = {:?}", self.slots, self.hostnames),
3973 };
3974 format!("[[ingress]] provider = {:?} ({sel})", self.provider.as_str())
3975 }
3976}
3977
3978/// A mirror's `ingress` declaration, in either spelling.
3979///
3980/// The list is the general form; the bare provider is shorthand for the single
3981/// edge fronting everything, and is kept rather than migrated because it is the
3982/// honest spelling for the common case — one service, one front door. Both
3983/// normalize to the same `Vec<IngressEdge>` through
3984/// [`MirrorConfig::ingress_edges`], so nothing downstream branches on which was
3985/// written.
3986#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
3987#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3988#[serde(untagged)]
3989pub enum IngressDecl {
3990 /// `ingress = "passway"` — one edge fronting every fronted slot, placed by
3991 /// the sibling [`MirrorConfig::ingress_machines`].
3992 Provider(IngressProvider),
3993 /// `[[ingress]]` — one entry per declared edge.
3994 Edges(Vec<IngressEdge>),
3995}
3996
3997/// Hand-written because `#[serde(untagged)]` throws the real error away.
3998///
3999/// A derived untagged `Deserialize` tries each variant and, on failure, reports
4000/// only `data did not match any variant of untagged enum IngressDecl` — so a
4001/// misspelled `provider = "passwya"` says nothing about providers, nothing about
4002/// the legal values, and points at the `[[ingress]]` header rather than the
4003/// field. Dispatching on the input shape first means each arm's own error
4004/// survives: a bad string names the legal provider vocabulary, a bad edge table
4005/// names the offending field.
4006impl<'de> Deserialize<'de> for IngressDecl {
4007 fn deserialize<D: serde::Deserializer<'de>>(d: D) -> std::result::Result<Self, D::Error> {
4008 struct DeclVisitor;
4009
4010 impl<'de> serde::de::Visitor<'de> for DeclVisitor {
4011 type Value = IngressDecl;
4012
4013 fn expecting(&self, f: &mut std::fmt::Formatter) -> std::fmt::Result {
4014 f.write_str(
4015 "a provider name (`ingress = \"passway\"`) or a list of edge tables \
4016 (`[[ingress]]`)",
4017 )
4018 }
4019
4020 fn visit_str<E: serde::de::Error>(self, v: &str) -> std::result::Result<Self::Value, E> {
4021 IngressProvider::deserialize(serde::de::value::StrDeserializer::new(v))
4022 .map(IngressDecl::Provider)
4023 }
4024
4025 fn visit_seq<A: serde::de::SeqAccess<'de>>(
4026 self,
4027 seq: A,
4028 ) -> std::result::Result<Self::Value, A::Error> {
4029 Vec::<IngressEdge>::deserialize(serde::de::value::SeqAccessDeserializer::new(seq))
4030 .map(IngressDecl::Edges)
4031 }
4032 }
4033
4034 d.deserialize_any(DeclVisitor)
4035 }
4036}
4037
4038/// No front door — the shape of every mirror that publishes to R2 behind a
4039/// Worker, or runs a mesh-only compute tier.
4040impl Default for IngressDecl {
4041 fn default() -> Self {
4042 Self::Provider(IngressProvider::None)
4043 }
4044}
4045
4046impl IngressDecl {
4047 /// `true` when this mirror declares no front door at all.
4048 pub fn is_absent(&self) -> bool {
4049 match self {
4050 Self::Provider(p) => !p.is_declared(),
4051 Self::Edges(e) => e.is_empty(),
4052 }
4053 }
4054}
4055
4056impl From<IngressProvider> for IngressDecl {
4057 fn from(p: IngressProvider) -> Self {
4058 Self::Provider(p)
4059 }
4060}
4061
4062/// A service mirror — the projection of a [`ServiceConfig`] onto concrete
4063/// infra. Lives at `.yah/services/<svc>/mirrors/<env>.toml`.
4064#[derive(Debug, Clone, Serialize, Deserialize)]
4065#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4066pub struct MirrorConfig {
4067 pub schema_version: u32,
4068 pub shape: MirrorShape,
4069 /// Public-ingress edges fronting this mirror (W267, W305 F2). Defaults to
4070 /// none.
4071 ///
4072 /// Two spellings, one meaning — see [`IngressDecl`]. `ingress = "passway"`
4073 /// is one edge fronting everything; `[[ingress]]` entries declare several,
4074 /// each naming its provider plus the slots or hostnames it fronts. Read it
4075 /// through [`ingress_edges`](Self::ingress_edges), never by matching on the
4076 /// enum, so the two spellings cannot drift apart.
4077 ///
4078 /// Declared at mirror scope rather than per provider slot because a front
4079 /// door does **fan-in**: one `cloudflared` (or one passway) on a node
4080 /// multiplexes every hostname→port rule it fronts, so pinning one to a
4081 /// single slot would mint one edge connection per slot for no gain. An
4082 /// edge's `slots` selector is the general form of that — it groups slots
4083 /// behind one front door, it does not split a front door per slot.
4084 #[serde(default, skip_serializing_if = "IngressDecl::is_absent")]
4085 pub ingress: IngressDecl,
4086 /// Machines the front door is placed on — **independent of where the
4087 /// fronted workload runs** (R330-F37).
4088 ///
4089 /// The single-edge spelling of [`IngressEdge::machines`]: it applies to the
4090 /// one edge `ingress = "<provider>"` declares, and combining it with
4091 /// `[[ingress]]` entries is an error rather than a silent precedence rule.
4092 ///
4093 /// Empty (the default) keeps the pre-existing behaviour: the front door is
4094 /// co-located with the fronted slot's own `machine` / `machines`. That was
4095 /// never a design choice, it was an artifact of bundles binding
4096 /// `127.0.0.1` — nothing off-node could reach a workload, so a proxy had to
4097 /// sit on top of it. R599-F12 landed mesh binding, which removes the
4098 /// constraint: passway is a reverse proxy, and a valid front door needs a
4099 /// cert and an upstream it can *reach*, not a local copy of the service.
4100 ///
4101 /// Listing several machines is what lets the ingress tier and the service
4102 /// tier scale independently — **N front doors over ONE deployment**. There
4103 /// is still exactly one rendered copy of the site, so fanning the front door
4104 /// out introduces no cache-coherence problem; that only appears if you
4105 /// deploy the *workload* to every node instead.
4106 ///
4107 /// ```toml
4108 /// ingress = "passway"
4109 /// ingress_machines = ["us-east-001", "us-west-001"]
4110 /// ```
4111 ///
4112 /// Declaring this without [`ingress`](Self::ingress) is an error, not a
4113 /// no-op — it always means the operator expected a front door somewhere.
4114 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4115 pub ingress_machines: Vec<String>,
4116 /// Provider slots, keyed by role (`"static"`, `"compute"`, …). Each value
4117 /// either references a provider declared under `.yah/infra/providers/` or
4118 /// inlines a local-only provider (no creds, no infra file).
4119 ///
4120 /// A role is normally service-wide — one slot serves every component that
4121 /// shares it — but [`ReconcileCtx::slot`](crate::reconciler::ReconcileCtx::slot)
4122 /// looks up the component-qualified key `"<role>:<component id>"` first.
4123 /// A service with two components of the same role (e.g. two
4124 /// `mesofact-static` components under one mirror) declares
4125 /// `providers."static:<id>"` per component to give each its own port;
4126 /// omitting the qualifier keeps the pre-existing single-slot behavior.
4127 #[serde(default)]
4128 pub providers: BTreeMap<String, MirrorProviderSlot>,
4129 /// Capability→driver bindings, keyed by **capability** (`"pg"`, `"s3"`, …)
4130 /// rather than by slot role (W265 §Drivers).
4131 ///
4132 /// This is the generalization of [`Self::providers`]: `providers.static` /
4133 /// `providers.object_store` are the special case where the slot name and
4134 /// the capability happen to coincide, and keying by capability is what stops
4135 /// the slot enum growing one arm per tier-specific implementation. A service
4136 /// says "I need pg"; the mirror says which implementation of pg *this tier*
4137 /// uses; the app talks the same wire protocol either way and never forks.
4138 ///
4139 /// ```toml
4140 /// [drivers.pg]
4141 /// kind = "local-pg-dev" # dev — kamaji-supervised loopback postgres
4142 /// ```
4143 ///
4144 /// Additive in P1: `drivers` lands *alongside* `providers`, and migrating
4145 /// the existing `providers.static` / `providers.object_store` declarations
4146 /// over is a separate pass (W265 §"Open follow-ups"). A mirror that declares
4147 /// neither is unchanged.
4148 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
4149 pub drivers: BTreeMap<String, MirrorProviderSlot>,
4150 /// Per-environment alias overrides for `kind = "static-asset"` components.
4151 ///
4152 /// Keys are logical names (e.g. `"whisper-default"`); values must be
4153 /// filenames present in the component's `workload.toml` catalog.
4154 /// **Resolution only** — this table may never introduce a filename absent
4155 /// from the catalog. Validated against the workload catalog at sync time.
4156 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
4157 pub asset_aliases: BTreeMap<String, String>,
4158}
4159
4160impl MirrorConfig {
4161 /// This mirror's declared edges, with both spellings normalized (W305 F2).
4162 ///
4163 /// The single place `ingress` + `ingress_machines` are reconciled, so no
4164 /// consumer has to know which spelling was written. Returns an empty vec
4165 /// when the mirror declares no front door.
4166 ///
4167 /// Errors are the declarations that cannot mean anything:
4168 ///
4169 /// - `ingress_machines` with no `ingress` — front-door placement with no
4170 /// front door to place, always a typo (R330-F37);
4171 /// - `ingress_machines` alongside `[[ingress]]` — placement declared twice,
4172 /// in a form where one silently wins;
4173 /// - `provider = "none"` on an edge — an edge that fronts with nothing.
4174 /// The `[[ingress]]` entries exactly as written, without normalizing the
4175 /// scalar spelling or validating anything.
4176 ///
4177 /// [`ingress_edges`](Self::ingress_edges) is the one to reach for; this
4178 /// exists for the checks that must run *before* a mirror is known to be
4179 /// well-formed — cross-reference validation walks every mirror in the
4180 /// workspace, and hard-failing there on an unrelated mirror's shape error
4181 /// would report the wrong file. Empty for the scalar spelling, which has no
4182 /// edge table to carry per-edge fields.
4183 pub fn ingress_edge_slice(&self) -> &[IngressEdge] {
4184 match &self.ingress {
4185 IngressDecl::Edges(edges) => edges,
4186 IngressDecl::Provider(_) => &[],
4187 }
4188 }
4189
4190 pub fn ingress_edges(&self) -> Result<Vec<IngressEdge>> {
4191 match &self.ingress {
4192 IngressDecl::Provider(p) if !p.is_declared() => {
4193 if !self.ingress_machines.is_empty() {
4194 bail!(
4195 "mirror declares `ingress_machines = {:?}` but no `ingress` provider — \
4196 front-door placement with no front door to place. Add \
4197 `ingress = \"passway\"` (or \"cloudflare-tunnel\"), or drop \
4198 `ingress_machines`.",
4199 self.ingress_machines
4200 );
4201 }
4202 Ok(Vec::new())
4203 }
4204 IngressDecl::Provider(p) => Ok(vec![IngressEdge::all_slots(
4205 *p,
4206 self.ingress_machines.clone(),
4207 )]),
4208 IngressDecl::Edges(edges) => {
4209 if !self.ingress_machines.is_empty() {
4210 bail!(
4211 "mirror declares both `[[ingress]]` edges and the single-edge \
4212 `ingress_machines = {:?}` — front-door placement stated twice. Move \
4213 those names onto the edge they place: `machines = [...]` inside the \
4214 `[[ingress]]` entry.",
4215 self.ingress_machines
4216 );
4217 }
4218 for edge in edges {
4219 if !edge.provider.is_declared() {
4220 bail!(
4221 "{}: `provider = \"none\"` fronts nothing. An edge exists to name a \
4222 front door — delete the entry instead.",
4223 edge.label()
4224 );
4225 }
4226 }
4227 Ok(edges.clone())
4228 }
4229 }
4230 }
4231
4232 /// Nodes this mirror's **passway** front doors are placed on, in
4233 /// declaration order and de-duplicated — or `None` when the mirror declares
4234 /// no passway edge at all.
4235 ///
4236 /// `Some(vec![])` is a real and different answer from `None`: a passway edge
4237 /// is declared but names no machine, so its placement falls back to the
4238 /// fronted slot's own. That fallback is placement *resolution* — it belongs
4239 /// to [`IngressRule::machines`](crate::reconciler::IngressRule::machines)
4240 /// and the plan it is built from, not to a mirror read in isolation — so it
4241 /// is reported as "declared, placement unknown from here" rather than
4242 /// half-derived. A caller that needs a node to dial has to say so.
4243 ///
4244 /// Passway-only because the caller is tenant DNS onboarding: only a passway
4245 /// node serves yubaba's `GET /domains/{domain}/onboarding`. A
4246 /// cloudflare-tunnel edge publishes through Cloudflare's own DNS and has no
4247 /// such record to hand a tenant, so folding its machines in would point the
4248 /// UI at a node that cannot answer.
4249 ///
4250 /// Read through [`ingress_edges`](Self::ingress_edges), so both spellings
4251 /// are covered by construction. A declaration that cannot mean anything
4252 /// (`ingress_machines` with no `ingress`, or both spellings at once) reads
4253 /// as `None` rather than propagating an error: those are reported by
4254 /// cross-reference validation, which can name the offending file.
4255 pub fn passway_machines(&self) -> Option<Vec<String>> {
4256 let edges = self.ingress_edges().ok()?;
4257 let mut declared = false;
4258 let mut machines: Vec<String> = Vec::new();
4259 for edge in edges
4260 .iter()
4261 .filter(|e| matches!(e.provider, IngressProvider::Passway))
4262 {
4263 declared = true;
4264 for m in &edge.machines {
4265 if !machines.iter().any(|seen| seen == m) {
4266 machines.push(m.clone());
4267 }
4268 }
4269 }
4270 declared.then_some(machines)
4271 }
4272
4273 /// Parse a single `mirrors/<env>.toml` file.
4274 pub fn load(path: &Path) -> Result<Self> {
4275 let src =
4276 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
4277 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
4278 }
4279
4280 /// Persist to `.yah/services/<service>/mirrors/<env>.toml`, creating the
4281 /// `mirrors/` directory if needed. Create-or-overwrite. The mirror file is
4282 /// named by `env` (its stem); `service` selects the owning service dir.
4283 pub fn save(&self, workspace_root: &Path, service: &str, env: &str) -> Result<()> {
4284 let dir = crate::paths::service_mirrors_dir(workspace_root, service);
4285 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
4286 let path = crate::paths::service_mirror_toml(workspace_root, service, env);
4287 let s = toml::to_string_pretty(self)
4288 .with_context(|| format!("serializing mirror {service}/{env}"))?;
4289 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
4290 }
4291
4292 /// Remove `.yah/services/<service>/mirrors/<env>.toml`. Returns `false`
4293 /// when the file was already absent. Leaves the service and its other
4294 /// mirrors untouched.
4295 ///
4296 /// Also checks legacy stems (e.g. `local-sim` when `env = "pond"`) so
4297 /// deleting a canonical tier name removes whichever file exists on disk.
4298 pub fn delete(workspace_root: &Path, service: &str, env: &str) -> Result<bool> {
4299 let path = crate::paths::service_mirror_toml(workspace_root, service, env);
4300 if path.exists() {
4301 std::fs::remove_file(&path).with_context(|| format!("removing {}", path.display()))?;
4302 return Ok(true);
4303 }
4304 // Try legacy file stems for canonical tier names.
4305 let legacy: &[&str] = match env {
4306 "dev" => &["local"],
4307 "pond" => &["local-sim", "sim"],
4308 "cloud" => &["prod"],
4309 _ => &[],
4310 };
4311 for stem in legacy {
4312 let alt = crate::paths::service_mirror_toml(workspace_root, service, stem);
4313 if alt.exists() {
4314 std::fs::remove_file(&alt)
4315 .with_context(|| format!("removing {}", alt.display()))?;
4316 return Ok(true);
4317 }
4318 }
4319 Ok(false)
4320 }
4321}
4322
4323/// A provider slot inside a [`MirrorConfig`]. Two shapes:
4324/// - **Reference** (`use = "<provider-id>"`) — point at an infra-declared
4325/// provider; extra fields are slot-specific (bucket, zone, dns, …).
4326/// - **Inline** (`kind = "local-*"`) — for providers that need no infra
4327/// declaration because they carry no credentials.
4328#[derive(Debug, Clone, Serialize, Deserialize)]
4329#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4330#[serde(untagged)]
4331pub enum MirrorProviderSlot {
4332 Reference {
4333 #[serde(rename = "use")]
4334 provider_id: String,
4335 #[serde(flatten)]
4336 #[cfg_attr(
4337 feature = "json-schema",
4338 schemars(with = "std::collections::BTreeMap<String, serde_json::Value>")
4339 )]
4340 fields: BTreeMap<String, toml::Value>,
4341 },
4342 Inline {
4343 kind: Provider,
4344 #[serde(flatten)]
4345 #[cfg_attr(
4346 feature = "json-schema",
4347 schemars(with = "std::collections::BTreeMap<String, serde_json::Value>")
4348 )]
4349 fields: BTreeMap<String, toml::Value>,
4350 },
4351}
4352
4353impl MirrorProviderSlot {
4354 /// Provider id this slot references, or `None` for inline slots.
4355 pub fn provider_id(&self) -> Option<&str> {
4356 match self {
4357 Self::Reference { provider_id, .. } => Some(provider_id),
4358 Self::Inline { .. } => None,
4359 }
4360 }
4361
4362 /// Provider kind for inline slots, or `None` for reference slots
4363 /// (resolve via the referenced [`ProviderConfig`]).
4364 pub fn inline_kind(&self) -> Option<Provider> {
4365 match self {
4366 Self::Reference { .. } => None,
4367 Self::Inline { kind, .. } => Some(*kind),
4368 }
4369 }
4370
4371 pub fn fields(&self) -> &BTreeMap<String, toml::Value> {
4372 match self {
4373 Self::Reference { fields, .. } | Self::Inline { fields, .. } => fields,
4374 }
4375 }
4376
4377 /// F16 placement: parse the optional `required = { … }` sub-table on this
4378 /// slot. Returns `None` when absent or unparseable (callers treat as no
4379 /// constraint). See [`RequiredSpec`] for the field grammar.
4380 pub fn required(&self) -> Option<RequiredSpec> {
4381 let v = self.fields().get("required")?.clone();
4382 v.try_into().ok()
4383 }
4384}
4385
4386/// F16 placement constraints declared on a [`MirrorProviderSlot`], lives under
4387/// `[providers.<role>] required = { regions = [...], mesh_tags = [...] }` in
4388/// `mirrors/<env>.toml`.
4389///
4390/// Hard (must-satisfy) axes, all AND-ed together:
4391/// - `regions` / `zones` / `providers` — *membership*: the machine's
4392/// `region` / `zone` / `provider` must be one of the listed values.
4393/// - `mesh_tags` — *superset*: the machine's `mesh_tags` must contain every
4394/// listed tag.
4395/// - `memory_mb` / `cpu_millis` — *capacity floor* (R572-F5): the machine's
4396/// `allocatable` budget must cover the demand. `0` = no constraint.
4397/// - *taint repulsion* — **unconditional** (R876-B7): the machine must not
4398/// carry any taint that [`taint_effect`] classifies as
4399/// [`TaintEffect::Repels`], unless that exact key is listed in
4400/// [`Self::tolerates`]. This axis is not declared; it applies to every spec.
4401/// - `requires_taint` — *taint affinity* (R572-F5): the machine must carry
4402/// this taint key (in `taints` or `mesh_tags`). `None` = no affinity.
4403///
4404/// These are the **only** readers of [`MachineConfig::taints`], which is
4405/// what makes [`taint_effect`]'s closed vocabulary well-founded.
4406///
4407/// [`MachineConfig::sovereign_group`] is deliberately **not** an axis here and
4408/// must not become one (W305/R742-F1). A sovereign group is a blast radius,
4409/// not a filter: which quorum a box votes in says nothing about whether a
4410/// workload may run on it, and a dev-group node exists precisely so dev-mode
4411/// services — stateful ones included — can be scheduled onto it. Filtering on
4412/// it would re-make the mistake W305 exists to undo, where one mechanism
4413/// silently carried three unrelated properties.
4414///
4415/// An empty / zero / None on every axis means "no constraint on that axis".
4416/// A fully-unconstrained `RequiredSpec` matches every machine (see
4417/// [`RequiredSpec::is_unconstrained`]).
4418#[derive(Debug, Clone, Default, Serialize, Deserialize)]
4419#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4420pub struct RequiredSpec {
4421 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4422 pub regions: Vec<String>,
4423 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4424 pub zones: Vec<String>,
4425 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4426 pub providers: Vec<String>,
4427 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4428 pub mesh_tags: Vec<String>,
4429
4430 /// R833-F8: **imperative** placement — the machine must be one of these by
4431 /// `name`. Empty (the default) = no constraint, which is every pre-R833-F8
4432 /// caller.
4433 ///
4434 /// This is the one axis that is not a *capability* the scheduler infers.
4435 /// The operator typed `--where=node:us-west-003`, so it composes with the
4436 /// other axes exactly like the rest — a named node that fails the capacity
4437 /// floor or carries a repelling taint still does not match, and the refusal
4438 /// names why rather than silently placing the work somewhere else.
4439 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4440 pub nodes: Vec<String>,
4441
4442 /// R572-F5: minimum memory (MiB) the target node must have in its
4443 /// declared `allocatable` budget. `0` = no constraint. Filled by
4444 /// [`CloudConfig::admit_workload`] from the workload's
4445 /// `memory_request_mb()` — its placement **request**, which is not the
4446 /// same number as the `resources.memory_mb` cgroup **ceiling**.
4447 #[serde(default, skip_serializing_if = "is_zero_u32")]
4448 pub memory_mb: u32,
4449 /// R572-F5: minimum CPU (millicores) the target node must have in its
4450 /// declared `allocatable` budget. `0` = no constraint. Filled by
4451 /// [`CloudConfig::admit_workload`] from the workload's `resources.cpu_millis`.
4452 #[serde(default, skip_serializing_if = "is_zero_u32")]
4453 pub cpu_millis: u32,
4454 /// R876-B7: repelling node taints this placement **opts back in to**.
4455 /// Each entry is a machine taint key spelled exactly as it appears in
4456 /// [`MachineConfig::taints`] — `"no-appliance"`, not `"appliance"` — so the
4457 /// node side and the workload side share one vocabulary and nothing has to
4458 /// translate between them.
4459 ///
4460 /// # Why this replaced `repel_archetypes`
4461 ///
4462 /// Repulsion used to be **opt-in-to-be-repelled**: the spec named the
4463 /// archetypes it was, and only a `no-<that archetype>` taint blocked it.
4464 /// That field was `#[serde(skip)]`, so a slot declared as
4465 /// `required = { ... }` in a mirror TOML always deserialized with it empty
4466 /// and [`Self::matches`] never read [`MachineConfig::taints`] at all. Node
4467 /// taints were therefore structurally inert for every mirror-declared
4468 /// placement, and inert *silently* — `no-server` is a legal key, so
4469 /// `yah cloud validate` passed and an operator draining a node before
4470 /// maintenance got a green run and a workload that never moved (R876-S2's
4471 /// drill measured exactly this against the real tree).
4472 ///
4473 /// The sense is now inverted, which is the only shape that can survive a
4474 /// field the wire does not carry: **repulsion is unconditional and
4475 /// toleration is declared.** A spec that says nothing is repelled by every
4476 /// repelling taint — the reading an operator writing `taints = ["no-server"]`
4477 /// on a machine already assumed they were getting.
4478 ///
4479 /// Toleration is per-key and absolute; there is no wildcard. Listing a key
4480 /// no machine declares is harmless and matches nothing.
4481 ///
4482 /// [`admission_spec`] fills this from the placement group's archetypes —
4483 /// every repelling key that is *not* the group's own class — which is what
4484 /// makes the `admit_workload` path behave identically across this change
4485 /// (R860-T4 / W338 §Placement consequences 2 still hold: the group's
4486 /// archetypes are the union over `local` requirement edges, so a `Server`
4487 /// bound to an `Appliance` tolerates neither `no-server` nor
4488 /// `no-appliance`).
4489 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4490 pub tolerates: Vec<String>,
4491 /// R572-F5: taint the workload requires the target node to carry
4492 /// (annotation `yah.placement.requires-taint`). The node must have the
4493 /// key in its `taints` list or `mesh_tags`. `None` = no affinity constraint.
4494 #[serde(skip)]
4495 pub requires_taint: Option<String>,
4496
4497 /// R844-F8: **how many** machines this constraint places onto. `None` — the
4498 /// only shape on disk before this field — means one, so every mirror in the
4499 /// tree resolves byte-identically across the change.
4500 ///
4501 /// This is not a match axis: it never appears in [`Self::matches`] and never
4502 /// changes whether a given machine qualifies. It is the *cardinality* of the
4503 /// answer, which is why it lives here rather than as another filter — the
4504 /// operator declares what is required and how many of it, and the scheduler
4505 /// picks which.
4506 ///
4507 /// **Declared, never inferred.** The count is emphatically not "how many
4508 /// machines happen to match": deriving it that way would make adding a box
4509 /// to the fleet silently scale a production front door. A constraint that
4510 /// matches four machines and asks for two places on two.
4511 ///
4512 /// **Fewer matches than asked is an error** ([`select_matching`]), not a
4513 /// partial placement. Placing one of two and reporting success is the
4514 /// subset-that-looks-like-it-worked failure R844 exists to close.
4515 ///
4516 /// Deliberately absent from [`Self::is_unconstrained`], which answers "does
4517 /// every machine match" — a question about the predicate, not the count. A
4518 /// `required = { replicas = 2 }` with no axis is therefore still
4519 /// unconstrained, and the deploy side still refuses it as an
4520 /// underspecified placement.
4521 #[serde(default, skip_serializing_if = "Option::is_none")]
4522 pub replicas: Option<u32>,
4523}
4524
4525impl RequiredSpec {
4526 /// How many machines this constraint places onto — [`Self::replicas`],
4527 /// resolving the absent case to the pre-R844-F8 answer of one.
4528 ///
4529 /// The single place that default is spelled, so the ingress planner and the
4530 /// deploy resolver cannot disagree about what "no replica count" means.
4531 pub fn replica_count(&self) -> usize {
4532 self.replicas.unwrap_or(1) as usize
4533 }
4534
4535 /// True when no *declared* axis carries a constraint — every untainted
4536 /// machine matches.
4537 ///
4538 /// R876-B7: taint repulsion is deliberately absent from this conjunction,
4539 /// unlike the `repel_archetypes` it replaced. Repulsion is no longer an axis
4540 /// a spec declares — it applies to every spec — so including it would make
4541 /// the answer a property of the fleet rather than of the constraint. Nor
4542 /// does [`Self::tolerates`] belong here: a toleration *widens* the candidate
4543 /// set, and the callers of this predicate ask "did the operator narrow
4544 /// anything" in order to refuse an underspecified placement. A slot that
4545 /// declares only a toleration has still narrowed nothing.
4546 pub fn is_unconstrained(&self) -> bool {
4547 self.regions.is_empty()
4548 && self.zones.is_empty()
4549 && self.providers.is_empty()
4550 && self.mesh_tags.is_empty()
4551 && self.nodes.is_empty()
4552 && self.memory_mb == 0
4553 && self.cpu_millis == 0
4554 && self.requires_taint.is_none()
4555 }
4556
4557 /// Whether `machine` satisfies every hard axis.
4558 ///
4559 /// - Membership axes (region/zone/provider): machine must carry the field
4560 /// and it must appear in the constraint list.
4561 /// - `mesh_tags`: machine tags must be a superset of the required set.
4562 /// - **R572-F5 capacity floor**: `machine.allocatable.{memory,cpu}` must
4563 /// cover `self.{memory,cpu}`. A machine with no `allocatable` block passes
4564 /// unconditionally (capacity unknown → no constraint enforced).
4565 /// - **Taint repulsion (R876-B7)**: machine must not carry *any* taint that
4566 /// [`taint_effect`] classifies as [`TaintEffect::Repels`], unless that key
4567 /// is listed in [`Self::tolerates`]. Applied unconditionally — this is the
4568 /// axis no spec has to declare, and the one that makes a node drainable.
4569 /// - **R572-F5 taint affinity**: if `requires_taint` is set, the machine
4570 /// must carry that key in its `taints` list or `mesh_tags`.
4571 ///
4572 /// A [`TaintEffect::Attracts`] key (today just `public-ip`) does **not**
4573 /// repel: it is the affinity vocabulary, so reading it as repulsion would
4574 /// evict every workload from the three nodes that carry it. Only the
4575 /// `no-<archetype>` class repels, and [`taint_effect`] is the single
4576 /// authority on which is which — which is why
4577 /// [`crate::validate::check_inert_taints`] refuses to let an unclassifiable
4578 /// key be declared: it would read as a constraint and be none.
4579 pub fn matches(&self, machine: &MachineConfig) -> bool {
4580 let member_ok = |constraint: &[String], value: Option<&str>| -> bool {
4581 constraint.is_empty() || value.map_or(false, |v| constraint.iter().any(|c| c == v))
4582 };
4583
4584 // R833-F8: imperative node pin, checked first because it is the axis a
4585 // human asserted rather than one the scheduler derived — a refusal
4586 // should read "us-west-003 does not match" and not lead with a tag set
4587 // the operator never typed.
4588 if !member_ok(&self.nodes, Some(machine.name.as_str())) {
4589 return false;
4590 }
4591
4592 // Membership + mesh-tags (pre-existing axes).
4593 if !member_ok(&self.regions, machine.region.as_deref())
4594 || !member_ok(&self.zones, machine.zone.as_deref())
4595 || !member_ok(&self.providers, Some(machine.provider.as_str()))
4596 || !self
4597 .mesh_tags
4598 .iter()
4599 .all(|t| machine.mesh_tags.iter().any(|mt| mt == t))
4600 {
4601 return false;
4602 }
4603
4604 // R572-F5: capacity floor. Skipped when machine has no allocatable
4605 // declaration (unknown capacity → passes, consistent with pre-F5 behaviour).
4606 if self.memory_mb > 0 || self.cpu_millis > 0 {
4607 if let Some(alloc) = &machine.allocatable {
4608 if self.memory_mb > alloc.memory_mb || self.cpu_millis > alloc.cpu_millis {
4609 return false;
4610 }
4611 }
4612 }
4613
4614 // R876-B7: taint repulsion, repel-by-default. Every repelling taint on
4615 // the machine blocks placement unless this spec names it in
4616 // `tolerates`. Driven off `machine.taints` rather than off a field of
4617 // `self`, which is the whole point: a spec that arrives by deserializing
4618 // a mirror's `required = {...}` carries no repulsion declaration and
4619 // never could, so making repulsion conditional on one made node taints
4620 // structurally unreadable on that path (R876-S2).
4621 for taint in &machine.taints {
4622 if !matches!(taint_effect(taint), TaintEffect::Repels(_)) {
4623 continue;
4624 }
4625 if !self.tolerates.iter().any(|t| t == taint) {
4626 return false;
4627 }
4628 }
4629
4630 // R572-F5: taint affinity. Machine must carry the required taint key
4631 // in either its `taints` list or `mesh_tags`.
4632 if let Some(req) = &self.requires_taint {
4633 let has_it = machine.taints.iter().any(|t| t == req)
4634 || machine.mesh_tags.iter().any(|t| t == req);
4635 if !has_it {
4636 return false;
4637 }
4638 }
4639
4640 true
4641 }
4642
4643 /// Human-readable summary of the constraints, for fail-loud error messages.
4644 /// Example: `required.regions=[us-west] + required.mesh_tags=[tag:cloud-runner]`.
4645 pub fn describe(&self) -> String {
4646 let mut parts = Vec::new();
4647 let mut push = |label: &str, vals: &[String]| {
4648 if !vals.is_empty() {
4649 parts.push(format!("required.{label}=[{}]", vals.join(",")));
4650 }
4651 };
4652 push("nodes", &self.nodes);
4653 push("regions", &self.regions);
4654 push("zones", &self.zones);
4655 push("providers", &self.providers);
4656 push("mesh_tags", &self.mesh_tags);
4657 // Kept with the other list axes, and NOT moved below: `push` borrows
4658 // `parts` mutably for as long as it is live, so interleaving it with the
4659 // direct `parts.push` calls under it does not compile.
4660 push("tolerates", &self.tolerates);
4661 if self.memory_mb > 0 {
4662 parts.push(format!("memory_mb>={}", self.memory_mb));
4663 }
4664 if self.cpu_millis > 0 {
4665 parts.push(format!("cpu_millis>={}", self.cpu_millis));
4666 }
4667 if let Some(req) = &self.requires_taint {
4668 parts.push(format!("requires_taint={req}"));
4669 }
4670 if parts.is_empty() {
4671 "no constraints".to_string()
4672 } else {
4673 parts.join(" + ")
4674 }
4675 }
4676}
4677
4678/// Which front door actually serves a domain's requests (R594-F12).
4679///
4680/// Every domain manifest must say this out loud. Before it existed the
4681/// difference between "R2 serves this hostname directly" and "a Worker
4682/// serves it" was expressed *only* by whether the file happened to carry
4683/// `[[routes]]` — so binding a route-carrying domain straight to R2 was
4684/// accepted silently and served 200s on its SSG half while losing clean
4685/// URLs, SPA shell fallback, deferred-route pointers and branded error
4686/// pages. All of those live in the Worker
4687/// (`oss/mesofact/packages/mesofact-edge/src/router.ts`) or in
4688/// mesofact-serve; an R2 custom domain has none of them.
4689///
4690/// The vocabulary mirrors `scripts/cf-apex-mode.sh` (worker | grey | orange)
4691/// — this moves the choice into the config where it can be checked instead
4692/// of living in one bash script.
4693///
4694/// A front door does **fan-in** only. The render cube (SSG / SPA / SSR /
4695/// deferred / 404) is mesofact's manifest, not this one — see W173 and
4696/// `.yah/docs/working/W267-sovereign-public-ingress.md`
4697/// §"Two front doors, one render contract".
4698#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
4699#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4700#[serde(rename_all = "kebab-case")]
4701pub enum FrontDoor {
4702 /// Cloudflare R2 custom domain. Requests hit R2 objects with edge
4703 /// caching and nothing else — no clean URLs, no SPA fallback, no
4704 /// branded errors. Correct for a pure asset tier (W175's verdict for
4705 /// `cdn.yah.dev`) and wrong for anything that renders pages.
4706 /// Implies zero `[[routes]]` and no `worker_bundle_path`.
4707 BucketDirect,
4708 /// Cloudflare Worker generated from this manifest's route table.
4709 Worker,
4710 /// Sovereign L7 ingress — the `passway` proxy on yah-owned metal
4711 /// (`oss/passway`, W267). Same route table as `worker`; different
4712 /// machine terminates TLS.
4713 Passway,
4714}
4715
4716impl FrontDoor {
4717 /// Whether this front door consumes the manifest's `[[routes]]` table.
4718 /// `bucket-direct` does not; the other two are nothing without it.
4719 pub fn is_route_driven(self) -> bool {
4720 matches!(self, FrontDoor::Worker | FrontDoor::Passway)
4721 }
4722
4723 /// The manifest spelling, for error messages.
4724 pub fn as_str(self) -> &'static str {
4725 match self {
4726 FrontDoor::BucketDirect => "bucket-direct",
4727 FrontDoor::Worker => "worker",
4728 FrontDoor::Passway => "passway",
4729 }
4730 }
4731}
4732
4733/// A routing manifest for one domain, from `.yah/domains/<name>.toml`.
4734///
4735/// The domain manifest is the *only* place that knows about path routing:
4736/// services declare static/backend components by opaque ID, and this
4737/// manifest binds those components to URL paths on a public-facing
4738/// domain. Generated Worker bundles consume this. See
4739/// `.yah/docs/working/W118-yah-domain-tiers.md` (R347).
4740#[derive(Debug, Clone, Serialize, Deserialize)]
4741#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4742pub struct DomainConfig {
4743 pub schema_version: u32,
4744 /// Stable identifier for this domain (file stem of the manifest).
4745 /// Example: `"yah-dev"` for the `yah.dev` zone.
4746 pub name: String,
4747 /// The fully-qualified domain this manifest routes for. Example:
4748 /// `"yah.dev"`, `"app.yah.dev"`.
4749 pub domain: String,
4750 /// Which front door serves this domain (R594-F12). **Required** — a
4751 /// default here would silently re-create the defect the field exists to
4752 /// close. Cross-checked against `routes` / `worker_bundle_path` by
4753 /// [`DomainConfig::validate_front_door`] at load time.
4754 pub front_door: FrontDoor,
4755 /// Public CDN bucket name. Static-mode route components publish into
4756 /// this bucket. Owned by the domain, *not* by any single service.
4757 pub cdn_bucket: String,
4758 /// Optional path (relative to workspace root) where the generated
4759 /// Worker bundle lands. `None` while the bundle generator (R347-F4)
4760 /// is still being wired up.
4761 #[serde(default, skip_serializing_if = "Option::is_none")]
4762 pub worker_bundle_path: Option<String>,
4763 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4764 pub routes: Vec<DomainRoute>,
4765}
4766
4767/// One entry in a [`DomainConfig`]'s route table.
4768///
4769/// The `mode` discriminator picks the variant's body via serde's
4770/// internally-tagged enum representation. Path patterns follow the
4771/// Worker convention: a trailing `*` matches everything underneath.
4772#[derive(Debug, Clone, Serialize, Deserialize)]
4773#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4774pub struct DomainRoute {
4775 /// URL pattern this route matches. Examples: `"/"`, `"/dashboard/*"`,
4776 /// `"/camp/ws"`.
4777 pub path: String,
4778 /// Response headers the front door sets on every response served under
4779 /// this route (R746). Empty by default.
4780 ///
4781 /// This is the manifest's answer to "who decides a path's response
4782 /// headers". Before it existed the answer was *nobody*: a `_headers` file
4783 /// is a Cloudflare Pages / Netlify convention, and neither of this
4784 /// repo's front doors reads one — a Worker returns what it fetched from
4785 /// R2, and R2 serves only the object's own httpMetadata. So a site could
4786 /// carry a `_headers` file declaring COOP/COEP and ship without them,
4787 /// which is exactly how it was found: `SharedArrayBuffer` is simply
4788 /// absent in a document served cross-origin-isolation-free, with no
4789 /// error anywhere to say why.
4790 ///
4791 /// Deliberately a free-form `name -> value` map rather than named fields
4792 /// for the isolation headers: the domain manifest has no business
4793 /// knowing which headers a route's payload happens to need. Ordering
4794 /// follows the route table's own rule — first matching route wins, no
4795 /// merging across routes (see the Worker's `applyRouteHeaders`).
4796 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
4797 pub headers: BTreeMap<String, String>,
4798 #[serde(flatten)]
4799 pub mode: RouteMode,
4800}
4801
4802/// Body of a [`DomainRoute`]. Three modes:
4803/// - **Static** — Worker reads from the domain's CDN bucket. Component
4804/// ref points at a `kind = "mesofact-static"` (or similar) service
4805/// component.
4806/// - **Backend** — Worker proxies to an HTTP origin owned by a backend
4807/// component (yubaba workload, gateway, etc.).
4808/// - **Redirect** — Worker emits a 30x to the target URL. Used to keep
4809/// old paths alive during domain refactors.
4810#[derive(Debug, Clone, Serialize, Deserialize)]
4811#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4812#[serde(tag = "mode", rename_all = "kebab-case")]
4813pub enum RouteMode {
4814 Static {
4815 /// Component reference `"<service>/<component-id>"`. Validated
4816 /// at [`CloudConfig::load`] time.
4817 component: String,
4818 },
4819 Backend {
4820 /// Component reference `"<service>/<component-id>"`. Validated
4821 /// at [`CloudConfig::load`] time.
4822 component: String,
4823 /// Origin URL the Worker `fetch()`es. Schema-permissive — could
4824 /// be `https://...`, `wss://...`, or a yah-internal mesh URL
4825 /// resolved by yubaba.
4826 origin: String,
4827 },
4828 Redirect {
4829 /// Absolute URL or path the Worker emits a 30x to.
4830 target: String,
4831 /// HTTP status code. Defaults to 308 (permanent + method-preserving)
4832 /// so deprecations don't silently turn POSTs into GETs.
4833 #[serde(default = "default_redirect_status")]
4834 status: u16,
4835 },
4836}
4837
4838fn default_redirect_status() -> u16 {
4839 308
4840}
4841
4842/// Normalize a component `mount` to a storage/URL key prefix: strip the
4843/// surrounding slashes. `"/app"`, `"app/"`, `"/app/"` → `"app"`; `"/"`, `""`
4844/// → `""` (the service root).
4845///
4846/// One producer on purpose — the publisher's key prefix, the route-path
4847/// cross-check and the front door's key lookup must all agree on what `/app`
4848/// means down to the byte, and three copies of `trim_matches('/')` is how they
4849/// stop agreeing.
4850pub fn normalize_mount(raw: &str) -> String {
4851 raw.trim_matches('/').to_string()
4852}
4853
4854/// The key prefix a domain route pattern serves under: `"/*"` → `""`,
4855/// `"/app/*"` and `"/app"` → `"app"`. The twin of [`normalize_mount`] on the
4856/// routing side.
4857pub fn route_path_prefix(path: &str) -> String {
4858 normalize_mount(path.strip_suffix('*').unwrap_or(path))
4859}
4860
4861/// The route-driven domain whose route table binds a component of `service`,
4862/// if any. Used by static publishers to pick up the per-route response
4863/// headers a service's paths were declared with.
4864///
4865/// Deterministic by `BTreeMap` key order when more than one domain routes the
4866/// same service (a legitimate shape: an apex and a staging host serving one
4867/// bundle). Returning the first is a real limitation, not a considered
4868/// choice — the day two such domains want *different* headers for one
4869/// component, this needs the domain identity threaded in rather than inferred.
4870pub fn domain_serving_service<'a>(
4871 domains: &'a BTreeMap<String, DomainConfig>,
4872 service: &str,
4873) -> Option<&'a DomainConfig> {
4874 domains
4875 .values()
4876 .find(|d| d.front_door.is_route_driven() && d.serves_service(service))
4877}
4878
4879/// The `ROUTE_HEADERS` Worker-binding value for `service`, read from the
4880/// workspace's domain manifests. `"[]"` when no route-driven domain routes the
4881/// service, or when the one that does declares no headers.
4882///
4883/// Reads `.yah/domains/` directly rather than taking a loaded [`CloudConfig`]:
4884/// the static reconcilers are handed a per-component [`ReconcileCtx`], not the
4885/// whole workspace config, and threading a config reference through all 22 of
4886/// its construction sites to reach one string would be a wide change for a
4887/// narrow read. Manifest parse errors propagate — a domain file that no longer
4888/// loads is a deploy-stopping fact, not a reason to ship a Worker with the
4889/// headers quietly missing.
4890pub fn route_headers_for_service(workspace_root: &Path, service: &str) -> Result<String> {
4891 let domains = load_domains(&crate::paths::domains_dir(workspace_root))?;
4892 Ok(domain_serving_service(&domains, service)
4893 .map(DomainConfig::route_headers_json)
4894 .unwrap_or_else(|| "[]".to_string()))
4895}
4896
4897impl DomainConfig {
4898 /// Parse a single `.yah/domains/<name>.toml`, rejecting a manifest whose
4899 /// declared front door contradicts its route table
4900 /// ([`Self::validate_front_door`]).
4901 pub fn load(path: &Path) -> Result<Self> {
4902 let src =
4903 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
4904 let dom: Self =
4905 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
4906 dom.validate_front_door()
4907 .with_context(|| format!("validating {}", path.display()))?;
4908 dom.validate_route_headers()
4909 .with_context(|| format!("validating {}", path.display()))?;
4910 Ok(dom)
4911 }
4912
4913 /// R594-F12 — the front door must agree with the rest of the manifest.
4914 ///
4915 /// - `bucket-direct` is an R2 custom domain: a Worker route table would
4916 /// never be consulted, so declaring one means the author expected
4917 /// Worker behaviour (clean URLs, SPA fallback, branded errors) from a
4918 /// surface that cannot provide it. Rejected rather than silently
4919 /// ignored. Same for `worker_bundle_path` — nothing would deploy it.
4920 /// - `worker` / `passway` with an empty route table is a silent 404
4921 /// machine: the front door exists, has nothing to serve, and every
4922 /// request falls through to the catch-all.
4923 ///
4924 /// Called from [`Self::load`], so both [`CloudConfig::load`] and
4925 /// [`CloudConfig::load_from_config_dir`] enforce it.
4926 pub fn validate_front_door(&self) -> Result<()> {
4927 match self.front_door {
4928 FrontDoor::BucketDirect => {
4929 if let Some(route) = self.routes.first() {
4930 anyhow::bail!(
4931 "front_door = \"bucket-direct\" but routes[0].path = \"{}\" — \
4932 an R2 custom domain never consults a route table, so this \
4933 route would silently do nothing (no clean URLs, no SPA \
4934 fallback, no branded errors). Set front_door = \"worker\" \
4935 (or \"passway\") to keep the routes, or drop the [[routes]] \
4936 to keep the bucket-direct binding.",
4937 route.path
4938 );
4939 }
4940 if let Some(path) = &self.worker_bundle_path {
4941 anyhow::bail!(
4942 "front_door = \"bucket-direct\" but worker_bundle_path = \
4943 \"{path}\" — nothing deploys a Worker bundle for a domain \
4944 bound straight to R2"
4945 );
4946 }
4947 }
4948 FrontDoor::Worker | FrontDoor::Passway => {
4949 if self.routes.is_empty() {
4950 anyhow::bail!(
4951 "front_door = \"{}\" but [[routes]] is empty — a front door \
4952 with no route table is a silent 404 machine. Declare at \
4953 least one route, or set front_door = \"bucket-direct\" if \
4954 this domain really is served straight from R2.",
4955 self.front_door.as_str()
4956 );
4957 }
4958 }
4959 }
4960 Ok(())
4961 }
4962
4963 /// The `ROUTE_HEADERS` Worker binding for this domain (R746) — the route
4964 /// table's `path` + `headers` pairs, in manifest order, with routes that
4965 /// declare no headers dropped. `"[]"` when nothing declares any.
4966 ///
4967 /// Order is load-bearing and must survive serialization: the front door
4968 /// applies the FIRST matching rule, so `/app/*` above `/*` is what gives
4969 /// the app its isolation headers and leaves the marketing site alone.
4970 /// That is why this is a `Vec` of pairs and not a map keyed by path.
4971 ///
4972 /// Infallible by design — [`Self::validate_route_headers`] has already run
4973 /// at [`Self::load`], so by the time a reconciler calls this the table is
4974 /// known to be one both front doors can apply.
4975 pub fn route_headers_json(&self) -> String {
4976 #[derive(Serialize)]
4977 struct Rule<'a> {
4978 path: &'a str,
4979 headers: &'a BTreeMap<String, String>,
4980 }
4981 let rules: Vec<Rule<'_>> = self
4982 .routes
4983 .iter()
4984 .filter(|r| !r.headers.is_empty())
4985 .map(|r| Rule {
4986 path: &r.path,
4987 headers: &r.headers,
4988 })
4989 .collect();
4990 serde_json::to_string(&rules).unwrap_or_else(|_| "[]".to_string())
4991 }
4992
4993 /// R749-T5 — everything [`Self::route_headers_json`] emits must be
4994 /// *applicable*, checked here where the table is PRODUCED.
4995 ///
4996 /// That method serializes a typed struct, so the table's JSON *shape* is
4997 /// sound by construction. Its contents are not: a route's `headers` map is
4998 /// a free-form `name -> value` read verbatim out of hand-written TOML, so
4999 /// `"Cross Origin Opener Policy"` (spaces instead of hyphens) or a value
5000 /// carrying a newline ships a structurally-valid table that neither front
5001 /// door can apply — and they fail *differently*, neither naming the
5002 /// manifest line responsible:
5003 ///
5004 /// - **passway** — `mesofact::route_headers::RouteHeaderTable::parse`
5005 /// refuses the start, so the origin is simply down.
5006 /// - **worker** — `validateRouteHeaderTable` accepts it (it checks shape,
5007 /// not header validity) and `applyRouteHeaders` then throws inside the
5008 /// exported `fetch`, which is a 500 on every request, not the
5009 /// serve-without-the-headers degradation that code intends.
5010 ///
5011 /// So the strictness lives at the producer: a table that cannot be applied
5012 /// fails `yah cloud apply` at manifest load, naming domain, route and
5013 /// header. This is deliberately *not* a second parser — the check is
5014 /// `HeaderName`/`HeaderValue`'s own, the very constructors the passway door
5015 /// runs on the far side, and route *matching* semantics stay defined once,
5016 /// at the doors. Only routes that contribute to the table are checked, so
5017 /// the invariant is exactly "`route_headers_json`'s output parses".
5018 ///
5019 /// Called from [`Self::load`], alongside [`Self::validate_front_door`].
5020 pub fn validate_route_headers(&self) -> Result<()> {
5021 use axum::http::{HeaderName, HeaderValue};
5022
5023 for route in self.routes.iter().filter(|r| !r.headers.is_empty()) {
5024 if route.path.is_empty() {
5025 anyhow::bail!(
5026 "domain \"{}\" declares response headers on a route whose `path` is \
5027 empty — a rule that matches nothing (or everything, depending on \
5028 which front door reads it) is not a policy",
5029 self.name
5030 );
5031 }
5032 for (name, value) in &route.headers {
5033 HeaderName::try_from(name.as_str()).with_context(|| {
5034 format!(
5035 "domain \"{}\" route \"{}\" declares {name:?}, which is not a valid \
5036 HTTP header name — names are token characters only, so it is \
5037 `Cross-Origin-Opener-Policy`, never `Cross Origin Opener Policy`",
5038 self.name, route.path
5039 )
5040 })?;
5041 HeaderValue::try_from(value.as_str()).with_context(|| {
5042 format!(
5043 "domain \"{}\" route \"{}\" declares {name} = {value:?}, which is not \
5044 a valid HTTP header value — no newlines and no control characters",
5045 self.name, route.path
5046 )
5047 })?;
5048 }
5049 }
5050 Ok(())
5051 }
5052
5053 /// Whether this domain's route table binds any component of `service`.
5054 pub fn serves_service(&self, service: &str) -> bool {
5055 self.routes.iter().any(|r| {
5056 r.mode
5057 .component()
5058 .and_then(split_component_ref)
5059 .is_some_and(|(svc, _)| svc == service)
5060 })
5061 }
5062
5063 /// Persist to `.yah/domains/<name>.toml`, creating the domains
5064 /// directory if needed. Create-or-overwrite.
5065 pub fn save(&self, workspace_root: &Path) -> Result<()> {
5066 let dir = crate::paths::domains_dir(workspace_root);
5067 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
5068 let path = crate::paths::domain_toml(workspace_root, &self.name);
5069 let s = toml::to_string_pretty(self)
5070 .with_context(|| format!("serializing domain {}", self.name))?;
5071 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
5072 }
5073
5074 /// Remove `.yah/domains/<name>.toml`. Returns `false` when the file
5075 /// was already absent.
5076 pub fn delete(workspace_root: &Path, name: &str) -> Result<bool> {
5077 let path = crate::paths::domain_toml(workspace_root, name);
5078 if !path.exists() {
5079 return Ok(false);
5080 }
5081 std::fs::remove_file(&path).with_context(|| format!("removing {}", path.display()))?;
5082 Ok(true)
5083 }
5084}
5085
5086impl RouteMode {
5087 /// Component reference for static/backend modes; `None` for redirects.
5088 pub fn component(&self) -> Option<&str> {
5089 match self {
5090 Self::Static { component } | Self::Backend { component, .. } => Some(component),
5091 Self::Redirect { .. } => None,
5092 }
5093 }
5094}
5095
5096// ─── Service-group vault (R706 / W294) ───────────────────────────────────────
5097
5098/// A camp's declaration of one cluster secret, from
5099/// `.yah/infra/secrets/<slug>.toml`.
5100///
5101/// This is the *authoring* side of the fleet's cluster-secret store: it names
5102/// where the value lives in the camp (a `fob` vault slot), what the fleet should
5103/// call it, and — the point of R706 — which workloads are allowed to mount it.
5104///
5105/// The declaration is not itself the enforcement point. `yah cloud secret put`
5106/// reads this file, seals the vault value under the cluster KEK, and ships the
5107/// ciphertext **with its access rule** into raft; yubaba's `ClusterResolver`
5108/// evaluates the rule on the node at mount time. Deleting this file does not
5109/// revoke anything — the record in raft is the live authority. That asymmetry is
5110/// deliberate: a rule that lived only in a git-tracked camp file would be
5111/// trivially bypassed by anyone who could reach the fleet without the camp.
5112///
5113/// ```toml
5114/// #:schema ../../schema/secret.toml.schema.json
5115/// schema_version = 1
5116/// name = "cheers/cloud-admin/verify-key"
5117/// vault_slot = "cheers-cloud-admin-verify-key"
5118/// description = "Ed25519 public key yah-cloud-admin verifies operator PASETOs with"
5119///
5120/// [access]
5121/// workloads = [{ workload = "yah-cloud-admin" }]
5122///
5123/// [target]
5124/// kind = "file"
5125/// path = "/run/secrets/cheers-verify.key"
5126/// mode = 0o400
5127/// ```
5128#[derive(Debug, Clone, Serialize, Deserialize)]
5129#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5130pub struct SecretConfig {
5131 pub schema_version: u32,
5132
5133 /// Logical cluster-secret key, as `SecretRef::Cluster { name }` spells it —
5134 /// e.g. `"tls/yah.dev/cert"`, `"cheers/cloud-admin/verify-key"`. May contain
5135 /// `/`; the file stem is a filesystem-safe slug and carries no meaning.
5136 pub name: String,
5137
5138 /// The `fob` vault slot in this camp holding the plaintext value. Read by
5139 /// `yah cloud secret put` at ship time and never recorded anywhere else — in
5140 /// particular the value is not in this file, so the declaration is safe to
5141 /// commit.
5142 pub vault_slot: String,
5143
5144 /// Human note for `yah cloud secret ls`. What this secret is and who minted
5145 /// it — the thing nobody remembers 6 months later.
5146 #[serde(default, skip_serializing_if = "Option::is_none")]
5147 pub description: Option<String>,
5148
5149 /// How the vault slot's text decodes into the bytes the consumer expects.
5150 ///
5151 /// `fob` slots hold strings, but plenty of real secrets are **binary** — an
5152 /// Ed25519 key is exactly 32 raw bytes, and `yah-cloud-admin` rejects a key
5153 /// file of any other length. Without this field the only way to ship such a
5154 /// key would be to hope its bytes happened to be valid UTF-8, which for a
5155 /// random key they are not.
5156 ///
5157 /// Defaults to [`SecretEncoding::Utf8`] — the right answer for tokens,
5158 /// passwords, and PEM, which is most secrets.
5159 #[serde(default)]
5160 pub encoding: SecretEncoding,
5161
5162 /// Who may mount it. Stamped onto the raft record verbatim.
5163 ///
5164 /// Defaults to [`SecretAccess::default`] — the deny-all empty allow-list. A
5165 /// declaration that forgets this field produces a secret nobody can mount,
5166 /// which is the correct direction to fail in.
5167 ///
5168 /// Three forms:
5169 ///
5170 /// ```toml
5171 /// access = "allow_any" # explicit escape hatch
5172 ///
5173 /// [access] # named workloads
5174 /// workloads = [{ workload = "yah-cloud-admin" }]
5175 ///
5176 /// [access] # signed recipes (R555-F5)
5177 /// recipes = [{ recipe = "rusty-v8-musl", key = "3d40…" }]
5178 /// ```
5179 ///
5180 /// Use the `recipes` form for a credential a **dispatched build** needs (the
5181 /// R2 write key, the cosign signing key). A remote QED run's workload name
5182 /// is a fresh `forge-<uuid>` every time, so `workloads` cannot name it and
5183 /// `allow_any` over-answers — see W235 §Seam (c) secret scoping. `key` is
5184 /// the hex Ed25519 public key from the recipe's `[admission]` block.
5185 #[serde(default)]
5186 pub access: SecretAccess,
5187
5188 /// Advisory: the mount shape a consuming workload should declare. Not
5189 /// enforced — yubaba honours whatever the `WorkloadSpec` asks for — but it
5190 /// lets `yah cloud secret put` print the exact `SecretMount` to paste, so
5191 /// the consumer and the declaration can't drift on path or mode.
5192 #[serde(default, skip_serializing_if = "Option::is_none")]
5193 pub target: Option<SecretTargetDecl>,
5194}
5195
5196/// How a [`SecretConfig`]'s vault text becomes the bytes delivered to the
5197/// container (R706 / W294).
5198#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
5199#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5200#[serde(rename_all = "kebab-case")]
5201pub enum SecretEncoding {
5202 /// Ship the vault string's UTF-8 bytes verbatim. Tokens, passwords, PEM.
5203 #[default]
5204 Utf8,
5205 /// The vault string is hex; ship the decoded bytes. Use for binary key
5206 /// material — e.g. a raw Ed25519 key, which must land as exactly 32 bytes.
5207 Hex,
5208}
5209
5210/// Advisory mount shape on a [`SecretConfig`]. Mirrors
5211/// `workload_spec::SecretTarget` in a TOML-friendly, externally-tagged-free
5212/// shape (a `kind` discriminator reads better in a hand-written manifest than
5213/// serde's default enum encoding).
5214#[derive(Debug, Clone, Serialize, Deserialize)]
5215#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5216#[serde(tag = "kind", rename_all = "kebab-case")]
5217pub enum SecretTargetDecl {
5218 /// Mounted as a tmpfs-backed file inside the container.
5219 File {
5220 /// Absolute path inside the container.
5221 path: String,
5222 /// Unix permission bits. Defaults to `0o400` (owner-read-only).
5223 #[serde(default = "default_secret_mode")]
5224 mode: u32,
5225 },
5226 /// Injected as an environment variable. Prefer `file` — env vars leak
5227 /// through subprocess environments and log dumps.
5228 EnvVar { name: String },
5229}
5230
5231fn default_secret_mode() -> u32 {
5232 0o400
5233}
5234
5235impl SecretTargetDecl {
5236 /// The `workload_spec` target this declaration describes.
5237 pub fn to_target(&self) -> workload_spec::SecretTarget {
5238 match self {
5239 Self::File { path, mode } => workload_spec::SecretTarget::File {
5240 path: path.into(),
5241 mode: *mode,
5242 },
5243 Self::EnvVar { name } => workload_spec::SecretTarget::EnvVar { name: name.clone() },
5244 }
5245 }
5246}
5247
5248impl SecretConfig {
5249 /// Parse a single `.yah/infra/secrets/<slug>.toml`.
5250 pub fn load(path: &Path) -> Result<Self> {
5251 let src =
5252 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
5253 let cfg: Self =
5254 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
5255 cfg.validate()
5256 .with_context(|| format!("validating {}", path.display()))?;
5257 Ok(cfg)
5258 }
5259
5260 /// Load every declaration in `dir`, keyed by logical secret name. A missing
5261 /// directory is an empty map (a camp with no cluster secrets is normal).
5262 ///
5263 /// Two files declaring the same `name` is a hard error, not a last-writer-
5264 /// wins merge: they would race to define the access rule for one record, and
5265 /// whichever lost would look correct in git while being inert on the fleet.
5266 pub fn load_dir(dir: &Path) -> Result<BTreeMap<String, Self>> {
5267 let mut out: BTreeMap<String, Self> = BTreeMap::new();
5268 if !dir.exists() {
5269 return Ok(out);
5270 }
5271 for entry in std::fs::read_dir(dir).with_context(|| format!("reading {}", dir.display()))? {
5272 let path = entry?.path();
5273 if path.extension().is_none_or(|e| e != "toml") {
5274 continue;
5275 }
5276 let cfg = Self::load(&path)?;
5277 if let Some(prev) = out.insert(cfg.name.clone(), cfg) {
5278 anyhow::bail!(
5279 "two secret declarations both claim name {:?} (one of them is {}); \
5280 a cluster secret must have exactly one declaration so its access \
5281 rule has one author",
5282 prev.name,
5283 path.display()
5284 );
5285 }
5286 }
5287 Ok(out)
5288 }
5289
5290 /// Reject declarations that would produce an unusable or dangerous record.
5291 pub fn validate(&self) -> Result<()> {
5292 if self.name.trim().is_empty() {
5293 anyhow::bail!("`name` must not be empty");
5294 }
5295 if self.vault_slot.trim().is_empty() {
5296 anyhow::bail!(
5297 "`vault_slot` must not be empty — it names the fob slot holding the value"
5298 );
5299 }
5300 // A deny-all rule is a *valid* record (it is the fail-closed default the
5301 // resolver relies on) but it is never a useful thing to deliberately
5302 // ship, so catching it here saves an operator the round-trip of
5303 // deploying a workload that mysteriously can't see its own secret.
5304 if let SecretAccess::Workloads(entries) = &self.access {
5305 if entries.is_empty() {
5306 anyhow::bail!(
5307 "`[access]` admits nobody: list the workloads allowed to mount {:?} \
5308 (e.g. `workloads = [{{ workload = \"my-service\" }}]`), or set \
5309 `access = \"allow_any\"` to store it unrestricted",
5310 self.name
5311 );
5312 }
5313 if let Some(bad) = entries.iter().find(|e| e.workload.trim().is_empty()) {
5314 anyhow::bail!("`[access]` entry has an empty `workload` name: {bad:?}");
5315 }
5316 }
5317 Ok(())
5318 }
5319}
5320
5321#[cfg(test)]
5322mod secret_config_tests {
5323 use super::*;
5324
5325 fn parse(body: &str) -> Result<SecretConfig> {
5326 let cfg: SecretConfig = toml::from_str(body)?;
5327 cfg.validate()?;
5328 Ok(cfg)
5329 }
5330
5331 #[test]
5332 fn minimal_declaration_parses_with_narrow_defaults() {
5333 let cfg = parse(
5334 r#"
5335schema_version = 1
5336name = "svc/token"
5337vault_slot = "svc-token"
5338[access]
5339workloads = [{ workload = "svc" }]
5340"#,
5341 )
5342 .unwrap();
5343
5344 assert_eq!(cfg.encoding, SecretEncoding::Utf8, "text is the default");
5345 assert!(cfg.target.is_none());
5346 // The omitted tenant/namespace must narrow to the singletons, not widen
5347 // to a wildcard.
5348 assert!(cfg
5349 .access
5350 .admits(&workload_spec::secrets::SecretConsumer::workload("svc")));
5351 assert!(!cfg
5352 .access
5353 .admits(&workload_spec::secrets::SecretConsumer::workload("other")));
5354 }
5355
5356 #[test]
5357 fn allow_any_is_spelled_as_a_bare_string() {
5358 // The operator-facing spelling, pinned: `access = "allow_any"`.
5359 let cfg = parse(
5360 r#"
5361schema_version = 1
5362name = "public/thing"
5363vault_slot = "slot"
5364access = "allow_any"
5365"#,
5366 )
5367 .unwrap();
5368 assert_eq!(cfg.access, SecretAccess::AllowAny);
5369 }
5370
5371 #[test]
5372 fn a_declaration_with_no_access_block_is_rejected() {
5373 // Omitting `[access]` defaults to deny-all, which is the correct
5374 // *runtime* default but never a correct authoring intent — so it must
5375 // not silently produce a secret nobody can mount.
5376 let err = parse(
5377 r#"
5378schema_version = 1
5379name = "svc/token"
5380vault_slot = "svc-token"
5381"#,
5382 )
5383 .unwrap_err()
5384 .to_string();
5385 assert!(err.contains("admits nobody"), "got {err}");
5386 }
5387
5388 #[test]
5389 fn empty_name_or_slot_is_rejected() {
5390 assert!(parse(
5391 r#"
5392schema_version = 1
5393name = ""
5394vault_slot = "slot"
5395access = "allow_any"
5396"#
5397 )
5398 .is_err());
5399 assert!(parse(
5400 r#"
5401schema_version = 1
5402name = "x"
5403vault_slot = " "
5404access = "allow_any"
5405"#
5406 )
5407 .is_err());
5408 }
5409
5410 #[test]
5411 fn target_declaration_maps_onto_the_workload_spec_type() {
5412 let cfg = parse(
5413 r#"
5414schema_version = 1
5415name = "svc/token"
5416vault_slot = "slot"
5417access = "allow_any"
5418[target]
5419kind = "file"
5420path = "/run/secrets/t"
5421"#,
5422 )
5423 .unwrap();
5424 match cfg.target.unwrap().to_target() {
5425 workload_spec::SecretTarget::File { path, mode } => {
5426 assert_eq!(path, std::path::PathBuf::from("/run/secrets/t"));
5427 assert_eq!(mode, 0o400, "owner-read-only by default");
5428 }
5429 other => panic!("expected File, got {other:?}"),
5430 }
5431 }
5432
5433 #[test]
5434 fn load_dir_is_empty_for_a_camp_with_no_secrets() {
5435 let tmp = tempfile::TempDir::new().unwrap();
5436 assert!(SecretConfig::load_dir(&tmp.path().join("nope"))
5437 .unwrap()
5438 .is_empty());
5439 }
5440}
5441
5442/// Split a `"<service>/<component-id>"` ref. Returns `None` if the ref
5443/// isn't shaped like `service/component`.
5444fn split_component_ref(s: &str) -> Option<(&str, &str)> {
5445 let (svc, comp) = s.split_once('/')?;
5446 if svc.is_empty() || comp.is_empty() || comp.contains('/') {
5447 return None;
5448 }
5449 Some((svc, comp))
5450}
5451
5452#[cfg(test)]
5453mod tests {
5454 use super::*;
5455 use std::path::PathBuf;
5456
5457 fn make_machine(name: &str, mesh_tags: Vec<&str>) -> MachineConfig {
5458 MachineConfig {
5459 name: name.into(),
5460 provider: "hetzner".into(),
5461 location: Some("hil".into()),
5462 server_type: Some("ccx13".into()),
5463 hosts_mirrors: vec![],
5464 mesh_tags: mesh_tags.into_iter().map(String::from).collect(),
5465 region: None,
5466 zone: None,
5467 arch: None,
5468 bucket: None,
5469 vendor: None,
5470 nickname: None,
5471 legacy_hostkey_fingerprint: None,
5472 registration: Default::default(),
5473 ssh_keys: vec![],
5474 cloudflared: None,
5475 hosts_operator_bridge: false,
5476 connect: None,
5477 allocatable: None,
5478 taints: vec![],
5479 sovereign_group: None,
5480 sovereign_role: None,
5481 ingress_floating_ip: None,
5482 }
5483 }
5484
5485 /// Like [`make_machine`] but with explicit topology axes for F16 tests.
5486 fn make_machine_topo(
5487 name: &str,
5488 provider: &str,
5489 region: &str,
5490 mesh_tags: Vec<&str>,
5491 ) -> MachineConfig {
5492 MachineConfig {
5493 provider: provider.into(),
5494 region: Some(region.into()),
5495 zone: Some(region.into()),
5496 ..make_machine(name, mesh_tags)
5497 }
5498 }
5499
5500 fn make_empty_cfg(machines: Vec<MachineConfig>) -> CloudConfig {
5501 CloudConfig {
5502 workspace_root: PathBuf::new(),
5503 machines,
5504 providers: vec![],
5505 machine_origins: BTreeMap::new(),
5506 provider_origins: BTreeMap::new(),
5507 services: BTreeMap::new(),
5508 domains: BTreeMap::new(),
5509 legacy_mirrors: vec![],
5510 workloads: vec![],
5511 topology: TopologyConfig::default(),
5512 legacy_services: vec![],
5513 }
5514 }
5515
5516 #[test]
5517 fn required_spec_parses_from_provider_fields() {
5518 let toml_src = r#"
5519use = "hetzner-primary"
5520[required]
5521mesh_tags = ["tag:cloud-runner"]
5522"#;
5523 let slot: MirrorProviderSlot = toml::from_str(toml_src).unwrap();
5524 let req = slot.required().expect("required block present");
5525 assert_eq!(req.mesh_tags, vec!["tag:cloud-runner"]);
5526 }
5527
5528 #[test]
5529 fn required_spec_absent_when_field_missing() {
5530 let slot: MirrorProviderSlot = toml::from_str(r#"use = "hetzner-primary""#).unwrap();
5531 assert!(slot.required().is_none());
5532 }
5533
5534 #[test]
5535 fn db_catalog_parses_all_env_blocks() {
5536 // W241 / R571-F8: a service.toml [db] table with dev/pond/cloud.
5537 let toml_src = r#"
5538schema_version = 1
5539name = "scrabcake"
5540domain = "scrabcake.net.yah.dev"
5541
5542[[db.dev]]
5543name = "main"
5544path = "data/dev.sqlite"
5545
5546[[db.pond]]
5547name = "main"
5548port = 5433
5549
5550[[db.pond]]
5551name = "pg"
5552port = 5432
5553kind = "postgres"
5554
5555[[db.cloud]]
5556name = "main"
5557url = "libsql://scrabcake.turso.io"
5558auth_token_env = "SCRABCAKE_TURSO_TOKEN"
5559"#;
5560 let svc: ServiceConfig = toml::from_str(toml_src).unwrap();
5561 assert_eq!(svc.db.dev.len(), 1);
5562 assert_eq!(svc.db.dev[0].path, "data/dev.sqlite");
5563 assert_eq!(svc.db.pond.len(), 2);
5564 assert_eq!(svc.db.pond[0].port, Some(5433));
5565 assert_eq!(svc.db.pond[0].kind, PondDbKind::Turso); // default
5566 assert_eq!(svc.db.pond[1].kind, PondDbKind::Postgres);
5567 assert_eq!(
5568 svc.db.cloud[0].auth_token_env.as_deref(),
5569 Some("SCRABCAKE_TURSO_TOKEN")
5570 );
5571 }
5572
5573 #[test]
5574 fn service_without_db_table_has_empty_catalog() {
5575 let svc: ServiceConfig =
5576 toml::from_str("schema_version = 1\nname = \"s\"\ndomain = \"s.dev\"\n").unwrap();
5577 assert!(svc.db.is_empty());
5578 // And an empty [db] must not appear when re-serialized.
5579 let out = toml::to_string(&svc).unwrap();
5580 assert!(
5581 !out.contains("[db"),
5582 "empty db table should be skipped: {out}"
5583 );
5584 }
5585
5586 #[test]
5587 fn camp_shared_cloud_toml_parses() {
5588 let src = r#"
5589[[cloud]]
5590name = "analytics"
5591url = "postgres://shared/analytics"
5592"#;
5593 let shared: CampCloudDbs = toml::from_str(src).unwrap();
5594 assert_eq!(shared.cloud.len(), 1);
5595 assert_eq!(shared.cloud[0].name, "analytics");
5596 }
5597
5598 #[test]
5599 fn resolve_machine_by_mesh_tags_superset_match() {
5600 let cfg = make_empty_cfg(vec![
5601 make_machine("yah-bnt-1", vec!["tag:primary-yah", "tag:tier-scratch"]),
5602 make_machine("us-west-001", vec!["tag:primary-yah", "tag:cloud-runner"]),
5603 ]);
5604 let picked = cfg
5605 .resolve_machine_by_mesh_tags(&["tag:cloud-runner".into()])
5606 .map(|m| m.name.as_str());
5607 assert_eq!(picked, Some("us-west-001"));
5608 }
5609
5610 #[test]
5611 fn resolve_machine_by_mesh_tags_returns_none_when_no_match() {
5612 let cfg = make_empty_cfg(vec![make_machine("yah-bnt-1", vec!["tag:primary-yah"])]);
5613 assert!(cfg
5614 .resolve_machine_by_mesh_tags(&["tag:cloud-runner".into()])
5615 .is_none());
5616 }
5617
5618 // ─── R590-F1 mesh-tag node-selector admission ───────────────────────────
5619
5620 /// Build a forge WorkloadSpec carrying the R594 node-selector annotation.
5621 /// `selector` is the comma-joined mesh-tag set; `None` omits the annotation
5622 /// entirely (pre-R594 "no constraint").
5623 fn ws_with_selector(selector: Option<&str>) -> WorkloadSpec {
5624 use workload_spec::{ImageRef, TierTag};
5625 let mut ws = WorkloadSpec::for_forge(
5626 "R590-F1-test",
5627 ImageRef {
5628 registry: "docker.io".into(),
5629 repository: "library/busybox".into(),
5630 tag: "latest".into(),
5631 digest: workload_spec::testing::test_digest(),
5632 },
5633 TierTag("infra".into()),
5634 vec![],
5635 );
5636 if let Some(sel) = selector {
5637 ws.annotations.insert(
5638 velveteen_exec::remote::NODE_SELECTOR_MESH_TAGS_ANNOTATION.into(),
5639 sel.into(),
5640 );
5641 }
5642 ws
5643 }
5644
5645 /// The build-worker fleet shape: one x86 node (us-west-002) and one arm
5646 /// node (a Pi5), both carrying `tag:build-worker`.
5647 fn build_worker_fleet() -> CloudConfig {
5648 make_empty_cfg(vec![
5649 make_machine("us-west-002", vec!["tag:build-worker", "arch:x86"]),
5650 make_machine("pi5-001", vec!["tag:build-worker", "arch:arm"]),
5651 ])
5652 }
5653
5654 #[test]
5655 fn admit_workload_routes_amd64_to_x86_worker() {
5656 let cfg = build_worker_fleet();
5657 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
5658 let picked = cfg.admit_workload(&ws).unwrap();
5659 assert_eq!(picked.name, "us-west-002");
5660 }
5661
5662 #[test]
5663 fn admit_workload_routes_arm64_to_pi5_worker() {
5664 let cfg = build_worker_fleet();
5665 let ws = ws_with_selector(Some("tag:build-worker,arch:arm"));
5666 let picked = cfg.admit_workload(&ws).unwrap();
5667 assert_eq!(picked.name, "pi5-001");
5668 }
5669
5670 /// A forge run must be admissible on a build-worker smaller than its own
5671 /// cgroup ceiling.
5672 ///
5673 /// The fleet's arm build-workers are 8 GiB Pi-5s and `for_forge` sets a
5674 /// 32 GiB ceiling, so while admission read `resources.memory_mb` as the
5675 /// capacity floor this returned "no candidates" and *every* offloaded qed
5676 /// step to those nodes failed at dispatch — measured on desktop-release run
5677 /// b04cef47, where the aarch64-linux row died in 1.6s. The other
5678 /// build-workers (16 GiB us-west-003, and the arm Pi-5s) were excluded the
5679 /// same way, leaving one 47 GiB node as the fleet's only legal target for
5680 /// remote CI.
5681 #[test]
5682 fn admit_workload_places_a_forge_run_on_a_worker_smaller_than_its_ceiling() {
5683 let mut pi = make_machine("pi5-001", vec!["tag:build-worker", "arch:arm"]);
5684 pi.allocatable = Some(NodeAllocatable {
5685 memory_mb: 8192,
5686 cpu_millis: 4000,
5687 });
5688 let cfg = make_empty_cfg(vec![pi]);
5689
5690 let ws = ws_with_selector(Some("tag:build-worker,arch:arm"));
5691 assert!(
5692 ws.resources.memory_mb > 8192,
5693 "precondition: the ceiling must exceed the node, or this proves nothing"
5694 );
5695
5696 let picked = cfg
5697 .admit_workload(&ws)
5698 .expect("an 8 GiB build-worker must admit a forge run");
5699 assert_eq!(picked.name, "pi5-001");
5700 }
5701
5702 /// The floor is still enforced — the fix separates two numbers, it does not
5703 /// disable the R572-F5 capacity check.
5704 #[test]
5705 fn admit_workload_still_rejects_a_node_below_the_declared_request() {
5706 let mut tiny = make_machine("tiny-001", vec!["tag:build-worker", "arch:arm"]);
5707 tiny.allocatable = Some(NodeAllocatable {
5708 memory_mb: 512,
5709 cpu_millis: 4000,
5710 });
5711 let cfg = make_empty_cfg(vec![tiny]);
5712
5713 let ws = ws_with_selector(Some("tag:build-worker,arch:arm"));
5714 assert!(
5715 cfg.admit_workload(&ws).is_err(),
5716 "a 512 MiB node cannot satisfy a 2 GiB forge request"
5717 );
5718 }
5719
5720 // ─── R833-F8 imperative node-selector admission ─────────────────────────
5721
5722 /// Build a forge WorkloadSpec carrying the R833-F8 imperative node
5723 /// selector — the operator's `--where=node:<machine>`.
5724 fn ws_pinned_to(node: &str) -> WorkloadSpec {
5725 let mut ws = ws_with_selector(None);
5726 ws.annotations.insert(
5727 velveteen_exec::remote::NODE_SELECTOR_NODE_ANNOTATION.into(),
5728 node.into(),
5729 );
5730 ws
5731 }
5732
5733 /// The ticket's acceptance shape: a named node wins over the
5734 /// declaration-order tie-break that would otherwise decide placement.
5735 /// `us-west-002` is declared first and carries every tag, so an inferred
5736 /// placement lands there; the pin must reach `pi5-001` regardless.
5737 #[test]
5738 fn admit_workload_honours_an_explicitly_named_node() {
5739 let cfg = build_worker_fleet();
5740 assert_eq!(
5741 cfg.admit_workload(&ws_with_selector(Some("tag:build-worker")))
5742 .unwrap()
5743 .name,
5744 "us-west-002",
5745 "precondition: inference elects the first-declared node",
5746 );
5747 assert_eq!(
5748 cfg.admit_workload(&ws_pinned_to("pi5-001")).unwrap().name,
5749 "pi5-001",
5750 );
5751 }
5752
5753 /// A pin at a machine that is not declared fails loud, naming the
5754 /// constraint and the pool — the operator mistyped a node, and silently
5755 /// running the build somewhere else is the one outcome that must not
5756 /// happen.
5757 #[test]
5758 fn admit_workload_refuses_a_node_that_is_not_declared() {
5759 let cfg = build_worker_fleet();
5760 let err = cfg
5761 .admit_workload(&ws_pinned_to("us-west-404"))
5762 .unwrap_err()
5763 .to_string();
5764 assert!(err.contains("required.nodes=[us-west-404]"), "{err}");
5765 assert!(err.contains("us-west-002"), "the pool must be named: {err}");
5766 }
5767
5768 /// The pin narrows the candidate set; it does not suspend the other axes.
5769 /// A named node that cannot fit the workload still refuses, rather than
5770 /// being handed work it has no room for.
5771 #[test]
5772 fn a_pinned_node_is_still_checked_against_capacity() {
5773 let mut tiny = make_machine("tiny-001", vec!["tag:build-worker", "arch:arm"]);
5774 tiny.allocatable = Some(NodeAllocatable {
5775 memory_mb: 512,
5776 cpu_millis: 4000,
5777 });
5778 let cfg = make_empty_cfg(vec![tiny]);
5779 assert!(cfg.admit_workload(&ws_pinned_to("tiny-001")).is_err());
5780 }
5781
5782 /// Inference is untouched: with no node annotation the `nodes` axis is
5783 /// empty, which is "no constraint" — every pre-R833-F8 workload is admitted
5784 /// exactly as before.
5785 #[test]
5786 fn an_unpinned_workload_carries_no_node_constraint() {
5787 assert!(node_selector_node(&ws_with_selector(Some("arch:x86"))).is_none());
5788 assert_eq!(
5789 node_selector_node(&ws_pinned_to("us-west-003")).as_deref(),
5790 Some("us-west-003")
5791 );
5792 assert!(RequiredSpec::default().is_unconstrained());
5793 assert!(!RequiredSpec {
5794 nodes: vec!["us-west-003".into()],
5795 ..Default::default()
5796 }
5797 .is_unconstrained());
5798 }
5799
5800 #[test]
5801 fn admit_workload_rejects_node_missing_required_tag() {
5802 // Only an arm worker exists; an x86 build must NOT land on it.
5803 let cfg = make_empty_cfg(vec![make_machine(
5804 "pi5-001",
5805 vec!["tag:build-worker", "arch:arm"],
5806 )]);
5807 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
5808 assert!(cfg.admit_workload(&ws).is_err());
5809 }
5810
5811 /// R555-S1 regression: with TWO nodes carrying the same tag set, which one
5812 /// admits must be decided by *declaration order* (file name), which is the
5813 /// contract `admit_workload` documents — not by `read_dir` order, which is
5814 /// filesystem-dependent and can change when an unrelated file appears in
5815 /// the directory. Written creation-order-reversed so a filesystem that
5816 /// yields creation order (rather than sorted order) trips it without the
5817 /// sort in `load_dir`.
5818 ///
5819 /// Live consequence this guards: `.yah/infra/machines/` carries both
5820 /// us-west-002 and us-west-003 on `[tag:build-worker, arch:x86, os:linux]`,
5821 /// so an x86 QED offload has two equal candidates. Unstable selection means
5822 /// a retried build cannot be relied on to land back on the node whose
5823 /// working state it left behind.
5824 #[test]
5825 fn equally_matching_machines_admit_in_file_name_order() {
5826 let tmp = tempfile::TempDir::new().unwrap();
5827 let machines = tmp.path().join(".yah").join("infra").join("machines");
5828 std::fs::create_dir_all(&machines).unwrap();
5829 let toml_for = |name: &str| {
5830 format!(
5831 r#"name = "{name}"
5832provider = "static"
5833mesh_tags = ["tag:build-worker", "arch:x86"]
5834"#
5835 )
5836 };
5837 // Reverse-of-sorted creation order on purpose.
5838 std::fs::write(machines.join("b-second.toml"), toml_for("b-second")).unwrap();
5839 std::fs::write(machines.join("a-first.toml"), toml_for("a-first")).unwrap();
5840
5841 let cfg = CloudConfig::load(tmp.path()).unwrap();
5842 assert_eq!(
5843 cfg.machines.iter().map(|m| m.name.as_str()).collect::<Vec<_>>(),
5844 vec!["a-first", "b-second"],
5845 "machines must load in file-name order, not read_dir order"
5846 );
5847
5848 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
5849 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "a-first");
5850
5851 // R605-T14: the same two nodes, seen as the pool they are. The head is
5852 // what `admit_workload` returns, and the tail is what the dispatcher
5853 // fails over to when the head does not answer — so these two views must
5854 // come from one predicate, not two.
5855 assert_eq!(
5856 cfg.admit_workload_candidates(&ws)
5857 .unwrap()
5858 .iter()
5859 .map(|m| m.name.as_str())
5860 .collect::<Vec<_>>(),
5861 vec!["a-first", "b-second"],
5862 "the pool must be every admissible node, in the same declaration order"
5863 );
5864 }
5865
5866 /// A pool of one is still a pool, and a pool of none is an `Err` that reads
5867 /// exactly like `admit_workload`'s — "nothing admits this" is one failure
5868 /// with one wording, not two.
5869 #[test]
5870 fn admit_workload_candidates_matches_admit_workload_on_the_edges() {
5871 let cfg = make_empty_cfg(vec![
5872 make_machine("x86-box", vec!["tag:build-worker", "arch:x86"]),
5873 make_machine("arm-box", vec!["tag:build-worker", "arch:arm"]),
5874 ]);
5875
5876 let one = ws_with_selector(Some("arch:arm"));
5877 assert_eq!(
5878 cfg.admit_workload_candidates(&one)
5879 .unwrap()
5880 .iter()
5881 .map(|m| m.name.as_str())
5882 .collect::<Vec<_>>(),
5883 vec!["arm-box"],
5884 "only one node carries arch:arm, so the pool is that one node"
5885 );
5886
5887 let none = ws_with_selector(Some("arch:riscv"));
5888 let pool_err = cfg.admit_workload_candidates(&none).unwrap_err().to_string();
5889 let single_err = cfg.admit_workload(&none).unwrap_err().to_string();
5890 assert_eq!(
5891 pool_err, single_err,
5892 "an empty pool must be refused in the same words as an unadmitted workload"
5893 );
5894 }
5895
5896 /// R844-B7 — the wrong-root half of the distinction. A directory with no
5897 /// `.yah/` at all used to load as a valid config with zero machines, so a
5898 /// caller pointed at the wrong directory got a green result that measured
5899 /// nothing. Asserting `load` merely *succeeds* is what let that through;
5900 /// the shape that catches it is a non-zero machine count, or — here — an
5901 /// `Err` naming the path that was looked for.
5902 #[test]
5903 fn loading_a_directory_that_is_not_a_yah_workspace_is_an_error() {
5904 let tmp = tempfile::TempDir::new().unwrap();
5905 // A plausible-looking package root: real files, real subdirectories,
5906 // no `.yah/`. This is exactly what `load_cloud(".")` reads when a test
5907 // runs under `cargo test` from a member crate.
5908 std::fs::create_dir_all(tmp.path().join("src")).unwrap();
5909 std::fs::write(tmp.path().join("Cargo.toml"), "[package]\nname = \"x\"\n").unwrap();
5910
5911 let err = CloudConfig::load(tmp.path()).expect_err(
5912 "a directory with no .yah/ is the WRONG DIRECTORY, not a fleet with no machines",
5913 );
5914 let msg = format!("{err:#}");
5915 assert!(
5916 msg.contains("not a yah workspace"),
5917 "error must say the root is not a workspace, got: {msg}"
5918 );
5919 assert!(
5920 msg.contains(&tmp.path().join(".yah").display().to_string()),
5921 "error must name the path it looked for so an operator sees the \
5922 wrong-root immediately, got: {msg}"
5923 );
5924 }
5925
5926 /// R844-B7 — the other half, and the reason the check is drawn at `.yah/`
5927 /// rather than at the machine list: a camp that declares no machines is a
5928 /// real workspace and must keep loading. Blanket-erroring on an empty
5929 /// fleet would conflate `unknown` with `answered with none`, which is the
5930 /// exact confusion the check exists to remove.
5931 #[test]
5932 fn a_workspace_with_no_machines_declared_still_loads() {
5933 let tmp = tempfile::TempDir::new().unwrap();
5934 std::fs::create_dir_all(tmp.path().join(".yah")).unwrap();
5935
5936 let cfg = CloudConfig::load(tmp.path())
5937 .expect("a `.yah/` with no infra/machines/ is an empty fleet, not a wrong root");
5938 assert!(cfg.machines.is_empty(), "nothing was declared");
5939 assert!(cfg.services.is_empty());
5940 assert!(cfg.providers.is_empty());
5941
5942 // And an existing-but-empty machines dir is the same answer, not a
5943 // second special case.
5944 std::fs::create_dir_all(crate::paths::machines_dir(tmp.path())).unwrap();
5945 let cfg = CloudConfig::load(tmp.path()).expect("an empty machines/ dir still loads");
5946 assert!(cfg.machines.is_empty());
5947 }
5948
5949 #[test]
5950 fn admit_workload_empty_selector_is_unconstrained() {
5951 // Absent annotation ⇒ no mesh-tag constraint ⇒ first declared machine
5952 // (pre-R594 behavior preserved).
5953 let cfg = build_worker_fleet();
5954 let ws = ws_with_selector(None);
5955 let picked = cfg.admit_workload(&ws).unwrap();
5956 assert_eq!(picked.name, "us-west-002");
5957 }
5958
5959 #[test]
5960 fn node_selector_mesh_tags_trims_and_drops_empties() {
5961 let ws = ws_with_selector(Some(" tag:build-worker , arch:x86 ,"));
5962 assert_eq!(
5963 node_selector_mesh_tags(&ws),
5964 vec!["tag:build-worker".to_string(), "arch:x86".to_string()]
5965 );
5966 assert!(node_selector_mesh_tags(&ws_with_selector(None)).is_empty());
5967 }
5968
5969 // ─── F16 topology-aware resolver ────────────────────────────────────────
5970
5971 fn two_region_fleet() -> CloudConfig {
5972 make_empty_cfg(vec![
5973 make_machine_topo(
5974 "us-west-001",
5975 "hetzner",
5976 "us-west",
5977 vec!["tag:cloud-runner"],
5978 ),
5979 make_machine_topo(
5980 "eu-west-001",
5981 "hetzner",
5982 "eu-west",
5983 vec!["tag:cloud-runner"],
5984 ),
5985 ])
5986 }
5987
5988 #[test]
5989 fn resolve_machine_matches_on_region_plus_mesh_tags() {
5990 let cfg = two_region_fleet();
5991 let req = RequiredSpec {
5992 regions: vec!["us-west".into()],
5993 mesh_tags: vec!["tag:cloud-runner".into()],
5994 ..Default::default()
5995 };
5996 let picked = cfg.resolve_machine(&req).unwrap();
5997 assert_eq!(picked.name, "us-west-001");
5998 }
5999
6000 #[test]
6001 fn resolve_machine_region_disambiguates_same_tag() {
6002 // Both boxes carry tag:cloud-runner; the region axis selects eu-west.
6003 let cfg = two_region_fleet();
6004 let req = RequiredSpec {
6005 regions: vec!["eu-west".into()],
6006 mesh_tags: vec!["tag:cloud-runner".into()],
6007 ..Default::default()
6008 };
6009 assert_eq!(cfg.resolve_machine(&req).unwrap().name, "eu-west-001");
6010 }
6011
6012 #[test]
6013 fn resolve_machine_fails_loud_with_constraint_summary() {
6014 let cfg = two_region_fleet();
6015 let req = RequiredSpec {
6016 regions: vec!["us-central".into()],
6017 mesh_tags: vec!["tag:cloud-runner".into()],
6018 ..Default::default()
6019 };
6020 let err = cfg.resolve_machine(&req).unwrap_err().to_string();
6021 assert!(err.contains("required.regions=[us-central]"), "got: {err}");
6022 assert!(
6023 err.contains("required.mesh_tags=[tag:cloud-runner]"),
6024 "got: {err}"
6025 );
6026 // Names the candidates it rejected.
6027 assert!(err.contains("us-west-001"), "got: {err}");
6028 }
6029
6030 #[test]
6031 fn resolve_machine_provider_axis_filters() {
6032 let cfg = make_empty_cfg(vec![
6033 make_machine_topo("aws-west-1", "aws", "us-west", vec!["tag:cloud-runner"]),
6034 make_machine_topo("hz-west-1", "hetzner", "us-west", vec!["tag:cloud-runner"]),
6035 ]);
6036 let req = RequiredSpec {
6037 regions: vec!["us-west".into()],
6038 providers: vec!["hetzner".into()],
6039 ..Default::default()
6040 };
6041 assert_eq!(cfg.resolve_machine(&req).unwrap().name, "hz-west-1");
6042 }
6043
6044 #[test]
6045 fn unconstrained_required_spec_matches_first_machine() {
6046 let cfg = two_region_fleet();
6047 assert!(RequiredSpec::default().is_unconstrained());
6048 assert_eq!(
6049 cfg.resolve_machine(&RequiredSpec::default()).unwrap().name,
6050 "us-west-001"
6051 );
6052 }
6053
6054 #[test]
6055 fn required_spec_parses_topology_axes_from_toml() {
6056 let toml_src = r#"
6057use = "hetzner-primary"
6058[required]
6059regions = ["us-west"]
6060mesh_tags = ["tag:cloud-runner"]
6061"#;
6062 let slot: MirrorProviderSlot = toml::from_str(toml_src).unwrap();
6063 let req = slot.required().expect("required block present");
6064 assert_eq!(req.regions, vec!["us-west"]);
6065 assert_eq!(req.mesh_tags, vec!["tag:cloud-runner"]);
6066 assert!(req.zones.is_empty());
6067 }
6068
6069 // ─── R844-F8 replica count ──────────────────────────────────────────────
6070
6071 fn three_runner_fleet() -> CloudConfig {
6072 make_empty_cfg(vec![
6073 make_machine_topo("us-east-001", "hetzner", "us-east", vec!["tag:cloud-runner"]),
6074 make_machine_topo(
6075 "us-south-001",
6076 "hetzner",
6077 "us-south",
6078 vec!["tag:cloud-runner"],
6079 ),
6080 make_machine_topo(
6081 "us-west-001",
6082 "hetzner",
6083 "us-west",
6084 vec!["tag:cloud-runner"],
6085 ),
6086 ])
6087 }
6088
6089 #[test]
6090 fn an_absent_replica_count_still_places_exactly_one_machine() {
6091 // The migration is additive: every mirror on disk omits `replicas`, and
6092 // must resolve byte-identically to the pre-R844-F8 answer.
6093 let cfg = three_runner_fleet();
6094 let req = RequiredSpec {
6095 mesh_tags: vec!["tag:cloud-runner".into()],
6096 ..Default::default()
6097 };
6098 assert_eq!(req.replica_count(), 1);
6099 let names: Vec<&str> = cfg
6100 .resolve_machines(&req)
6101 .unwrap()
6102 .iter()
6103 .map(|m| m.name.as_str())
6104 .collect();
6105 assert_eq!(names, vec!["us-east-001"]);
6106 assert_eq!(cfg.resolve_machine(&req).unwrap().name, "us-east-001");
6107 }
6108
6109 #[test]
6110 fn a_replica_count_places_that_many_machines_not_every_match() {
6111 // Three machines match; two are asked for; two are placed. Inferring the
6112 // count from the match count would make adding a box to the fleet
6113 // silently scale a production front door.
6114 let cfg = three_runner_fleet();
6115 let req = RequiredSpec {
6116 mesh_tags: vec!["tag:cloud-runner".into()],
6117 replicas: Some(2),
6118 ..Default::default()
6119 };
6120 let names: Vec<&str> = cfg
6121 .resolve_machines(&req)
6122 .unwrap()
6123 .iter()
6124 .map(|m| m.name.as_str())
6125 .collect();
6126 assert_eq!(names, vec!["us-east-001", "us-south-001"]);
6127 }
6128
6129 #[test]
6130 fn fewer_matches_than_replicas_is_an_error_naming_both_numbers() {
6131 // Never a partial placement: one of two reported as success is the
6132 // subset-that-looks-like-it-worked failure in its purest form.
6133 let cfg = three_runner_fleet();
6134 let req = RequiredSpec {
6135 regions: vec!["us-east".into()],
6136 replicas: Some(2),
6137 ..Default::default()
6138 };
6139 let err = cfg.resolve_machines(&req).unwrap_err().to_string();
6140 assert!(err.contains("only 1 of 2"), "got: {err}");
6141 assert!(err.contains("required.regions=[us-east]"), "got: {err}");
6142 // …and names the pool it searched, like every other placement refusal.
6143 assert!(err.contains("declared machines"), "got: {err}");
6144 assert!(err.contains("us-south-001"), "got: {err}");
6145 }
6146
6147 #[test]
6148 fn zero_replicas_is_refused_rather_than_placing_nothing() {
6149 let cfg = three_runner_fleet();
6150 let req = RequiredSpec {
6151 mesh_tags: vec!["tag:cloud-runner".into()],
6152 replicas: Some(0),
6153 ..Default::default()
6154 };
6155 let err = cfg.resolve_machines(&req).unwrap_err().to_string();
6156 assert!(err.contains("replicas = 0"), "got: {err}");
6157 }
6158
6159 #[test]
6160 fn replicas_parses_from_the_inline_required_form() {
6161 // The INLINE form specifically: `[providers.bundle.required]` as a table
6162 // HEADER ends the slot's table and reparents every key below it.
6163 let slot: MirrorProviderSlot = toml::from_str(
6164 r#"
6165use = "hetzner-primary"
6166port = 8080
6167required = { regions = ["us-east"], mesh_tags = ["tag:cloud-runner"], replicas = 2 }
6168"#,
6169 )
6170 .unwrap();
6171 assert_eq!(
6172 slot.fields().get("port").and_then(|v| v.as_integer()),
6173 Some(8080),
6174 "the inline form leaves the slot's other keys where they were"
6175 );
6176 let req = slot.required().expect("required block present");
6177 assert_eq!(req.replicas, Some(2));
6178 assert_eq!(req.replica_count(), 2);
6179 // A count is not a match axis — it says how many, not which.
6180 assert!(!req.is_unconstrained());
6181 assert!(RequiredSpec {
6182 replicas: Some(2),
6183 ..Default::default()
6184 }
6185 .is_unconstrained());
6186 }
6187
6188 #[test]
6189 fn round_trip_machine() {
6190 let cfg = MachineConfig {
6191 name: "test-pdx-1".into(),
6192 provider: "hetzner".into(),
6193 location: Some("pdx".into()),
6194 server_type: Some("cpx22".into()),
6195 hosts_mirrors: vec!["noisetable".into()],
6196 mesh_tags: vec!["region:pdx".into()],
6197 region: Some("us-west".into()),
6198 zone: Some("pdx".into()),
6199 arch: None,
6200 bucket: Some(BucketSpec {
6201 name: "test-assets-pdx-1".into(),
6202 public_read: false,
6203 }),
6204 vendor: None,
6205 nickname: None,
6206 legacy_hostkey_fingerprint: None,
6207 registration: Default::default(),
6208 ssh_keys: vec![],
6209 cloudflared: None,
6210 hosts_operator_bridge: false,
6211 connect: None,
6212 allocatable: None,
6213 taints: vec![],
6214 sovereign_group: None,
6215 sovereign_role: None,
6216 ingress_floating_ip: None,
6217 };
6218 let s = toml::to_string(&cfg).unwrap();
6219 let back: MachineConfig = toml::from_str(&s).unwrap();
6220 assert_eq!(back.name, cfg.name);
6221 assert_eq!(back.location, cfg.location);
6222 assert_eq!(back.region.as_deref(), Some("us-west"));
6223 assert_eq!(back.zone.as_deref(), Some("pdx"));
6224 }
6225
6226 #[test]
6227 fn round_trip_mirror() {
6228 let cfg = LegacyMirrorConfig {
6229 camp: "noisetable".into(),
6230 regions: vec!["pdx".into(), "iad".into()],
6231 workloads: vec!["asset-registry".into()],
6232 cloud_domain: None,
6233 };
6234 let s = toml::to_string(&cfg).unwrap();
6235 let back: LegacyMirrorConfig = toml::from_str(&s).unwrap();
6236 assert_eq!(back.camp, cfg.camp);
6237 assert_eq!(back.regions, cfg.regions);
6238 assert_eq!(back.workloads, cfg.workloads);
6239 }
6240
6241 #[test]
6242 fn mirror_serialises_as_camp_key() {
6243 // Serialised form should use `camp`, not `rig`.
6244 let cfg = LegacyMirrorConfig {
6245 camp: "noisetable".into(),
6246 regions: vec!["pdx".into()],
6247 workloads: vec![],
6248 cloud_domain: None,
6249 };
6250 let s = toml::to_string(&cfg).unwrap();
6251 assert!(
6252 s.contains("camp = "),
6253 "serialised key should be 'camp': {s}"
6254 );
6255 assert!(!s.contains("rig = "), "old key should not appear: {s}");
6256 }
6257
6258 #[test]
6259 fn mirror_rig_alias_still_loads() {
6260 // Old mirrors/*.toml files use `rig = "..."` before the R137 rename;
6261 // the alias keeps them loading until the one-time `sed` migration runs.
6262 let toml_str =
6263 "rig = \"noisetable\"\nregions = [\"pdx\"]\nworkloads = [\"asset-registry\"]\n";
6264 let cfg: LegacyMirrorConfig = toml::from_str(toml_str).unwrap();
6265 assert_eq!(cfg.camp, "noisetable");
6266 }
6267
6268 #[test]
6269 fn mirror_services_alias_still_loads() {
6270 // Old mirrors/*.toml files use `services = [...]`; the alias keeps them
6271 // loading without a migration step.
6272 let toml_str =
6273 "camp = \"noisetable\"\nregions = [\"pdx\"]\nservices = [\"asset-registry\"]\n";
6274 let cfg: LegacyMirrorConfig = toml::from_str(toml_str).unwrap();
6275 assert_eq!(cfg.workloads, vec!["asset-registry"]);
6276 }
6277
6278 #[test]
6279 fn round_trip_service_legacy() {
6280 let cfg = LegacyServiceConfig {
6281 name: "asset-registry".into(),
6282 image: "ghcr.io/noisetable/asset-registry".into(),
6283 version: "v1.0.0".into(),
6284 env: HashMap::new(),
6285 ports: vec![PortMapping {
6286 host: 8080,
6287 container: 8080,
6288 }],
6289 mesh_only: false,
6290 bind_interface: None,
6291 tenant: TenantId::singleton(),
6292 };
6293 let s = toml::to_string(&cfg).unwrap();
6294 let back: LegacyServiceConfig = toml::from_str(&s).unwrap();
6295 assert_eq!(back.name, cfg.name);
6296 assert_eq!(back.image, cfg.image);
6297 }
6298
6299 #[test]
6300 fn service_bind_interface_round_trips() {
6301 let cfg = LegacyServiceConfig {
6302 name: "postgres".into(),
6303 image: "postgres".into(),
6304 version: "16".into(),
6305 env: HashMap::new(),
6306 ports: vec![PortMapping {
6307 host: 5432,
6308 container: 5432,
6309 }],
6310 mesh_only: true,
6311 bind_interface: Some("tailscale0".into()),
6312 tenant: TenantId::singleton(),
6313 };
6314 let s = toml::to_string(&cfg).unwrap();
6315 let back: LegacyServiceConfig = toml::from_str(&s).unwrap();
6316 assert_eq!(back.bind_interface.as_deref(), Some("tailscale0"));
6317 }
6318
6319 #[test]
6320 fn service_bind_interface_absent_is_none() {
6321 let toml_str = "name = \"app\"\nimage = \"app\"\nversion = \"v1\"\n";
6322 let cfg: LegacyServiceConfig = toml::from_str(toml_str).unwrap();
6323 assert!(
6324 cfg.bind_interface.is_none(),
6325 "bind_interface should default to None"
6326 );
6327 }
6328
6329 #[test]
6330 fn service_bind_interface_skipped_when_none() {
6331 let cfg = LegacyServiceConfig {
6332 name: "app".into(),
6333 image: "app".into(),
6334 version: "v1".into(),
6335 env: HashMap::new(),
6336 ports: vec![],
6337 mesh_only: false,
6338 bind_interface: None,
6339 tenant: TenantId::singleton(),
6340 };
6341 let s = toml::to_string(&cfg).unwrap();
6342 assert!(!s.contains("bind_interface"), "None should be skipped: {s}");
6343 }
6344
6345 #[test]
6346 fn load_dir_missing_is_empty() {
6347 let dir = std::path::PathBuf::from("/nonexistent/path");
6348 let result: Vec<MachineConfig> = load_dir(dir).unwrap();
6349 assert!(result.is_empty());
6350 }
6351
6352 #[test]
6353 fn topology_round_trip() {
6354 let topo = TopologyConfig {
6355 assignments: vec![
6356 MirrorAssignment {
6357 mirror: "noisetable-pdx".into(),
6358 machine: "noisetable-pdx-1".into(),
6359 },
6360 MirrorAssignment {
6361 mirror: "noisetable-iad".into(),
6362 machine: "noisetable-iad-1".into(),
6363 },
6364 ],
6365 buckets: vec![],
6366 };
6367 let s = toml::to_string(&topo).unwrap();
6368 let back: TopologyConfig = toml::from_str(&s).unwrap();
6369 assert_eq!(back.assignments.len(), 2);
6370 assert_eq!(back.assignments[0].mirror, "noisetable-pdx");
6371 assert_eq!(back.assignments[1].machine, "noisetable-iad-1");
6372 }
6373
6374 #[test]
6375 fn topology_absent_returns_default() {
6376 let tmp = tempfile::TempDir::new().unwrap();
6377 let path = tmp.path().join("topology.toml");
6378 // file doesn't exist
6379 let topo = load_topology(path).unwrap();
6380 assert!(topo.assignments.is_empty());
6381 }
6382
6383 /// Helper: lay out a `<workspace_root>/.yah/cloud/` legacy tree for the
6384 /// pre-R215 cargo tests below; returns the legacy cloud_dir for writes.
6385 fn make_legacy_cloud_dir(root: &std::path::Path) -> std::path::PathBuf {
6386 let cloud_dir = root.join(".yah").join("cloud");
6387 std::fs::create_dir_all(&cloud_dir).unwrap();
6388 cloud_dir
6389 }
6390
6391 #[test]
6392 fn cloud_config_load_and_lookup() {
6393 let tmp = tempfile::TempDir::new().unwrap();
6394 let root = tmp.path();
6395 let cloud_dir = make_legacy_cloud_dir(root);
6396
6397 let machine = MachineConfig {
6398 name: "noisetable-pdx-1".into(),
6399 provider: "hetzner".into(),
6400 location: Some("pdx".into()),
6401 server_type: Some("cpx22".into()),
6402 hosts_mirrors: vec!["noisetable".into(), "yah".into()],
6403 mesh_tags: vec!["region:pdx".into(), "tier:t2".into()],
6404 region: None,
6405 zone: None,
6406 arch: None,
6407 bucket: Some(BucketSpec {
6408 name: "noisetable-assets-pdx-1".into(),
6409 public_read: false,
6410 }),
6411 vendor: None,
6412 nickname: None,
6413 legacy_hostkey_fingerprint: None,
6414 registration: Default::default(),
6415 ssh_keys: vec![],
6416 cloudflared: None,
6417 hosts_operator_bridge: false,
6418 connect: None,
6419 allocatable: None,
6420 taints: vec![],
6421 sovereign_group: None,
6422 sovereign_role: None,
6423 ingress_floating_ip: None,
6424 };
6425 // Land in the legacy tree so the legacy machine loader picks it up.
6426 machine.save(&cloud_dir).unwrap();
6427
6428 let mirror_toml = "camp = \"noisetable\"\nregions = [\"pdx\", \"iad\", \"fsn\"]\nworkloads = [\"asset-registry\"]\n";
6429 std::fs::create_dir_all(cloud_dir.join("mirrors")).unwrap();
6430 std::fs::write(cloud_dir.join("mirrors/noisetable.toml"), mirror_toml).unwrap();
6431
6432 // Legacy services/ dir (backward compat)
6433 let svc_toml = "name = \"asset-registry\"\nimage = \"ghcr.io/noisetable/asset-registry\"\nversion = \"v1.0.0\"\nmesh_only = false\n";
6434 std::fs::create_dir_all(cloud_dir.join("services")).unwrap();
6435 std::fs::write(cloud_dir.join("services/asset-registry.toml"), svc_toml).unwrap();
6436
6437 let cfg = CloudConfig::load(root).unwrap();
6438
6439 assert_eq!(cfg.machines.len(), 1);
6440 assert_eq!(cfg.legacy_mirrors.len(), 1);
6441 assert_eq!(cfg.legacy_services.len(), 1);
6442 assert_eq!(cfg.workloads.len(), 0); // no workloads/ dir yet
6443 assert!(cfg.services.is_empty(), "no R215+ services/ tree");
6444 assert!(cfg.providers.is_empty(), "no R215+ providers/ tree");
6445
6446 let m = cfg.machine("noisetable-pdx-1").unwrap();
6447 assert_eq!(m.location(), "pdx");
6448 assert_eq!(m.bucket.as_ref().unwrap().name, "noisetable-assets-pdx-1");
6449
6450 let mir = cfg.legacy_mirror("noisetable").unwrap();
6451 assert_eq!(mir.regions, vec!["pdx", "iad", "fsn"]);
6452 assert_eq!(mir.workloads, vec!["asset-registry"]);
6453 }
6454
6455 #[test]
6456 fn mirror_folder_layout_loads() {
6457 // Folder layout: mirrors/<id>/mirror.toml — new preferred form.
6458 let tmp = tempfile::TempDir::new().unwrap();
6459 let root = tmp.path();
6460 let cloud_dir = make_legacy_cloud_dir(root);
6461 let mirror_dir = cloud_dir.join("mirrors").join("yah-com");
6462 std::fs::create_dir_all(&mirror_dir).unwrap();
6463 std::fs::write(
6464 mirror_dir.join("mirror.toml"),
6465 "camp = \"yah\"\nregions = [\"pdx\"]\nworkloads = [\"yah-web\"]\n",
6466 )
6467 .unwrap();
6468
6469 let cfg = CloudConfig::load(root).unwrap();
6470 assert_eq!(cfg.legacy_mirrors.len(), 1);
6471 let mir = cfg.legacy_mirror("yah").unwrap();
6472 assert_eq!(mir.camp, "yah");
6473 assert_eq!(mir.workloads, vec!["yah-web"]);
6474 }
6475
6476 #[test]
6477 fn mirror_folder_and_flat_coexist() {
6478 // Both layouts may coexist in the same mirrors/ directory.
6479 let tmp = tempfile::TempDir::new().unwrap();
6480 let root = tmp.path();
6481 let cloud_dir = make_legacy_cloud_dir(root);
6482 let mirrors_root = cloud_dir.join("mirrors");
6483 std::fs::create_dir_all(&mirrors_root).unwrap();
6484
6485 // Flat legacy mirror
6486 std::fs::write(
6487 mirrors_root.join("noisetable.toml"),
6488 "camp = \"noisetable\"\nregions = [\"pdx\"]\nworkloads = []\n",
6489 )
6490 .unwrap();
6491
6492 // Folder-form mirror
6493 let yah_com_dir = mirrors_root.join("yah-com");
6494 std::fs::create_dir_all(&yah_com_dir).unwrap();
6495 std::fs::write(
6496 yah_com_dir.join("mirror.toml"),
6497 "camp = \"yah\"\nregions = [\"pdx\"]\nworkloads = []\n",
6498 )
6499 .unwrap();
6500
6501 let cfg = CloudConfig::load(root).unwrap();
6502 assert_eq!(cfg.legacy_mirrors.len(), 2);
6503 assert!(cfg.legacy_mirror("noisetable").is_some());
6504 assert!(cfg.legacy_mirror("yah").is_some());
6505 }
6506
6507 #[test]
6508 fn mirror_malformed_fails_with_field_path() {
6509 // A malformed mirror.toml should fail at load with a clear error
6510 // that includes the file path.
6511 let tmp = tempfile::TempDir::new().unwrap();
6512 let root = tmp.path();
6513 let cloud_dir = make_legacy_cloud_dir(root);
6514 let mirror_dir = cloud_dir.join("mirrors").join("bad");
6515 std::fs::create_dir_all(&mirror_dir).unwrap();
6516 // Missing required `camp` field
6517 std::fs::write(
6518 mirror_dir.join("mirror.toml"),
6519 "regions = [\"pdx\"]\nworkloads = []\n",
6520 )
6521 .unwrap();
6522
6523 let err = CloudConfig::load(root).unwrap_err();
6524 let msg = err.to_string();
6525 assert!(
6526 msg.contains("mirror.toml"),
6527 "error should reference the file path, got: {msg}"
6528 );
6529 }
6530
6531 #[test]
6532 fn workload_config_load_and_validate() {
6533 use workload_spec::{
6534 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
6535 RestartPolicy, SchemaVersion, StopPolicy, TenantId, TierTag, WorkloadSpec,
6536 };
6537
6538 let tmp = tempfile::TempDir::new().unwrap();
6539 let root = tmp.path();
6540 let cloud_dir = make_legacy_cloud_dir(root);
6541 std::fs::create_dir_all(cloud_dir.join("workloads")).unwrap();
6542
6543 let spec = WorkloadSpec {
6544 schema_version: SchemaVersion::V1,
6545 name: "asset-registry".into(),
6546 image: ImageRef {
6547 registry: "ghcr.io".into(),
6548 repository: "noisetable/asset-registry".into(),
6549 tag: "v1.0.0".into(),
6550 digest: workload_spec::testing::test_digest(),
6551 },
6552 tier: TierTag("tenant".into()),
6553 replicas: 1,
6554 command: None,
6555 entrypoint: None,
6556 workdir: None,
6557 user: None,
6558 env: vec![],
6559 secrets: vec![],
6560 volumes: vec![],
6561 resources: ResourceLimits {
6562 memory_mb: 256,
6563 cpu_millis: 512,
6564 ephemeral_storage_mb: 512,
6565 },
6566 depends_on: vec![],
6567 requires: vec![],
6568 healthcheck: None,
6569 restart_policy: RestartPolicy::Always,
6570 archetype: None,
6571 stop_policy: StopPolicy {
6572 signal: 15,
6573 grace_period: workload_spec::Millis::from_secs(10),
6574 },
6575 expose: ExposeSpec {
6576 mesh: MeshExpose {
6577 identity: MeshIdent("asset-registry.pdx".into()),
6578 ports: MeshExpose::anonymous_ports([8080]),
6579 allow_from: vec![],
6580 },
6581 public: None,
6582 operator: None,
6583 },
6584 tenant: TenantId::singleton(),
6585 namespace: NamespaceId::singleton(),
6586 labels: Default::default(),
6587 annotations: Default::default(),
6588 files: Vec::new(),
6589 };
6590
6591 let toml_str = toml::to_string_pretty(&spec).unwrap();
6592 std::fs::write(cloud_dir.join("workloads/asset-registry.toml"), &toml_str).unwrap();
6593
6594 let cfg = CloudConfig::load(root).unwrap();
6595 assert_eq!(cfg.workloads.len(), 1);
6596 assert_eq!(cfg.workloads[0].spec.name, "asset-registry");
6597 assert_eq!(cfg.workload("asset-registry").unwrap().spec.replicas, 1);
6598 }
6599
6600 /// Minimal valid spec for the R215+ loader tests below. Kept as a helper so
6601 /// the two tests differ only in *where* the file lands, which is the whole
6602 /// thing under test.
6603 #[cfg(test)]
6604 fn minimal_spec(name: &str, replicas: u32) -> workload_spec::WorkloadSpec {
6605 use workload_spec::{
6606 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
6607 RestartPolicy, SchemaVersion, StopPolicy, TenantId, TierTag, WorkloadSpec,
6608 };
6609 WorkloadSpec {
6610 schema_version: SchemaVersion::V1,
6611 name: name.into(),
6612 image: ImageRef {
6613 registry: "cr.yah.dev".into(),
6614 repository: name.into(),
6615 tag: "v1".into(),
6616 digest: workload_spec::testing::test_digest(),
6617 },
6618 tier: TierTag("infra".into()),
6619 replicas,
6620 command: None,
6621 entrypoint: None,
6622 workdir: None,
6623 user: None,
6624 env: vec![],
6625 secrets: vec![],
6626 volumes: vec![],
6627 resources: ResourceLimits {
6628 memory_mb: 256,
6629 cpu_millis: 250,
6630 ephemeral_storage_mb: 128,
6631 },
6632 depends_on: vec![],
6633 requires: vec![],
6634 healthcheck: None,
6635 restart_policy: RestartPolicy::Always,
6636 archetype: None,
6637 stop_policy: StopPolicy {
6638 signal: 15,
6639 grace_period: workload_spec::Millis::from_secs(10),
6640 },
6641 expose: ExposeSpec {
6642 mesh: MeshExpose {
6643 identity: MeshIdent(name.into()),
6644 ports: MeshExpose::anonymous_ports([4325]),
6645 allow_from: vec![],
6646 },
6647 public: None,
6648 operator: None,
6649 },
6650 tenant: TenantId::singleton(),
6651 namespace: NamespaceId::singleton(),
6652 labels: Default::default(),
6653 annotations: Default::default(),
6654 files: Vec::new(),
6655 }
6656 }
6657
6658 /// R568-T7. Workloads must load from the R215+ tree.
6659 ///
6660 /// Before the fix this function tested, `CloudConfig::load` read workloads
6661 /// ONLY from the pre-R215 `.yah/cloud/workloads/` — which R222-B1 emptied —
6662 /// so in any modern camp `cfg.workload(name)` returned `None` for every
6663 /// name and the entire `yah cloud workload …` surface was unreachable. The
6664 /// CLI's own error text has said `.yah/infra/workloads/` throughout, so the
6665 /// bug read as "you must have typoed the filename".
6666 ///
6667 /// Note the fixture writes NO legacy `.yah/cloud/` dir at all: that is the
6668 /// shape of a real post-R215 camp, and it is exactly the shape the old code
6669 /// could not serve.
6670 #[test]
6671 fn workloads_load_from_the_infra_tree() {
6672 let tmp = tempfile::TempDir::new().unwrap();
6673 let root = tmp.path();
6674 let dir = crate::paths::workloads_dir(root);
6675 std::fs::create_dir_all(&dir).unwrap();
6676 std::fs::write(
6677 dir.join("yah-cloud-admin.toml"),
6678 toml::to_string_pretty(&minimal_spec("yah-cloud-admin", 1)).unwrap(),
6679 )
6680 .unwrap();
6681
6682 let cfg = CloudConfig::load(root).unwrap();
6683 assert_eq!(cfg.workloads.len(), 1);
6684 assert_eq!(
6685 cfg.workload("yah-cloud-admin").unwrap().spec.replicas,
6686 1,
6687 "a workload declared under .yah/infra/workloads/ must be resolvable by name"
6688 );
6689 }
6690
6691 /// A camp mid-migration can have both trees. R215+ wins on a name
6692 /// collision — same precedence the machine loader applies — so moving a
6693 /// declaration into `.yah/infra/workloads/` takes effect immediately
6694 /// instead of being silently shadowed by the copy left behind.
6695 #[test]
6696 fn infra_workload_shadows_the_legacy_copy_of_the_same_name() {
6697 let tmp = tempfile::TempDir::new().unwrap();
6698 let root = tmp.path();
6699
6700 let legacy = make_legacy_cloud_dir(root);
6701 std::fs::create_dir_all(legacy.join("workloads")).unwrap();
6702 std::fs::write(
6703 legacy.join("workloads/shared.toml"),
6704 toml::to_string_pretty(&minimal_spec("shared", 9)).unwrap(),
6705 )
6706 .unwrap();
6707 // Legacy-only name, to prove the old tree is still read rather than
6708 // replaced wholesale.
6709 std::fs::write(
6710 legacy.join("workloads/legacy-only.toml"),
6711 toml::to_string_pretty(&minimal_spec("legacy-only", 3)).unwrap(),
6712 )
6713 .unwrap();
6714
6715 let infra = crate::paths::workloads_dir(root);
6716 std::fs::create_dir_all(&infra).unwrap();
6717 std::fs::write(
6718 infra.join("shared.toml"),
6719 toml::to_string_pretty(&minimal_spec("shared", 1)).unwrap(),
6720 )
6721 .unwrap();
6722
6723 let cfg = CloudConfig::load(root).unwrap();
6724 assert_eq!(cfg.workloads.len(), 2, "one `shared`, plus `legacy-only`");
6725 assert_eq!(
6726 cfg.workload("shared").unwrap().spec.replicas,
6727 1,
6728 "the .yah/infra/ copy must win over the legacy one"
6729 );
6730 assert_eq!(cfg.workload("legacy-only").unwrap().spec.replicas, 3);
6731 }
6732
6733 #[test]
6734 fn workload_loader_rejects_bad_spec() {
6735 use workload_spec::{
6736 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
6737 RestartPolicy, SchemaVersion, StopPolicy, TenantId, TierTag, WorkloadSpec,
6738 };
6739
6740 let tmp = tempfile::TempDir::new().unwrap();
6741 let root = tmp.path();
6742 let cloud_dir = make_legacy_cloud_dir(root);
6743 std::fs::create_dir_all(cloud_dir.join("workloads")).unwrap();
6744
6745 // Construct a spec that round-trips through TOML but fails shape
6746 // validation: replicas = 200 is above the max of 100.
6747 let mut spec = WorkloadSpec {
6748 schema_version: SchemaVersion::V1,
6749 name: "asset-registry".into(),
6750 image: ImageRef {
6751 registry: "ghcr.io".into(),
6752 repository: "test/app".into(),
6753 tag: "v1".into(),
6754 digest: workload_spec::testing::test_digest(),
6755 },
6756 tier: TierTag("tenant".into()),
6757 replicas: 200, // ← invalid: exceeds max 100
6758 command: None,
6759 entrypoint: None,
6760 workdir: None,
6761 user: None,
6762 env: vec![],
6763 secrets: vec![],
6764 volumes: vec![],
6765 resources: ResourceLimits {
6766 memory_mb: 256,
6767 cpu_millis: 512,
6768 ephemeral_storage_mb: 512,
6769 },
6770 depends_on: vec![],
6771 requires: vec![],
6772 healthcheck: None,
6773 restart_policy: RestartPolicy::Always,
6774 archetype: None,
6775 stop_policy: StopPolicy {
6776 signal: 15,
6777 grace_period: workload_spec::Millis::from_secs(10),
6778 },
6779 expose: ExposeSpec {
6780 mesh: MeshExpose {
6781 identity: MeshIdent("asset-registry.pdx".into()),
6782 ports: MeshExpose::anonymous_ports([8080]),
6783 allow_from: vec![],
6784 },
6785 public: None,
6786 operator: None,
6787 },
6788 tenant: TenantId::singleton(),
6789 namespace: NamespaceId::singleton(),
6790 labels: Default::default(),
6791 annotations: Default::default(),
6792 files: Vec::new(),
6793 };
6794
6795 let toml_str = toml::to_string_pretty(&spec).unwrap();
6796 std::fs::write(cloud_dir.join("workloads/bad.toml"), &toml_str).unwrap();
6797
6798 let result = CloudConfig::load(root);
6799 assert!(
6800 result.is_err(),
6801 "loading a WorkloadSpec with replicas=200 should return Err"
6802 );
6803 let msg = result.unwrap_err().to_string();
6804 assert!(
6805 msg.contains("shape validation")
6806 || msg.contains("Replicas")
6807 || msg.contains("replicas"),
6808 "error should mention shape validation or replicas field, got: {msg}"
6809 );
6810
6811 // The `spec` binding is only used for the write — suppress warning.
6812 let _ = &mut spec;
6813 }
6814
6815 #[test]
6816 fn workload_config_save_round_trip() {
6817 use workload_spec::{
6818 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
6819 RestartPolicy, SchemaVersion, StopPolicy, TenantId, TierTag, WorkloadSpec,
6820 };
6821
6822 let tmp = tempfile::TempDir::new().unwrap();
6823 let root = tmp.path();
6824
6825 let spec = WorkloadSpec {
6826 schema_version: SchemaVersion::V1,
6827 name: "signing-service".into(),
6828 image: ImageRef {
6829 registry: "ghcr.io".into(),
6830 repository: "noisetable/signing".into(),
6831 tag: "v2.0.0".into(),
6832 digest: workload_spec::testing::test_digest(),
6833 },
6834 tier: TierTag("private".into()),
6835 replicas: 2,
6836 command: None,
6837 entrypoint: None,
6838 workdir: None,
6839 user: None,
6840 env: vec![],
6841 secrets: vec![],
6842 volumes: vec![],
6843 resources: ResourceLimits {
6844 memory_mb: 128,
6845 cpu_millis: 256,
6846 ephemeral_storage_mb: 256,
6847 },
6848 depends_on: vec![],
6849 requires: vec![],
6850 healthcheck: None,
6851 restart_policy: RestartPolicy::Always,
6852 archetype: None,
6853 stop_policy: StopPolicy {
6854 signal: 15,
6855 grace_period: workload_spec::Millis::from_secs(5),
6856 },
6857 expose: ExposeSpec {
6858 mesh: MeshExpose {
6859 identity: MeshIdent("signing.pdx".into()),
6860 ports: MeshExpose::anonymous_ports([9090]),
6861 allow_from: vec![],
6862 },
6863 public: None,
6864 operator: None,
6865 },
6866 tenant: TenantId::singleton(),
6867 namespace: NamespaceId::singleton(),
6868 labels: Default::default(),
6869 annotations: Default::default(),
6870 files: Vec::new(),
6871 };
6872
6873 let wc = WorkloadConfig { spec };
6874 let cloud_dir = make_legacy_cloud_dir(root);
6875 wc.save(&cloud_dir).unwrap();
6876
6877 let loaded = CloudConfig::load(root).unwrap();
6878 assert_eq!(loaded.workloads.len(), 1);
6879 assert_eq!(loaded.workloads[0].spec.name, "signing-service");
6880 assert_eq!(loaded.workloads[0].spec.replicas, 2);
6881 }
6882
6883 #[test]
6884 fn machine_save_write_back_fingerprint() {
6885 let tmp = tempfile::TempDir::new().unwrap();
6886 let root = tmp.path();
6887
6888 let mut machine = MachineConfig {
6889 name: "test-pdx-1".into(),
6890 provider: "hetzner".into(),
6891 location: Some("pdx".into()),
6892 server_type: Some("cpx22".into()),
6893 hosts_mirrors: vec![],
6894 mesh_tags: vec![],
6895 region: None,
6896 zone: None,
6897 arch: None,
6898 bucket: None,
6899 vendor: None,
6900 nickname: None,
6901 legacy_hostkey_fingerprint: None,
6902 registration: Default::default(),
6903 ssh_keys: vec![],
6904 cloudflared: None,
6905 hosts_operator_bridge: false,
6906 connect: None,
6907 allocatable: None,
6908 taints: vec![],
6909 sovereign_group: None,
6910 sovereign_role: None,
6911 ingress_floating_ip: None,
6912 };
6913 machine.save(root).unwrap();
6914
6915 // Simulate A4: write back the hostkey fingerprint after provision.
6916 // R707-T1: registration is the write target; the accessor is the read.
6917 machine.registration.hostkey_fingerprint = Some("SHA256:abc123".into());
6918 machine.save(root).unwrap();
6919
6920 let reloaded: Vec<MachineConfig> = load_dir(root.join("machines")).unwrap();
6921 assert_eq!(reloaded.len(), 1);
6922 assert_eq!(reloaded[0].hostkey_fingerprint(), Some("SHA256:abc123"));
6923 }
6924
6925 // ─── New-shape (R222 B2) parse tests ────────────────────────────────────
6926 //
6927 // These mirror the Phase-A manifests committed under `.yah/services/` and
6928 // `.yah/infra/providers/`. Keeping the test strings inline (rather than
6929 // reading the on-disk files) so the loader stays runnable in any workdir
6930 // and so accidental edits to the on-disk files don't silently change
6931 // schema expectations.
6932
6933 #[test]
6934 fn provider_cloudflare_round_trips() {
6935 let src = r#"
6936schema_version = 1
6937id = "cloudflare"
6938kind = "cloudflare"
6939credentials = "keystore://cloudflare/yah"
6940default_zone = "yah.dev"
6941"#;
6942 let cfg: ProviderConfig = toml::from_str(src).unwrap();
6943 assert_eq!(cfg.id, "cloudflare");
6944 assert_eq!(cfg.kind, Provider::Cloudflare);
6945 assert_eq!(
6946 cfg.credentials.as_deref(),
6947 Some("keystore://cloudflare/yah")
6948 );
6949 assert_eq!(
6950 cfg.fields.get("default_zone").and_then(|v| v.as_str()),
6951 Some("yah.dev"),
6952 );
6953 let back = toml::to_string(&cfg).unwrap();
6954 let again: ProviderConfig = toml::from_str(&back).unwrap();
6955 assert_eq!(again.id, cfg.id);
6956 assert_eq!(again.kind, cfg.kind);
6957 }
6958
6959 #[test]
6960 fn provider_hetzner_round_trips() {
6961 let src = r#"
6962schema_version = 1
6963id = "hetzner"
6964kind = "hetzner"
6965credentials = "keystore://hetzner/yah"
6966default_location = "pdx"
6967default_server_type = "cpx11"
6968ssh_keys = []
6969"#;
6970 let cfg: ProviderConfig = toml::from_str(src).unwrap();
6971 assert_eq!(cfg.kind, Provider::Hetzner);
6972 assert_eq!(
6973 cfg.fields.get("default_location").and_then(|v| v.as_str()),
6974 Some("pdx"),
6975 );
6976 assert!(
6977 cfg.fields
6978 .get("ssh_keys")
6979 .map(|v| v.as_array().unwrap().is_empty())
6980 .unwrap_or(false),
6981 "ssh_keys must round-trip as empty array, got {:?}",
6982 cfg.fields.get("ssh_keys"),
6983 );
6984 }
6985
6986 #[test]
6987 fn provider_orbstack_local_container_round_trips() {
6988 let src = r#"
6989schema_version = 1
6990id = "orbstack"
6991kind = "local-container"
6992runtime = "auto"
6993
6994[discovery]
6995orbstack = "~/.orbstack/run/docker.sock"
6996colima = "~/.colima/default/docker.sock"
6997docker = "/var/run/docker.sock"
6998"#;
6999 let cfg: ProviderConfig = toml::from_str(src).unwrap();
7000 assert_eq!(cfg.kind, Provider::LocalContainer);
7001 assert_eq!(
7002 cfg.fields.get("runtime").and_then(|v| v.as_str()),
7003 Some("auto"),
7004 );
7005 let discovery = cfg
7006 .fields
7007 .get("discovery")
7008 .and_then(|v| v.as_table())
7009 .expect("discovery table");
7010 assert!(discovery.contains_key("orbstack"));
7011 assert!(discovery.contains_key("colima"));
7012 assert!(discovery.contains_key("docker"));
7013 }
7014
7015 #[test]
7016 fn provider_unknown_kind_fails() {
7017 let src = r#"
7018schema_version = 1
7019id = "made-up"
7020kind = "fly-io"
7021"#;
7022 let err = toml::from_str::<ProviderConfig>(src).unwrap_err();
7023 let msg = err.to_string();
7024 assert!(
7025 msg.contains("kind") || msg.contains("variant"),
7026 "unknown provider kind should surface as a serde error, got: {msg}"
7027 );
7028 }
7029
7030 #[test]
7031 fn service_dev_yah_round_trips() {
7032 let src = r#"
7033schema_version = 1
7034name = "dev-yah"
7035domain = "yah.dev"
7036
7037[[components]]
7038id = "site"
7039kind = "mesofact-static"
7040path = "app/yah/web"
7041role = "static"
7042"#;
7043 let cfg: ServiceConfig = toml::from_str(src).unwrap();
7044 assert_eq!(cfg.name, "dev-yah");
7045 assert_eq!(cfg.domain, "yah.dev");
7046 assert_eq!(cfg.components.len(), 1);
7047 let c = &cfg.components[0];
7048 assert_eq!(c.id, "site");
7049 assert_eq!(c.kind, "mesofact-static");
7050 assert_eq!(c.path, "app/yah/web");
7051 assert_eq!(c.role, "static");
7052 assert!(c.publishes.is_none());
7053
7054 let back = toml::to_string(&cfg).unwrap();
7055 let again: ServiceConfig = toml::from_str(&back).unwrap();
7056 assert_eq!(again.name, cfg.name);
7057 assert_eq!(again.components[0].kind, c.kind);
7058 }
7059
7060 #[test]
7061 fn mirror_prod_cloudflare_reference_parses() {
7062 let src = r#"
7063schema_version = 1
7064shape = "single-machine"
7065
7066[providers.static]
7067use = "cloudflare"
7068bucket = "yah-dev"
7069zone = "yah.dev"
7070dns = { record = "@", type = "CNAME" }
7071"#;
7072 let cfg: MirrorConfig = toml::from_str(src).unwrap();
7073 assert_eq!(cfg.shape, MirrorShape::SingleMachine);
7074 let slot = cfg.providers.get("static").expect("static slot");
7075 assert_eq!(slot.provider_id(), Some("cloudflare"));
7076 assert!(slot.inline_kind().is_none());
7077 if let MirrorProviderSlot::Reference { fields, .. } = slot {
7078 assert_eq!(
7079 fields.get("bucket").and_then(|v| v.as_str()),
7080 Some("yah-dev")
7081 );
7082 assert_eq!(fields.get("zone").and_then(|v| v.as_str()), Some("yah.dev"));
7083 let dns = fields
7084 .get("dns")
7085 .and_then(|v| v.as_table())
7086 .expect("dns table");
7087 assert_eq!(dns.get("record").and_then(|v| v.as_str()), Some("@"));
7088 assert_eq!(dns.get("type").and_then(|v| v.as_str()), Some("CNAME"));
7089 } else {
7090 panic!("expected Reference slot");
7091 }
7092 }
7093
7094 #[test]
7095 fn mirror_local_inline_static_and_orbstack_compute_parse() {
7096 let src = r#"
7097schema_version = 1
7098shape = "local"
7099
7100[providers.static]
7101kind = "local-static"
7102port = 4321
7103artifact_dir = ".yah/infra/state/local/static"
7104
7105[providers.compute]
7106use = "orbstack"
7107"#;
7108 let cfg: MirrorConfig = toml::from_str(src).unwrap();
7109 assert_eq!(cfg.shape, MirrorShape::Local);
7110
7111 let static_slot = cfg.providers.get("static").expect("static slot");
7112 assert_eq!(static_slot.inline_kind(), Some(Provider::LocalStatic));
7113 assert!(static_slot.provider_id().is_none());
7114 if let MirrorProviderSlot::Inline { fields, .. } = static_slot {
7115 assert_eq!(fields.get("port").and_then(|v| v.as_integer()), Some(4321));
7116 assert_eq!(
7117 fields.get("artifact_dir").and_then(|v| v.as_str()),
7118 Some(".yah/infra/state/local/static"),
7119 );
7120 } else {
7121 panic!("expected Inline slot for static");
7122 }
7123
7124 let compute_slot = cfg.providers.get("compute").expect("compute slot");
7125 assert_eq!(compute_slot.provider_id(), Some("orbstack"));
7126 }
7127
7128 #[test]
7129 fn mirror_pond_miniflare_minio_parse() {
7130 // pond-tier mirror: miniflare-container + minio, both inline.
7131 // T1 just needs these inline kinds to parse — the reconciler dispatch
7132 // arrives in R256-T3.
7133 let src = r#"
7134schema_version = 1
7135shape = "local"
7136
7137[providers.static]
7138kind = "miniflare-container"
7139port = 4322
7140bucket = "yah-dev"
7141
7142[providers.object_store]
7143kind = "minio-container"
7144api_port = 9000
7145console_port = 9001
7146bucket = "yah-dev"
7147"#;
7148 let cfg: MirrorConfig = toml::from_str(src).unwrap();
7149 assert_eq!(cfg.shape, MirrorShape::Local);
7150
7151 let static_slot = cfg.providers.get("static").expect("static slot");
7152 assert_eq!(
7153 static_slot.inline_kind(),
7154 Some(Provider::MiniflareContainer)
7155 );
7156 if let MirrorProviderSlot::Inline { fields, .. } = static_slot {
7157 assert_eq!(fields.get("port").and_then(|v| v.as_integer()), Some(4322));
7158 assert_eq!(
7159 fields.get("bucket").and_then(|v| v.as_str()),
7160 Some("yah-dev")
7161 );
7162 } else {
7163 panic!("expected Inline slot for miniflare-container static");
7164 }
7165
7166 let object_store_slot = cfg
7167 .providers
7168 .get("object_store")
7169 .expect("object_store slot");
7170 assert_eq!(
7171 object_store_slot.inline_kind(),
7172 Some(Provider::MinioContainer)
7173 );
7174 if let MirrorProviderSlot::Inline { fields, .. } = object_store_slot {
7175 assert_eq!(
7176 fields.get("api_port").and_then(|v| v.as_integer()),
7177 Some(9000)
7178 );
7179 assert_eq!(
7180 fields.get("console_port").and_then(|v| v.as_integer()),
7181 Some(9001)
7182 );
7183 assert_eq!(
7184 fields.get("bucket").and_then(|v| v.as_str()),
7185 Some("yah-dev")
7186 );
7187 } else {
7188 panic!("expected Inline slot for minio-container object_store");
7189 }
7190 }
7191
7192 #[test]
7193 fn provider_miniflare_container_kind_round_trips() {
7194 // Inline-only kind; never declared as a standalone provider file but
7195 // the enum round-trip is still exercised through ProviderConfig because
7196 // schemars/serde share the variant table.
7197 let cfg = MirrorProviderSlot::Inline {
7198 kind: Provider::MiniflareContainer,
7199 fields: BTreeMap::new(),
7200 };
7201 let s = toml::to_string(&cfg).unwrap();
7202 assert!(
7203 s.contains("kind = \"miniflare-container\""),
7204 "kebab-case wire form expected, got: {s}"
7205 );
7206 let back: MirrorProviderSlot = toml::from_str(&s).unwrap();
7207 assert_eq!(back.inline_kind(), Some(Provider::MiniflareContainer));
7208 }
7209
7210 #[test]
7211 fn provider_minio_container_kind_round_trips() {
7212 let cfg = MirrorProviderSlot::Inline {
7213 kind: Provider::MinioContainer,
7214 fields: BTreeMap::new(),
7215 };
7216 let s = toml::to_string(&cfg).unwrap();
7217 assert!(
7218 s.contains("kind = \"minio-container\""),
7219 "kebab-case wire form expected, got: {s}"
7220 );
7221 let back: MirrorProviderSlot = toml::from_str(&s).unwrap();
7222 assert_eq!(back.inline_kind(), Some(Provider::MinioContainer));
7223 }
7224
7225 #[test]
7226 fn mirror_compute_slot_with_machine_reference_parses() {
7227 // The on-disk prod.toml has a commented-out compute slot; this test
7228 // covers the form Phase B will need once yubaba is provisioned.
7229 let src = r#"
7230schema_version = 1
7231shape = "single-machine"
7232
7233[providers.compute]
7234use = "hetzner"
7235machine = "yah-cloud-1"
7236"#;
7237 let cfg: MirrorConfig = toml::from_str(src).unwrap();
7238 let slot = cfg.providers.get("compute").expect("compute slot");
7239 assert_eq!(slot.provider_id(), Some("hetzner"));
7240 if let MirrorProviderSlot::Reference { fields, .. } = slot {
7241 assert_eq!(
7242 fields.get("machine").and_then(|v| v.as_str()),
7243 Some("yah-cloud-1"),
7244 );
7245 }
7246 }
7247
7248 #[test]
7249 fn machine_yah_cloud_1_round_trips_with_existing_shape() {
7250 // The current machine TOML predates B2 — MachineConfig hasn't been
7251 // reshaped yet. This locks the expected shape so we notice if B3
7252 // accidentally regresses it.
7253 let src = r#"
7254name = "yah-cloud-1"
7255provider = "hetzner"
7256location = "pdx"
7257server_type = "cpx11"
7258hosts_mirrors = []
7259mesh_tags = ["tag:tier-scratch", "tag:primary-yah"]
7260ssh_keys = [111513970, 111525493]
7261"#;
7262 let cfg: MachineConfig = toml::from_str(src).unwrap();
7263 assert_eq!(cfg.name, "yah-cloud-1");
7264 assert_eq!(cfg.provider, "hetzner");
7265 assert_eq!(cfg.ssh_keys.len(), 2);
7266 }
7267
7268 #[test]
7269 fn static_node_omits_location_server_type_and_carries_connect() {
7270 // BYO Phase-0: a `static` node we brought up over SSH has no provider
7271 // DC code or SKU; it declares reach in `[connect]` instead. Must load.
7272 let src = r#"
7273name = "us-south-001"
7274provider = "static"
7275region = "us-south"
7276mesh_tags = ["tag:cloud-runner", "tag:voter-candidate"]
7277
7278[connect]
7279address = "45.32.194.254"
7280ssh = "root@45.32.194.254"
7281identity_file = "~/.ssh/yah"
7282yubaba = "http://127.0.0.1:7443"
7283arch = "x86_64"
7284"#;
7285 let cfg: MachineConfig = toml::from_str(src).unwrap();
7286 assert_eq!(cfg.provider, "static");
7287 assert!(cfg.location.is_none());
7288 assert!(cfg.server_type.is_none());
7289 assert_eq!(cfg.location(), ""); // accessor defaults empty
7290 let c = cfg.connect.as_ref().expect("connect block");
7291 assert_eq!(c.ssh, "root@45.32.194.254");
7292 // Loopback is a *declared* reach placeholder, so it stays in [connect]
7293 // verbatim and composes straight through (R707-T1).
7294 assert_eq!(c.yubaba.as_deref(), Some("http://127.0.0.1:7443"));
7295 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://127.0.0.1:7443"));
7296 assert_eq!(cfg.mesh_ipv4(), None);
7297 // Static providers have no driver, so validate() is a no-op pass.
7298 assert!(!provider_has_machine_driver(&cfg.provider));
7299 cfg.validate().unwrap();
7300 }
7301
7302 // ─── R707-T1: declaration / registration split ──────────────────────────
7303
7304 /// The pre-split shape — top-level `hostkey_fingerprint`, mesh IP baked
7305 /// into `[connect].yubaba` — must keep parsing, and must read back through
7306 /// the accessors identically. Every machine TOML in the fleet was written
7307 /// this way, and other camps' inventories still are.
7308 #[test]
7309 fn legacy_shape_still_parses_and_reads_through_accessors() {
7310 let src = r#"
7311name = "us-west-001"
7312provider = "static"
7313region = "us-west"
7314arch = "x86_64"
7315mesh_tags = ["tag:cloud-runner"]
7316hostkey_fingerprint = "SHA256:dmpq"
7317
7318[connect]
7319address = "15.204.89.240"
7320ssh = "debian@15.204.89.240"
7321identity_file = "~/.ssh/yah"
7322yubaba = "http://100.64.0.1:7443"
7323"#;
7324 let cfg: MachineConfig = toml::from_str(src).unwrap();
7325 assert_eq!(cfg.hostkey_fingerprint(), Some("SHA256:dmpq"));
7326 assert_eq!(cfg.mesh_ipv4(), Some("100.64.0.1"));
7327 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://100.64.0.1:7443"));
7328 }
7329
7330 /// The post-split shape reads identically to the legacy one above — same
7331 /// three accessor answers from a file that separates the two halves. This
7332 /// is the "unchanged in meaning" guarantee the fleet migration rests on.
7333 #[test]
7334 fn split_shape_is_equivalent_to_legacy_shape() {
7335 let legacy = r#"
7336name = "m"
7337provider = "static"
7338mesh_tags = []
7339hostkey_fingerprint = "SHA256:dmpq"
7340
7341[connect]
7342address = "15.204.89.240"
7343ssh = "debian@15.204.89.240"
7344identity_file = "~/.ssh/yah"
7345yubaba = "http://100.64.0.1:7443"
7346"#;
7347 let split = r#"
7348name = "m"
7349provider = "static"
7350mesh_tags = []
7351
7352[connect]
7353address = "15.204.89.240"
7354ssh = "debian@15.204.89.240"
7355identity_file = "~/.ssh/yah"
7356
7357[registration]
7358hostkey_fingerprint = "SHA256:dmpq"
7359mesh_ipv4 = "100.64.0.1"
7360"#;
7361 let old: MachineConfig = toml::from_str(legacy).unwrap();
7362 let new: MachineConfig = toml::from_str(split).unwrap();
7363 assert_eq!(old.hostkey_fingerprint(), new.hostkey_fingerprint());
7364 assert_eq!(old.mesh_ipv4(), new.mesh_ipv4());
7365 assert_eq!(old.yubaba_url(), new.yubaba_url());
7366 }
7367
7368 /// A non-default `[connect].yubaba_port` is declared reach and composes
7369 /// with the observed mesh address rather than being pinned into a URL.
7370 #[test]
7371 fn declared_port_composes_with_observed_mesh_address() {
7372 let src = r#"
7373name = "m"
7374provider = "static"
7375mesh_tags = []
7376
7377[connect]
7378address = "10.0.0.1"
7379ssh = "yah@10.0.0.1"
7380identity_file = "~/.ssh/yah"
7381yubaba_port = 9443
7382
7383[registration]
7384mesh_ipv4 = "100.64.0.9"
7385"#;
7386 let cfg: MachineConfig = toml::from_str(src).unwrap();
7387 assert_eq!(cfg.connect.as_ref().unwrap().yubaba_port(), 9443);
7388 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://100.64.0.9:9443"));
7389 }
7390
7391 /// R605-T10 inverts R707-T6 for the private-literal case, and this is the
7392 /// node it was inverted for: us-west-014's shape, mesh-joined AND declaring
7393 /// a LAN `[connect].yubaba`. R707-T6 made the literal win outright so
7394 /// `rollout::yubaba::membership_to_nodes` could match the dev group's
7395 /// LAN-addressed raft membership — which fused identity into reach and made
7396 /// every automated dial go to an address only bldg-2506 can route.
7397 /// `lan_endpoint()` now serves that match, so the mesh address wins the
7398 /// dial and the literal is inert.
7399 #[test]
7400 fn a_private_literal_loses_to_the_registered_mesh_address() {
7401 let src = r#"
7402name = "us-west-014"
7403provider = "static"
7404mesh_tags = []
7405
7406[connect]
7407address = "192.168.10.14"
7408ssh = "yah@192.168.10.14"
7409identity_file = "~/.ssh/yah"
7410yubaba = "http://192.168.10.14:7443"
7411
7412[registration]
7413mesh_ipv4 = "100.64.0.6"
7414"#;
7415 let cfg: MachineConfig = toml::from_str(src).unwrap();
7416 assert_eq!(cfg.mesh_ipv4(), Some("100.64.0.6"), "still mesh-joined");
7417 assert_eq!(
7418 cfg.yubaba_url().as_deref(),
7419 Some("http://100.64.0.6:7443"),
7420 "automation dials the mesh, never the LAN literal"
7421 );
7422 assert_eq!(
7423 cfg.lan_endpoint().as_deref(),
7424 Some("192.168.10.14:7443"),
7425 "the LAN address is still recorded — as identity, not as reach"
7426 );
7427 }
7428
7429 /// The refusal R605-T10 asks for: a node whose ONLY declared reach is a LAN
7430 /// literal is unresolvable, and says so by name rather than returning a URL
7431 /// that will time out. us-west-011's shape before this ticket.
7432 #[test]
7433 fn a_lan_only_node_refuses_with_a_named_reason() {
7434 let src = r#"
7435name = "us-west-011"
7436provider = "static"
7437mesh_tags = []
7438
7439[connect]
7440address = "192.168.10.11"
7441ssh = "yah@192.168.10.11"
7442identity_file = "~/.ssh/yah"
7443yubaba = "http://192.168.10.11:7443"
7444"#;
7445 let cfg: MachineConfig = toml::from_str(src).unwrap();
7446 assert_eq!(cfg.yubaba_url(), None);
7447 let err = cfg.reach().unwrap_err();
7448 assert!(err.contains("us-west-011"), "{err}");
7449 assert!(err.contains("192.168.10.11"), "{err}");
7450 assert!(err.contains("mesh_ipv4"), "{err}");
7451 }
7452
7453 /// The loopback placeholder is a genuine declaration ("reach me through the
7454 /// SSH tunnel"), not a LAN literal — 127/8 is not RFC1918. It must keep
7455 /// resolving verbatim; `hub::coordinator::is_loopback_url` is what judges it
7456 /// downstream.
7457 #[test]
7458 fn a_loopback_placeholder_still_resolves_verbatim() {
7459 let src = r#"
7460name = "m"
7461provider = "static"
7462mesh_tags = []
7463
7464[connect]
7465address = "192.168.10.99"
7466ssh = "yah@192.168.10.99"
7467identity_file = "~/.ssh/yah"
7468yubaba = "http://127.0.0.1:7443"
7469"#;
7470 let cfg: MachineConfig = toml::from_str(src).unwrap();
7471 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://127.0.0.1:7443"));
7472 }
7473
7474 #[test]
7475 fn private_ranges_are_exactly_rfc1918() {
7476 for lan in [
7477 "http://192.168.10.11:7443",
7478 "http://10.0.0.5:7443",
7479 "http://172.16.4.1:7443",
7480 ] {
7481 assert!(private_ipv4_from_url(lan).is_some(), "{lan}");
7482 }
7483 for not_lan in [
7484 "http://100.64.0.6:7443", // mesh
7485 "http://127.0.0.1:7443", // loopback
7486 "http://172.32.0.1:7443", // just past 172.16/12
7487 "http://45.32.194.254:80", // public
7488 "http://us-west-001:7443", // name, not a literal
7489 ] {
7490 assert!(private_ipv4_from_url(not_lan).is_none(), "{not_lan}");
7491 }
7492 }
7493
7494 /// `normalize` migrates in place: the legacy fingerprint moves into
7495 /// `[registration]`, the mesh IP is lifted out of the URL, and the derived
7496 /// `[connect].yubaba` is cleared so the two halves cannot drift.
7497 #[test]
7498 fn normalize_migrates_legacy_fields_and_is_idempotent() {
7499 let src = r#"
7500name = "m"
7501provider = "static"
7502mesh_tags = []
7503hostkey_fingerprint = "SHA256:dmpq"
7504
7505[connect]
7506address = "15.204.89.240"
7507ssh = "debian@15.204.89.240"
7508identity_file = "~/.ssh/yah"
7509yubaba = "http://100.64.0.1:7443"
7510"#;
7511 let mut cfg: MachineConfig = toml::from_str(src).unwrap();
7512 cfg.normalize();
7513 assert!(cfg.legacy_hostkey_fingerprint.is_none());
7514 assert_eq!(
7515 cfg.registration.hostkey_fingerprint.as_deref(),
7516 Some("SHA256:dmpq")
7517 );
7518 assert_eq!(cfg.registration.mesh_ipv4.as_deref(), Some("100.64.0.1"));
7519 assert!(cfg.connect.as_ref().unwrap().yubaba.is_none());
7520 // Accessors still answer the same, and re-running changes nothing.
7521 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://100.64.0.1:7443"));
7522 let once = format!("{cfg:?}");
7523 cfg.normalize();
7524 assert_eq!(once, format!("{cfg:?}"));
7525 }
7526
7527 /// A loopback `[connect].yubaba` is a declaration ("no mesh address yet —
7528 /// reach me through the SSH tunnel"), not a stale observation, so
7529 /// `normalize` must leave it alone. us-west-003/011/013 depend on this.
7530 #[test]
7531 fn normalize_leaves_pre_mesh_loopback_declaration_intact() {
7532 let src = r#"
7533name = "m"
7534provider = "static"
7535mesh_tags = []
7536
7537[connect]
7538address = "192.168.10.11"
7539ssh = "yah@192.168.10.11"
7540identity_file = "~/.ssh/yah"
7541yubaba = "http://127.0.0.1:7443"
7542"#;
7543 let mut cfg: MachineConfig = toml::from_str(src).unwrap();
7544 cfg.normalize();
7545 assert_eq!(
7546 cfg.connect.as_ref().unwrap().yubaba.as_deref(),
7547 Some("http://127.0.0.1:7443")
7548 );
7549 assert!(cfg.registration.is_empty());
7550 assert_eq!(cfg.mesh_ipv4(), None);
7551 }
7552
7553 /// `save` normalizes, so a legacy file that round-trips through the writer
7554 /// comes back on the split shape with nothing lost — the property that
7555 /// keeps `yah cloud machine attach` from re-emitting the old layout.
7556 #[test]
7557 fn save_writes_the_split_shape_from_a_legacy_config() {
7558 let tmp = tempfile::TempDir::new().unwrap();
7559 let root = tmp.path();
7560 let src = r#"
7561name = "m"
7562provider = "static"
7563mesh_tags = []
7564hostkey_fingerprint = "SHA256:dmpq"
7565
7566[connect]
7567address = "15.204.89.240"
7568ssh = "debian@15.204.89.240"
7569identity_file = "~/.ssh/yah"
7570yubaba = "http://100.64.0.1:7443"
7571"#;
7572 let cfg: MachineConfig = toml::from_str(src).unwrap();
7573 cfg.save(root).unwrap();
7574
7575 let written = std::fs::read_to_string(root.join("machines/m.toml")).unwrap();
7576 let reg_at = written
7577 .find("[registration]")
7578 .unwrap_or_else(|| panic!("no [registration] table: {written}"));
7579 let fp_at = written
7580 .find("hostkey_fingerprint")
7581 .unwrap_or_else(|| panic!("fingerprint dropped: {written}"));
7582 assert!(
7583 fp_at > reg_at,
7584 "legacy top-level field must not be re-emitted: {written}"
7585 );
7586 assert!(
7587 !written.contains("yubaba ="),
7588 "derived URL must not be re-emitted alongside mesh_ipv4: {written}"
7589 );
7590
7591 let reloaded: MachineConfig = toml::from_str(&written).unwrap();
7592 assert_eq!(reloaded.hostkey_fingerprint(), Some("SHA256:dmpq"));
7593 assert_eq!(
7594 reloaded.yubaba_url().as_deref(),
7595 Some("http://100.64.0.1:7443")
7596 );
7597 }
7598
7599 /// `[registration]` is omitted entirely for a machine nothing has been
7600 /// observed about — a scaffolded declaration stays clean.
7601 #[test]
7602 fn empty_registration_is_omitted_on_serialize() {
7603 let src = r#"
7604name = "m"
7605provider = "static"
7606mesh_tags = []
7607"#;
7608 let cfg: MachineConfig = toml::from_str(src).unwrap();
7609 assert!(cfg.registration.is_empty());
7610 let out = toml::to_string_pretty(&cfg).unwrap();
7611 assert!(!out.contains("[registration]"), "{out}");
7612 }
7613
7614 #[test]
7615 fn driver_provider_without_location_fails_validate() {
7616 // A driver-backed provider (hetzner/vultr) still MUST carry location +
7617 // server_type — the driver can't create a server without them. The
7618 // contract moved from load-time (required field) to provision-time
7619 // (validate), so the TOML loads but validate() rejects it.
7620 let src = r#"
7621name = "us-west-001"
7622provider = "hetzner"
7623mesh_tags = []
7624"#;
7625 let cfg: MachineConfig = toml::from_str(src).unwrap();
7626 assert!(provider_has_machine_driver(&cfg.provider));
7627 let err = cfg.validate().unwrap_err().to_string();
7628 assert!(
7629 err.contains("location"),
7630 "expected location complaint: {err}"
7631 );
7632 }
7633
7634 /// Helper for the new-tree integration tests below: lay out
7635 /// `<workspace>/.yah/{infra,services}/` with `dev-yah` + its mirrors and
7636 /// the three Phase-A providers (cloudflare, hetzner, orbstack).
7637 fn make_new_tree_with_dev_yah(root: &std::path::Path) {
7638 let infra = root.join(".yah").join("infra");
7639 let providers = infra.join("providers");
7640 std::fs::create_dir_all(&providers).unwrap();
7641 std::fs::write(
7642 providers.join("cloudflare.toml"),
7643 r#"schema_version = 1
7644id = "cloudflare"
7645kind = "cloudflare"
7646credentials = "keystore://cloudflare/yah"
7647default_zone = "yah.dev"
7648"#,
7649 )
7650 .unwrap();
7651 std::fs::write(
7652 providers.join("hetzner.toml"),
7653 r#"schema_version = 1
7654id = "hetzner"
7655kind = "hetzner"
7656credentials = "keystore://hetzner/yah"
7657default_location = "pdx"
7658default_server_type = "cpx11"
7659ssh_keys = []
7660"#,
7661 )
7662 .unwrap();
7663 std::fs::write(
7664 providers.join("orbstack.toml"),
7665 r#"schema_version = 1
7666id = "orbstack"
7667kind = "local-container"
7668runtime = "auto"
7669
7670[discovery]
7671orbstack = "~/.orbstack/run/docker.sock"
7672"#,
7673 )
7674 .unwrap();
7675
7676 let svc = root.join(".yah").join("services").join("dev-yah");
7677 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
7678 std::fs::write(
7679 svc.join("service.toml"),
7680 r#"schema_version = 1
7681name = "dev-yah"
7682domain = "yah.dev"
7683
7684[[components]]
7685id = "site"
7686kind = "mesofact-static"
7687path = "app/yah/web"
7688role = "static"
7689"#,
7690 )
7691 .unwrap();
7692 std::fs::write(
7693 svc.join("mirrors/prod.toml"),
7694 r#"schema_version = 1
7695shape = "single-machine"
7696
7697[providers.static]
7698use = "cloudflare"
7699bucket = "yah-dev"
7700zone = "yah.dev"
7701"#,
7702 )
7703 .unwrap();
7704 std::fs::write(
7705 svc.join("mirrors/local.toml"),
7706 r#"schema_version = 1
7707shape = "local"
7708
7709[providers.static]
7710kind = "local-static"
7711port = 4321
7712
7713[providers.compute]
7714use = "orbstack"
7715"#,
7716 )
7717 .unwrap();
7718 }
7719
7720 #[test]
7721 fn cloud_config_load_new_tree_populates_providers_and_services() {
7722 let tmp = tempfile::TempDir::new().unwrap();
7723 let root = tmp.path();
7724 make_new_tree_with_dev_yah(root);
7725
7726 let cfg = CloudConfig::load(root).unwrap();
7727 assert_eq!(cfg.providers.len(), 3, "three providers loaded");
7728 assert!(cfg.provider("cloudflare").is_some());
7729 assert!(cfg.provider("hetzner").is_some());
7730 assert!(cfg.provider("orbstack").is_some());
7731
7732 let dev = cfg.service("dev-yah").expect("dev-yah service");
7733 assert_eq!(dev.service.domain, "yah.dev");
7734 assert_eq!(dev.service.components.len(), 1);
7735 assert_eq!(dev.mirrors.len(), 2);
7736 // Legacy file stems "prod" and "local" are normalised to canonical tier names.
7737 assert!(dev.mirrors.contains_key("cloud"), "prod.toml → cloud tier");
7738 assert!(dev.mirrors.contains_key("dev"), "local.toml → dev tier");
7739 assert_eq!(dev.mirrors["cloud"].shape, MirrorShape::SingleMachine);
7740 assert_eq!(dev.mirrors["dev"].shape, MirrorShape::Local);
7741
7742 // Legacy fields stay empty when no .yah/cloud/ exists.
7743 assert!(cfg.legacy_mirrors.is_empty());
7744 assert!(cfg.legacy_services.is_empty());
7745 assert!(cfg.workloads.is_empty());
7746 }
7747
7748 #[test]
7749 fn cloud_config_cross_ref_fails_on_missing_provider() {
7750 // Mirror references a provider id that doesn't exist.
7751 let tmp = tempfile::TempDir::new().unwrap();
7752 let root = tmp.path();
7753 let svc = root.join(".yah").join("services").join("dev-yah");
7754 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
7755 std::fs::write(
7756 svc.join("service.toml"),
7757 "schema_version = 1\nname = \"dev-yah\"\ndomain = \"yah.dev\"\n",
7758 )
7759 .unwrap();
7760 std::fs::write(
7761 svc.join("mirrors/prod.toml"),
7762 "schema_version = 1\nshape = \"single-machine\"\n\n[providers.static]\nuse = \"fly-io\"\n",
7763 ).unwrap();
7764
7765 let err = CloudConfig::load(root).unwrap_err();
7766 let msg = err.to_string();
7767 assert!(
7768 msg.contains("fly-io"),
7769 "error should name the missing provider id, got: {msg}"
7770 );
7771 assert!(
7772 msg.contains("providers/fly-io.toml") || msg.contains("no such provider"),
7773 "error should hint at remedy, got: {msg}"
7774 );
7775 }
7776
7777 #[test]
7778 fn cloud_config_cross_ref_fails_on_missing_provider_named_by_an_ingress_edge() {
7779 // R845: the edge's own `use` is a provider reference like any other, so
7780 // a typo has to fail here rather than at the Cloudflare arm of apply.
7781 let tmp = tempfile::TempDir::new().unwrap();
7782 let root = tmp.path();
7783 let svc = root.join(".yah").join("services").join("dev-yah");
7784 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
7785 std::fs::write(
7786 svc.join("service.toml"),
7787 "schema_version = 1\nname = \"dev-yah\"\ndomain = \"yah.dev\"\n",
7788 )
7789 .unwrap();
7790 std::fs::write(
7791 svc.join("mirrors/prod.toml"),
7792 "schema_version = 1\nshape = \"single-machine\"\n\n\
7793 [providers.compute]\nkind = \"static\"\nmachine = \"borrowed-01\"\n\
7794 zone = \"a.yah.dev\"\nport = 8080\n\n\
7795 [[ingress]]\nprovider = \"cloudflare-tunnel\"\nuse = \"cloudflar\"\n",
7796 )
7797 .unwrap();
7798
7799 let msg = CloudConfig::load(root).unwrap_err().to_string();
7800 assert!(
7801 msg.contains("ingress[0].use") && msg.contains("cloudflar"),
7802 "error should name the edge and the typo'd id, got: {msg}"
7803 );
7804 }
7805
7806 #[test]
7807 fn cloud_config_cross_ref_passes_on_inline_only_mirror() {
7808 // Inline `kind = "local-static"` doesn't require an infra provider.
7809 let tmp = tempfile::TempDir::new().unwrap();
7810 let root = tmp.path();
7811 let svc = root.join(".yah").join("services").join("local-only");
7812 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
7813 std::fs::write(
7814 svc.join("service.toml"),
7815 "schema_version = 1\nname = \"local-only\"\ndomain = \"local.test\"\n",
7816 )
7817 .unwrap();
7818 std::fs::write(
7819 svc.join("mirrors/local.toml"),
7820 "schema_version = 1\nshape = \"local\"\n\n[providers.static]\nkind = \"local-static\"\nport = 8080\n",
7821 ).unwrap();
7822
7823 // Should load fine: no `use=` references, no providers required.
7824 let cfg = CloudConfig::load(root).unwrap();
7825 assert!(cfg.service("local-only").is_some());
7826 }
7827
7828 fn mirror(src: &str) -> MirrorConfig {
7829 toml::from_str(&format!("schema_version = 1\nshape = \"single-machine\"\n{src}"))
7830 .expect("parse mirror")
7831 }
7832
7833 #[test]
7834 fn passway_machines_reads_both_ingress_spellings_the_same_way() {
7835 // The whole reason this is derived in Rust rather than read off a field
7836 // by the UI: these two mirrors say the identical thing, and a consumer
7837 // that reaches for `ingress_machines` sees the second one as empty.
7838 let scalar = mirror("ingress = \"passway\"\ningress_machines = [\"us-east-001\"]\n");
7839 let edges = mirror(
7840 "[[ingress]]\nprovider = \"passway\"\nmachines = [\"us-east-001\"]\n",
7841 );
7842 assert_eq!(scalar.passway_machines(), Some(vec!["us-east-001".into()]));
7843 assert_eq!(scalar.passway_machines(), edges.passway_machines());
7844 }
7845
7846 #[test]
7847 fn passway_machines_skips_a_cloudflare_tunnel_edge() {
7848 // A cloudflared node publishes through Cloudflare's DNS and does not
7849 // serve `GET /domains/{d}/onboarding`, so naming it here would point
7850 // the custom-domain UI at a node that cannot answer.
7851 let cf_only =
7852 mirror("[[ingress]]\nprovider = \"cloudflare-tunnel\"\nmachines = [\"cf-01\"]\n");
7853 assert_eq!(cf_only.passway_machines(), None);
7854
7855 let mixed = mirror(
7856 "[[ingress]]\nprovider = \"cloudflare-tunnel\"\nmachines = [\"cf-01\"]\n\
7857 slots = [\"static\"]\n\n\
7858 [[ingress]]\nprovider = \"passway\"\nmachines = [\"us-east-001\"]\n\
7859 slots = [\"bundle\"]\n",
7860 );
7861 assert_eq!(mixed.passway_machines(), Some(vec!["us-east-001".into()]));
7862 }
7863
7864 #[test]
7865 fn passway_machines_separates_declared_but_unplaced_from_undeclared() {
7866 // Some(vec![]) means "a passway front door exists, but its placement
7867 // falls back to the fronted slot's and is not knowable from the mirror".
7868 // None means there is no passway front door at all. Collapsing the two
7869 // would make a co-located edge indistinguishable from no edge.
7870 assert_eq!(mirror("ingress = \"passway\"\n").passway_machines(), Some(vec![]));
7871 assert_eq!(mirror("").passway_machines(), None);
7872 assert_eq!(mirror("ingress = \"none\"\n").passway_machines(), None);
7873 }
7874
7875 #[test]
7876 fn passway_machines_is_none_for_a_declaration_that_cannot_mean_anything() {
7877 // `ingress_machines` with no `ingress` is an error `ingress_edges` names
7878 // properly; swallowing it to None here is deliberate, because this is
7879 // read while loading every service in the workspace and hard-failing
7880 // would report an unrelated mirror's shape error from the wrong place.
7881 let orphaned = mirror("ingress_machines = [\"us-east-001\"]\n");
7882 assert!(orphaned.ingress_edges().is_err());
7883 assert_eq!(orphaned.passway_machines(), None);
7884 }
7885
7886 #[test]
7887 fn cloud_config_load_derives_passway_machines_only_for_passway_envs() {
7888 let tmp = tempfile::TempDir::new().unwrap();
7889 let root = tmp.path();
7890 let svc = root.join(".yah").join("services").join("dev-yah");
7891 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
7892 std::fs::write(
7893 svc.join("service.toml"),
7894 "schema_version = 1\nname = \"dev-yah\"\ndomain = \"yah.dev\"\n",
7895 )
7896 .unwrap();
7897 std::fs::write(
7898 svc.join("mirrors/cloud.toml"),
7899 "schema_version = 1\nshape = \"single-machine\"\n\
7900 ingress = \"passway\"\ningress_machines = [\"us-east-001\", \"us-west-001\"]\n\n\
7901 [providers.static]\nkind = \"local-static\"\nport = 8080\n",
7902 )
7903 .unwrap();
7904 std::fs::write(
7905 svc.join("mirrors/local.toml"),
7906 "schema_version = 1\nshape = \"local\"\n\n\
7907 [providers.static]\nkind = \"local-static\"\nport = 8080\n",
7908 )
7909 .unwrap();
7910
7911 let cfg = CloudConfig::load(root).unwrap();
7912 let svc = cfg.service("dev-yah").unwrap();
7913 assert_eq!(
7914 svc.passway_machines.get("cloud"),
7915 Some(&vec!["us-east-001".to_string(), "us-west-001".to_string()])
7916 );
7917 assert!(
7918 !svc.passway_machines.contains_key("local"),
7919 "an env with no front door must be absent, not empty: {:?}",
7920 svc.passway_machines
7921 );
7922 }
7923
7924 #[test]
7925 fn cloud_config_load_coexists_legacy_and_new_trees() {
7926 // Both trees present — both fields populated independently.
7927 let tmp = tempfile::TempDir::new().unwrap();
7928 let root = tmp.path();
7929 make_new_tree_with_dev_yah(root);
7930
7931 let cloud_dir = make_legacy_cloud_dir(root);
7932 std::fs::create_dir_all(cloud_dir.join("mirrors")).unwrap();
7933 std::fs::write(
7934 cloud_dir.join("mirrors/noisetable.toml"),
7935 "camp = \"noisetable\"\nregions = [\"pdx\"]\nworkloads = []\n",
7936 )
7937 .unwrap();
7938
7939 let cfg = CloudConfig::load(root).unwrap();
7940 assert_eq!(cfg.providers.len(), 3);
7941 assert!(cfg.service("dev-yah").is_some());
7942 assert_eq!(cfg.legacy_mirrors.len(), 1);
7943 assert!(cfg.legacy_mirror("noisetable").is_some());
7944 }
7945
7946 #[test]
7947 fn web_workload_round_trips() {
7948 // app/yah/web/workload.toml is parsed as a WorkloadSpec via the
7949 // workload-spec crate. The minimum-viable manifest here exercises
7950 // schema_version + kind + build fields.
7951 //
7952 // The on-disk file uses the abbreviated v1 form (kind + build); the
7953 // full WorkloadSpec is verbose, so this test asserts the new
7954 // mesofact-static abbreviated form parses as raw TOML (B3 will plumb
7955 // it through WorkloadSpec proper).
7956 // `routes` above [build] — it is a top-level field, and TOML would
7957 // scope it into that table if written below the header (R658-B1).
7958 let src = r#"
7959schema_version = 1
7960kind = "mesofact-static"
7961
7962routes = "./routes.ts"
7963
7964[build]
7965command = "bun run build"
7966out_dir = "dist"
7967"#;
7968 let v: toml::Value = toml::from_str(src).unwrap();
7969 assert_eq!(
7970 v.get("schema_version").and_then(|x| x.as_integer()),
7971 Some(1)
7972 );
7973 assert_eq!(
7974 v.get("kind").and_then(|x| x.as_str()),
7975 Some("mesofact-static")
7976 );
7977 let build = v
7978 .get("build")
7979 .and_then(|x| x.as_table())
7980 .expect("build table");
7981 assert_eq!(
7982 build.get("command").and_then(|x| x.as_str()),
7983 Some("bun run build")
7984 );
7985 assert_eq!(build.get("out_dir").and_then(|x| x.as_str()), Some("dist"));
7986 }
7987
7988 // ─── Canonical CRUD: ServiceConfig/MirrorConfig save + delete (R323-F1) ──
7989
7990 #[test]
7991 fn service_config_save_creates_canonical_toml_and_round_trips() {
7992 let tmp = tempfile::TempDir::new().unwrap();
7993 let root = tmp.path();
7994
7995 let svc = ServiceConfig {
7996 schema_version: 1,
7997 name: "dev-yah".into(),
7998 domain: "yah.dev".into(),
7999 db: DbCatalog::default(),
8000 components: vec![ServiceComponent {
8001 mount: None,
8002 id: "site".into(),
8003 kind: "mesofact-static".into(),
8004 path: "app/yah/web".into(),
8005 role: "static".into(),
8006 publishes: Some("static".into()),
8007 wave: 0,
8008 git: None,
8009 deploy: Default::default(),
8010 }],
8011 };
8012 svc.save(root).unwrap();
8013
8014 // Landed at the canonical path.
8015 let path = crate::paths::service_toml(root, "dev-yah");
8016 assert!(
8017 path.exists(),
8018 "service.toml should exist at {}",
8019 path.display()
8020 );
8021
8022 // Reloads through the full CloudConfig loader (no mirrors yet).
8023 let cfg = CloudConfig::load(root).unwrap();
8024 let loaded = cfg.service("dev-yah").expect("dev-yah service");
8025 assert_eq!(loaded.service.domain, "yah.dev");
8026 assert_eq!(loaded.service.components.len(), 1);
8027 assert_eq!(
8028 loaded.service.components[0].publishes.as_deref(),
8029 Some("static")
8030 );
8031 assert!(loaded.mirrors.is_empty());
8032 }
8033
8034 #[test]
8035 fn service_config_save_overwrites_in_place() {
8036 let tmp = tempfile::TempDir::new().unwrap();
8037 let root = tmp.path();
8038
8039 let mut svc = ServiceConfig {
8040 schema_version: 1,
8041 name: "dev-yah".into(),
8042 domain: "yah.dev".into(),
8043 components: vec![],
8044 db: DbCatalog::default(),
8045 };
8046 svc.save(root).unwrap();
8047 svc.domain = "yah.example".into();
8048 svc.save(root).unwrap();
8049
8050 let cfg = CloudConfig::load(root).unwrap();
8051 assert_eq!(
8052 cfg.service("dev-yah").unwrap().service.domain,
8053 "yah.example"
8054 );
8055 }
8056
8057 #[test]
8058 fn mirror_config_save_round_trips_reference_and_inline_slots() {
8059 let tmp = tempfile::TempDir::new().unwrap();
8060 let root = tmp.path();
8061
8062 // A service must exist so the loader walks the mirrors/ dir.
8063 ServiceConfig {
8064 schema_version: 1,
8065 name: "dev-yah".into(),
8066 domain: "yah.dev".into(),
8067 components: vec![],
8068 db: DbCatalog::default(),
8069 }
8070 .save(root)
8071 .unwrap();
8072
8073 // The cloudflare provider the reference slot points at must resolve,
8074 // or CloudConfig::load's cross-ref check rejects the tree.
8075 let providers = crate::paths::providers_dir(root);
8076 std::fs::create_dir_all(&providers).unwrap();
8077 std::fs::write(
8078 providers.join("cloudflare.toml"),
8079 "schema_version = 1\nid = \"cloudflare\"\nkind = \"cloudflare\"\n",
8080 )
8081 .unwrap();
8082
8083 let mut providers_map = BTreeMap::new();
8084 providers_map.insert(
8085 "static".to_string(),
8086 MirrorProviderSlot::Reference {
8087 provider_id: "cloudflare".into(),
8088 fields: {
8089 let mut f = BTreeMap::new();
8090 f.insert("bucket".to_string(), toml::Value::String("yah-dev".into()));
8091 f
8092 },
8093 },
8094 );
8095 providers_map.insert(
8096 "compute".to_string(),
8097 MirrorProviderSlot::Inline {
8098 kind: Provider::LocalStatic,
8099 fields: {
8100 let mut f = BTreeMap::new();
8101 f.insert("port".to_string(), toml::Value::Integer(4321));
8102 f
8103 },
8104 },
8105 );
8106 let mirror = MirrorConfig {
8107 schema_version: 1,
8108 shape: MirrorShape::SingleMachine,
8109 providers: providers_map,
8110 ingress: Default::default(),
8111 ingress_machines: Vec::new(),
8112 drivers: Default::default(),
8113 asset_aliases: Default::default(),
8114 };
8115 // Save with canonical name; legacy "prod" is normalised to "cloud" on load.
8116 mirror.save(root, "dev-yah", "cloud").unwrap();
8117
8118 let path = crate::paths::service_mirror_toml(root, "dev-yah", "cloud");
8119 assert!(
8120 path.exists(),
8121 "mirror toml should exist at {}",
8122 path.display()
8123 );
8124
8125 let cfg = CloudConfig::load(root).unwrap();
8126 let loaded = &cfg.service("dev-yah").unwrap().mirrors["cloud"];
8127 assert_eq!(loaded.shape, MirrorShape::SingleMachine);
8128 assert_eq!(loaded.providers["static"].provider_id(), Some("cloudflare"));
8129 assert_eq!(
8130 loaded.providers["compute"].inline_kind(),
8131 Some(Provider::LocalStatic)
8132 );
8133 }
8134
8135 #[test]
8136 fn service_delete_removes_dir_and_mirrors() {
8137 let tmp = tempfile::TempDir::new().unwrap();
8138 let root = tmp.path();
8139
8140 let svc = ServiceConfig {
8141 schema_version: 1,
8142 name: "dev-yah".into(),
8143 domain: "yah.dev".into(),
8144 components: vec![],
8145 db: DbCatalog::default(),
8146 };
8147 svc.save(root).unwrap();
8148 MirrorConfig {
8149 schema_version: 1,
8150 shape: MirrorShape::Local,
8151 providers: BTreeMap::new(),
8152 ingress: Default::default(),
8153 ingress_machines: Vec::new(),
8154 drivers: Default::default(),
8155 asset_aliases: Default::default(),
8156 }
8157 .save(root, "dev-yah", "local")
8158 .unwrap();
8159
8160 assert!(
8161 ServiceConfig::delete(root, "dev-yah").unwrap(),
8162 "first delete reports true"
8163 );
8164 assert!(!crate::paths::service_dir(root, "dev-yah").exists());
8165 // Idempotent: deleting again is a no-op that reports false.
8166 assert!(!ServiceConfig::delete(root, "dev-yah").unwrap());
8167
8168 let cfg = CloudConfig::load(root).unwrap();
8169 assert!(cfg.service("dev-yah").is_none());
8170 }
8171
8172 #[test]
8173 fn mirror_delete_leaves_other_mirrors_and_service_intact() {
8174 let tmp = tempfile::TempDir::new().unwrap();
8175 let root = tmp.path();
8176
8177 ServiceConfig {
8178 schema_version: 1,
8179 name: "dev-yah".into(),
8180 domain: "yah.dev".into(),
8181 components: vec![],
8182 db: DbCatalog::default(),
8183 }
8184 .save(root)
8185 .unwrap();
8186 for env in ["prod", "local"] {
8187 MirrorConfig {
8188 schema_version: 1,
8189 shape: MirrorShape::Local,
8190 providers: BTreeMap::new(),
8191 ingress: Default::default(),
8192 ingress_machines: Vec::new(),
8193 drivers: Default::default(),
8194 asset_aliases: Default::default(),
8195 }
8196 .save(root, "dev-yah", env)
8197 .unwrap();
8198 }
8199
8200 assert!(MirrorConfig::delete(root, "dev-yah", "prod").unwrap());
8201 assert!(!MirrorConfig::delete(root, "dev-yah", "prod").unwrap());
8202
8203 let cfg = CloudConfig::load(root).unwrap();
8204 let svc = cfg
8205 .service("dev-yah")
8206 .expect("service survives mirror delete");
8207 // Legacy file stems are normalised on load: "prod" → "cloud", "local" → "dev".
8208 assert!(!svc.mirrors.contains_key("cloud"));
8209 assert!(svc.mirrors.contains_key("dev"));
8210 }
8211
8212 // ─── DomainConfig (R347-F2) ────────────────────────────────────────────
8213
8214 fn write_marketing_service(root: &Path) {
8215 let svc = ServiceConfig {
8216 schema_version: 1,
8217 name: "yah-marketing".into(),
8218 domain: "yah.dev".into(),
8219 db: DbCatalog::default(),
8220 components: vec![ServiceComponent {
8221 mount: None,
8222 id: "site".into(),
8223 kind: "mesofact-static".into(),
8224 path: "app/yah/web".into(),
8225 role: "static".into(),
8226 publishes: None,
8227 wave: 0,
8228 git: None,
8229 deploy: Default::default(),
8230 }],
8231 };
8232 svc.save(root).unwrap();
8233 }
8234
8235 #[test]
8236 fn round_trip_domain_with_each_route_mode() {
8237 let dom = DomainConfig {
8238 schema_version: 1,
8239 name: "yah-dev".into(),
8240 domain: "yah.dev".into(),
8241 front_door: FrontDoor::Worker,
8242 cdn_bucket: "yah-dev".into(),
8243 worker_bundle_path: Some(".yah/workers/yah-dev/".into()),
8244 routes: vec![
8245 DomainRoute {
8246 headers: Default::default(),
8247 path: "/".into(),
8248 mode: RouteMode::Static {
8249 component: "yah-marketing/site".into(),
8250 },
8251 },
8252 DomainRoute {
8253 headers: Default::default(),
8254 path: "/dashboard/api/*".into(),
8255 mode: RouteMode::Backend {
8256 component: "yah-dashboard/api".into(),
8257 origin: "https://api.dashboard.yah.dev".into(),
8258 },
8259 },
8260 DomainRoute {
8261 headers: Default::default(),
8262 path: "/old".into(),
8263 mode: RouteMode::Redirect {
8264 target: "https://yah.dev/blog".into(),
8265 status: 308,
8266 },
8267 },
8268 ],
8269 };
8270 let s = toml::to_string(&dom).unwrap();
8271 let back: DomainConfig = toml::from_str(&s).unwrap();
8272 assert_eq!(back.name, "yah-dev");
8273 assert_eq!(back.routes.len(), 3);
8274 assert!(matches!(back.routes[0].mode, RouteMode::Static { .. }));
8275 assert!(matches!(back.routes[1].mode, RouteMode::Backend { .. }));
8276 assert!(matches!(back.routes[2].mode, RouteMode::Redirect { .. }));
8277 }
8278
8279 #[test]
8280 fn redirect_status_defaults_to_308() {
8281 let src = r#"
8282schema_version = 1
8283name = "yah-dev"
8284domain = "yah.dev"
8285front_door = "worker"
8286cdn_bucket = "yah-dev"
8287
8288[[routes]]
8289path = "/old"
8290mode = "redirect"
8291target = "https://yah.dev/blog"
8292"#;
8293 let dom: DomainConfig = toml::from_str(src).unwrap();
8294 let RouteMode::Redirect { status, .. } = &dom.routes[0].mode else {
8295 panic!("expected redirect");
8296 };
8297 assert_eq!(*status, 308);
8298 }
8299
8300 #[test]
8301 fn missing_domains_dir_is_empty() {
8302 let tmp = tempfile::TempDir::new().unwrap();
8303 // R844-B7: `.yah/` must exist or this is a wrong-root error rather
8304 // than an empty tree. The absent directory under test is `domains/`.
8305 std::fs::create_dir_all(tmp.path().join(".yah")).unwrap();
8306 let cfg = CloudConfig::load(tmp.path()).unwrap();
8307 assert!(cfg.domains.is_empty());
8308 }
8309
8310 #[test]
8311 fn save_reload_roundtrip() {
8312 let tmp = tempfile::TempDir::new().unwrap();
8313 let root = tmp.path();
8314 write_marketing_service(root);
8315
8316 let dom = DomainConfig {
8317 schema_version: 1,
8318 name: "yah-dev".into(),
8319 domain: "yah.dev".into(),
8320 front_door: FrontDoor::Worker,
8321 cdn_bucket: "yah-dev".into(),
8322 worker_bundle_path: None,
8323 routes: vec![DomainRoute {
8324 headers: Default::default(),
8325 path: "/".into(),
8326 mode: RouteMode::Static {
8327 component: "yah-marketing/site".into(),
8328 },
8329 }],
8330 };
8331 dom.save(root).unwrap();
8332
8333 let cfg = CloudConfig::load(root).unwrap();
8334 let loaded = cfg.domain("yah-dev").expect("yah-dev domain");
8335 assert_eq!(loaded.domain, "yah.dev");
8336 assert_eq!(loaded.routes.len(), 1);
8337 }
8338
8339 #[test]
8340 fn delete_returns_false_when_absent() {
8341 let tmp = tempfile::TempDir::new().unwrap();
8342 assert!(!DomainConfig::delete(tmp.path(), "no-such-domain").unwrap());
8343 }
8344
8345 #[test]
8346 fn delete_returns_true_first_time() {
8347 let tmp = tempfile::TempDir::new().unwrap();
8348 let root = tmp.path();
8349 let dom = DomainConfig {
8350 schema_version: 1,
8351 name: "yah-dev".into(),
8352 domain: "yah.dev".into(),
8353 front_door: FrontDoor::BucketDirect,
8354 cdn_bucket: "yah-dev".into(),
8355 worker_bundle_path: None,
8356 routes: vec![],
8357 };
8358 dom.save(root).unwrap();
8359 assert!(DomainConfig::delete(root, "yah-dev").unwrap());
8360 assert!(!DomainConfig::delete(root, "yah-dev").unwrap());
8361 }
8362
8363 // ---- R594-F12: front-door discriminator ------------------------------
8364
8365 /// Write a raw domain manifest so the tests exercise the deserialize +
8366 /// validate path, not a hand-built struct that skipped serde.
8367 fn write_domain_toml(root: &Path, stem: &str, body: &str) {
8368 let dir = root.join(".yah").join("domains");
8369 std::fs::create_dir_all(&dir).unwrap();
8370 std::fs::write(dir.join(format!("{stem}.toml")), body).unwrap();
8371 }
8372
8373 #[test]
8374 fn front_door_is_required() {
8375 let tmp = tempfile::TempDir::new().unwrap();
8376 let root = tmp.path();
8377 write_marketing_service(root);
8378 write_domain_toml(
8379 root,
8380 "yah-dev",
8381 r#"
8382schema_version = 1
8383name = "yah-dev"
8384domain = "yah.dev"
8385cdn_bucket = "yah-dev"
8386[[routes]]
8387path = "/*"
8388mode = "static"
8389component = "yah-marketing/site"
8390"#,
8391 );
8392 let err = CloudConfig::load(root).unwrap_err().to_string();
8393 // serde's own missing-field message; the point is that omitting the
8394 // discriminator is not a silently-defaulted state.
8395 assert!(err.contains("yah-dev.toml"), "{err}");
8396 }
8397
8398 #[test]
8399 fn bucket_direct_with_routes_is_rejected() {
8400 let tmp = tempfile::TempDir::new().unwrap();
8401 let root = tmp.path();
8402 write_marketing_service(root);
8403 write_domain_toml(
8404 root,
8405 "cdn-yah-dev",
8406 r#"
8407schema_version = 1
8408name = "cdn-yah-dev"
8409domain = "cdn.yah.dev"
8410front_door = "bucket-direct"
8411cdn_bucket = "yah-dev"
8412[[routes]]
8413path = "/docs/*"
8414mode = "static"
8415component = "yah-marketing/site"
8416"#,
8417 );
8418 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8419 assert!(err.contains("front_door"), "{err}");
8420 assert!(err.contains("/docs/*"), "{err}");
8421 }
8422
8423 #[test]
8424 fn bucket_direct_with_worker_bundle_path_is_rejected() {
8425 let tmp = tempfile::TempDir::new().unwrap();
8426 let root = tmp.path();
8427 write_domain_toml(
8428 root,
8429 "cdn-yah-dev",
8430 r#"
8431schema_version = 1
8432name = "cdn-yah-dev"
8433domain = "cdn.yah.dev"
8434front_door = "bucket-direct"
8435cdn_bucket = "yah-dev"
8436worker_bundle_path = ".yah/workers/cdn-yah-dev/"
8437"#,
8438 );
8439 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8440 assert!(err.contains("worker_bundle_path"), "{err}");
8441 }
8442
8443 // ── R746: per-route response headers + component mounts ──────────────────
8444
8445 /// A two-component service: `site` at the root, `app` mounted at `/app`
8446 /// with isolation headers on its route. This is the noisetable.com shape
8447 /// the primitive was built for.
8448 fn write_two_component_service(root: &Path) {
8449 let svc = ServiceConfig {
8450 schema_version: 1,
8451 name: "yah-marketing".into(),
8452 domain: "yah.dev".into(),
8453 db: DbCatalog::default(),
8454 components: vec![
8455 ServiceComponent {
8456 mount: None,
8457 id: "site".into(),
8458 kind: "mesofact-static".into(),
8459 path: "app/yah/web".into(),
8460 role: "static".into(),
8461 publishes: None,
8462 wave: 0,
8463 git: None,
8464 deploy: Default::default(),
8465 },
8466 ServiceComponent {
8467 mount: Some("/app".into()),
8468 id: "app".into(),
8469 kind: "mesofact-static".into(),
8470 path: "app/browser".into(),
8471 role: "static".into(),
8472 publishes: None,
8473 wave: 0,
8474 git: None,
8475 deploy: Default::default(),
8476 },
8477 ],
8478 };
8479 svc.save(root).unwrap();
8480 }
8481
8482 const MOUNTED_DOMAIN: &str = r#"
8483schema_version = 1
8484name = "yah-dev"
8485domain = "yah.dev"
8486front_door = "worker"
8487cdn_bucket = "yah-dev"
8488
8489[[routes]]
8490path = "/app/*"
8491mode = "static"
8492component = "yah-marketing/app"
8493headers = { "Cross-Origin-Opener-Policy" = "same-origin", "Cross-Origin-Embedder-Policy" = "require-corp" }
8494
8495[[routes]]
8496path = "/*"
8497mode = "static"
8498component = "yah-marketing/site"
8499"#;
8500
8501 #[test]
8502 fn a_mounted_component_routed_at_its_mount_loads() {
8503 let tmp = tempfile::TempDir::new().unwrap();
8504 let root = tmp.path();
8505 write_two_component_service(root);
8506 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
8507 let cfg = CloudConfig::load(root).unwrap();
8508 let dom = cfg.domain("yah-dev").unwrap();
8509 assert_eq!(dom.routes.len(), 2);
8510 assert_eq!(
8511 dom.routes[0].headers.get("Cross-Origin-Opener-Policy").map(String::as_str),
8512 Some("same-origin")
8513 );
8514 assert!(dom.routes[1].headers.is_empty());
8515 }
8516
8517 /// The header table reaches the Worker in MANIFEST order with headerless
8518 /// routes dropped. Order is the whole contract — the front door applies the
8519 /// first match, so `/app/*` before `/*` is what isolates the app without
8520 /// isolating the marketing site.
8521 #[test]
8522 fn route_headers_json_preserves_order_and_drops_headerless_routes() {
8523 let tmp = tempfile::TempDir::new().unwrap();
8524 let root = tmp.path();
8525 write_two_component_service(root);
8526 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
8527 let cfg = CloudConfig::load(root).unwrap();
8528 let json = cfg.domain("yah-dev").unwrap().route_headers_json();
8529
8530 let parsed: serde_json::Value = serde_json::from_str(&json).unwrap();
8531 let rules = parsed.as_array().unwrap();
8532 assert_eq!(rules.len(), 1, "the headerless catch-all is dropped: {json}");
8533 assert_eq!(rules[0]["path"], "/app/*");
8534 assert_eq!(rules[0]["headers"]["Cross-Origin-Embedder-Policy"], "require-corp");
8535 }
8536
8537 #[test]
8538 fn route_headers_json_is_an_empty_array_when_nothing_declares_headers() {
8539 let tmp = tempfile::TempDir::new().unwrap();
8540 let root = tmp.path();
8541 write_marketing_service(root);
8542 write_domain_toml(
8543 root,
8544 "yah-dev",
8545 r#"
8546schema_version = 1
8547name = "yah-dev"
8548domain = "yah.dev"
8549front_door = "worker"
8550cdn_bucket = "yah-dev"
8551
8552[[routes]]
8553path = "/*"
8554mode = "static"
8555component = "yah-marketing/site"
8556"#,
8557 );
8558 let cfg = CloudConfig::load(root).unwrap();
8559 assert_eq!(cfg.domain("yah-dev").unwrap().route_headers_json(), "[]");
8560 }
8561
8562 /// The reconciler's own entry point: given a workspace root and a service
8563 /// name, produce the binding value. `"[]"` when nothing routes the service.
8564 #[test]
8565 fn route_headers_for_service_reads_the_workspace_domains() {
8566 let tmp = tempfile::TempDir::new().unwrap();
8567 let root = tmp.path();
8568 write_two_component_service(root);
8569 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
8570 assert!(route_headers_for_service(root, "yah-marketing")
8571 .unwrap()
8572 .contains("require-corp"));
8573 assert_eq!(route_headers_for_service(root, "some-other-svc").unwrap(), "[]");
8574 }
8575
8576 // ---- R749-T5: a broken table fails the DEPLOY, not the edge -----------
8577
8578 /// The manifest's `headers` map is hand-written TOML, so a header name with
8579 /// spaces in it is one keystroke away — and it survives serialization into
8580 /// a structurally-valid table that neither front door can apply. Fail at
8581 /// load, naming the domain, the route and the header, instead of shipping a
8582 /// binding the Worker throws on and an origin that refuses to boot.
8583 #[test]
8584 fn a_route_header_name_that_is_not_a_header_name_fails_the_load() {
8585 let tmp = tempfile::TempDir::new().unwrap();
8586 let root = tmp.path();
8587 write_marketing_service(root);
8588 write_domain_toml(
8589 root,
8590 "yah-dev",
8591 r#"
8592schema_version = 1
8593name = "yah-dev"
8594domain = "yah.dev"
8595front_door = "worker"
8596cdn_bucket = "yah-dev"
8597
8598[[routes]]
8599path = "/*"
8600mode = "static"
8601component = "yah-marketing/site"
8602headers = { "Cross Origin Opener Policy" = "same-origin" }
8603"#,
8604 );
8605 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8606 assert!(err.contains("yah-dev"), "{err}");
8607 assert!(err.contains("/*"), "{err}");
8608 assert!(err.contains("Cross Origin Opener Policy"), "{err}");
8609 assert!(err.contains("not a valid HTTP header name"), "{err}");
8610 }
8611
8612 /// A newline in a value is header injection if it ever reached the wire, so
8613 /// both doors reject it and so does this.
8614 #[test]
8615 fn a_route_header_value_that_is_not_a_header_value_fails_the_load() {
8616 let tmp = tempfile::TempDir::new().unwrap();
8617 let root = tmp.path();
8618 write_marketing_service(root);
8619 write_domain_toml(
8620 root,
8621 "yah-dev",
8622 r#"
8623schema_version = 1
8624name = "yah-dev"
8625domain = "yah.dev"
8626front_door = "worker"
8627cdn_bucket = "yah-dev"
8628
8629[[routes]]
8630path = "/*"
8631mode = "static"
8632component = "yah-marketing/site"
8633headers = { "X-Frame-Options" = "DENY\nSet-Cookie: pwned=1" }
8634"#,
8635 );
8636 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8637 assert!(err.contains("X-Frame-Options"), "{err}");
8638 assert!(err.contains("not a valid HTTP header value"), "{err}");
8639 }
8640
8641 /// The invariant this gate exists to hold: everything `route_headers_json`
8642 /// emits is applicable. A headerless route contributes no rule, so its path
8643 /// is not the table's business — only rules that ship are checked.
8644 #[test]
8645 fn a_headerless_route_is_not_subject_to_the_route_header_gate() {
8646 let tmp = tempfile::TempDir::new().unwrap();
8647 let root = tmp.path();
8648 write_two_component_service(root);
8649 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
8650 let cfg = CloudConfig::load(root).unwrap();
8651 cfg.domain("yah-dev")
8652 .unwrap()
8653 .validate_route_headers()
8654 .unwrap();
8655 }
8656
8657 /// A `bucket-direct` domain has no front door to set headers on, so it must
8658 /// not be picked up as a service's header source.
8659 #[test]
8660 fn route_headers_ignores_domains_that_are_not_route_driven() {
8661 let doms: BTreeMap<String, DomainConfig> = [(
8662 "cdn".to_string(),
8663 DomainConfig {
8664 schema_version: 1,
8665 name: "cdn".into(),
8666 domain: "cdn.yah.dev".into(),
8667 front_door: FrontDoor::BucketDirect,
8668 cdn_bucket: "yah-dev".into(),
8669 worker_bundle_path: None,
8670 routes: vec![],
8671 },
8672 )]
8673 .into_iter()
8674 .collect();
8675 assert!(domain_serving_service(&doms, "yah-marketing").is_none());
8676 }
8677
8678 #[test]
8679 fn a_mount_that_disagrees_with_its_route_path_is_rejected() {
8680 let tmp = tempfile::TempDir::new().unwrap();
8681 let root = tmp.path();
8682 write_two_component_service(root);
8683 write_domain_toml(
8684 root,
8685 "yah-dev",
8686 r#"
8687schema_version = 1
8688name = "yah-dev"
8689domain = "yah.dev"
8690front_door = "worker"
8691cdn_bucket = "yah-dev"
8692
8693[[routes]]
8694path = "/studio/*"
8695mode = "static"
8696component = "yah-marketing/app"
8697"#,
8698 );
8699 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8700 assert!(err.contains("mount = \"/app\""), "{err}");
8701 assert!(err.contains("/studio/*"), "{err}");
8702 }
8703
8704 /// The other direction: routing an unmounted component under a sub-path
8705 /// points requests at a prefix nothing published to.
8706 #[test]
8707 fn routing_an_unmounted_component_under_a_subpath_is_rejected() {
8708 let tmp = tempfile::TempDir::new().unwrap();
8709 let root = tmp.path();
8710 write_marketing_service(root);
8711 write_domain_toml(
8712 root,
8713 "yah-dev",
8714 r#"
8715schema_version = 1
8716name = "yah-dev"
8717domain = "yah.dev"
8718front_door = "worker"
8719cdn_bucket = "yah-dev"
8720
8721[[routes]]
8722path = "/docs/*"
8723mode = "static"
8724component = "yah-marketing/site"
8725"#,
8726 );
8727 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8728 assert!(err.contains("no `mount`"), "{err}");
8729 assert!(err.contains("/docs"), "{err}");
8730 }
8731
8732 #[test]
8733 fn mount_and_route_prefix_normalization_agree() {
8734 for m in ["/app", "app", "app/", "/app/"] {
8735 assert_eq!(normalize_mount(m), "app", "mount {m:?}");
8736 }
8737 assert_eq!(normalize_mount("/"), "");
8738 assert_eq!(route_path_prefix("/*"), "");
8739 assert_eq!(route_path_prefix("/app/*"), "app");
8740 assert_eq!(route_path_prefix("/app"), "app");
8741 assert_eq!(route_path_prefix("/"), "");
8742 }
8743
8744 // ── R870-B11: a mount is owned by exactly one bundle-tier component ────
8745
8746 /// Two bundle-tier components at the same explicit mount would stage into
8747 /// the same `app/dist/<mount>/` prefix inside one assembled bundle and
8748 /// silently clobber each other — reject at load, before that happens.
8749 #[test]
8750 fn two_bundle_components_at_the_same_mount_are_rejected() {
8751 let tmp = tempfile::TempDir::new().unwrap();
8752 let root = tmp.path();
8753 let svc = ServiceConfig {
8754 schema_version: 1,
8755 name: "noisetable-marketing".into(),
8756 domain: "noisetable.com".into(),
8757 db: DbCatalog::default(),
8758 components: vec![
8759 ServiceComponent {
8760 mount: Some("/app".into()),
8761 id: "app".into(),
8762 kind: "mesofact-static".into(),
8763 path: "app/browser".into(),
8764 role: "static".into(),
8765 publishes: None,
8766 wave: 0,
8767 git: None,
8768 deploy: Default::default(),
8769 },
8770 ServiceComponent {
8771 mount: Some("app/".into()),
8772 id: "app2".into(),
8773 kind: "mesofact-spa".into(),
8774 path: "app/other".into(),
8775 role: "static".into(),
8776 publishes: None,
8777 wave: 0,
8778 git: None,
8779 deploy: Default::default(),
8780 },
8781 ],
8782 };
8783 svc.save(root).unwrap();
8784 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8785 assert!(err.contains("\"app\""), "{err}");
8786 assert!(err.contains("\"app2\""), "{err}");
8787 assert!(err.contains("mount = \"/app\""), "{err}");
8788 }
8789
8790 /// The unmounted case: two bundle-tier components both leaving `mount`
8791 /// unset both claim the service root, which collides exactly the same
8792 /// way — this is the noisetable shape the ticket was filed against, if
8793 /// `app`'s mount had been forgotten instead of declared.
8794 #[test]
8795 fn two_bundle_components_with_no_mount_are_rejected() {
8796 let tmp = tempfile::TempDir::new().unwrap();
8797 let root = tmp.path();
8798 let svc = ServiceConfig {
8799 schema_version: 1,
8800 name: "noisetable-marketing".into(),
8801 domain: "noisetable.com".into(),
8802 db: DbCatalog::default(),
8803 components: vec![
8804 ServiceComponent {
8805 mount: None,
8806 id: "site".into(),
8807 kind: "mesofact-spa".into(),
8808 path: "web/landing".into(),
8809 role: "static".into(),
8810 publishes: None,
8811 wave: 0,
8812 git: None,
8813 deploy: Default::default(),
8814 },
8815 ServiceComponent {
8816 mount: None,
8817 id: "app".into(),
8818 kind: "mesofact-static".into(),
8819 path: "app/browser".into(),
8820 role: "static".into(),
8821 publishes: None,
8822 wave: 0,
8823 git: None,
8824 deploy: Default::default(),
8825 },
8826 ],
8827 };
8828 svc.save(root).unwrap();
8829 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8830 assert!(err.contains("\"site\""), "{err}");
8831 assert!(err.contains("\"app\""), "{err}");
8832 assert!(err.contains("the service root"), "{err}");
8833 }
8834
8835 /// R870-F23 widened the same loop to the workload tier. Two components in
8836 /// DIFFERENT tiers at one mount is the same clobber read from the routing
8837 /// side: the inner door's table names one upstream for that prefix, so a
8838 /// request reaches either the bundle or the workload and nothing says
8839 /// which.
8840 #[test]
8841 fn a_bundle_component_and_a_workload_component_at_one_mount_are_rejected() {
8842 let tmp = tempfile::TempDir::new().unwrap();
8843 let root = tmp.path();
8844 let svc = ServiceConfig {
8845 schema_version: 1,
8846 name: "noisetable".into(),
8847 domain: "noisetable.com".into(),
8848 db: DbCatalog::default(),
8849 components: vec![
8850 ServiceComponent {
8851 mount: Some("app".into()),
8852 id: "app-bundle".into(),
8853 kind: "mesofact-spa".into(),
8854 path: "app/browser".into(),
8855 role: "static".into(),
8856 publishes: None,
8857 wave: 0,
8858 git: None,
8859 deploy: DeployTier::Bundle,
8860 },
8861 ServiceComponent {
8862 mount: Some("/app/".into()),
8863 id: "app-service".into(),
8864 kind: "container".into(),
8865 path: "app/server".into(),
8866 role: "compute".into(),
8867 publishes: None,
8868 wave: 0,
8869 git: None,
8870 deploy: DeployTier::Workload,
8871 },
8872 ],
8873 };
8874 svc.save(root).unwrap();
8875 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8876 assert!(err.contains("\"app-bundle\""), "{err}");
8877 assert!(err.contains("\"app-service\""), "{err}");
8878 assert!(err.contains("deploys as its own workload"), "{err}");
8879 }
8880
8881 /// And two WORKLOAD-tier components at one mount, which the pre-R870-F23
8882 /// loop skipped entirely (it filtered on `kind`, and a workload-tier
8883 /// component need not be a mesofact kind at all).
8884 #[test]
8885 fn two_workload_components_at_one_mount_are_rejected() {
8886 let tmp = tempfile::TempDir::new().unwrap();
8887 let root = tmp.path();
8888 let svc = ServiceConfig {
8889 schema_version: 1,
8890 name: "noisetable".into(),
8891 domain: "noisetable.com".into(),
8892 db: DbCatalog::default(),
8893 components: vec![
8894 ServiceComponent {
8895 mount: None,
8896 id: "api".into(),
8897 kind: "container".into(),
8898 path: "svc/api".into(),
8899 role: "compute".into(),
8900 publishes: None,
8901 wave: 0,
8902 git: None,
8903 deploy: DeployTier::Workload,
8904 },
8905 ServiceComponent {
8906 mount: Some("/".into()),
8907 id: "api2".into(),
8908 kind: "container".into(),
8909 path: "svc/api2".into(),
8910 role: "compute".into(),
8911 publishes: None,
8912 wave: 0,
8913 git: None,
8914 deploy: DeployTier::Workload,
8915 },
8916 ],
8917 };
8918 svc.save(root).unwrap();
8919 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8920 assert!(err.contains("inner-door route table"), "{err}");
8921 assert!(err.contains("the service root"), "{err}");
8922 }
8923
8924 /// The legitimate shape (distinct mounts) is untouched — regression guard
8925 /// so the new check does not become the next silent-overwrite bug.
8926 #[test]
8927 fn bundle_components_at_distinct_mounts_still_load() {
8928 let tmp = tempfile::TempDir::new().unwrap();
8929 let root = tmp.path();
8930 write_two_component_service(root);
8931 assert!(CloudConfig::load(root).is_ok());
8932 }
8933
8934 #[test]
8935 fn worker_with_no_routes_is_rejected() {
8936 let tmp = tempfile::TempDir::new().unwrap();
8937 let root = tmp.path();
8938 write_domain_toml(
8939 root,
8940 "yah-dev",
8941 r#"
8942schema_version = 1
8943name = "yah-dev"
8944domain = "yah.dev"
8945front_door = "worker"
8946cdn_bucket = "yah-dev"
8947"#,
8948 );
8949 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8950 assert!(err.contains("front_door = \"worker\""), "{err}");
8951 assert!(err.contains("404"), "{err}");
8952 }
8953
8954 #[test]
8955 fn passway_with_no_routes_is_rejected_too() {
8956 let tmp = tempfile::TempDir::new().unwrap();
8957 let root = tmp.path();
8958 write_domain_toml(
8959 root,
8960 "yah-dev",
8961 r#"
8962schema_version = 1
8963name = "yah-dev"
8964domain = "yah.dev"
8965front_door = "passway"
8966cdn_bucket = "yah-dev"
8967"#,
8968 );
8969 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8970 assert!(err.contains("front_door = \"passway\""), "{err}");
8971 }
8972
8973 #[test]
8974 fn bucket_direct_without_routes_loads() {
8975 let tmp = tempfile::TempDir::new().unwrap();
8976 let root = tmp.path();
8977 // Exactly the shape .yah/domains/cdn-yah-dev.toml ships (W175: a pure
8978 // asset tier deliberately has no Worker behaviours).
8979 write_domain_toml(
8980 root,
8981 "cdn-yah-dev",
8982 r#"
8983schema_version = 1
8984name = "cdn-yah-dev"
8985domain = "cdn.yah.dev"
8986front_door = "bucket-direct"
8987cdn_bucket = "yah-dev"
8988"#,
8989 );
8990 let cfg = CloudConfig::load(root).unwrap();
8991 let dom = cfg.domain("cdn-yah-dev").expect("cdn-yah-dev domain");
8992 assert_eq!(dom.front_door, FrontDoor::BucketDirect);
8993 assert!(!dom.front_door.is_route_driven());
8994 }
8995
8996 #[test]
8997 fn front_door_round_trips_through_save() {
8998 let tmp = tempfile::TempDir::new().unwrap();
8999 let root = tmp.path();
9000 write_marketing_service(root);
9001 let dom = DomainConfig {
9002 schema_version: 1,
9003 name: "yah-dev".into(),
9004 domain: "yah.dev".into(),
9005 front_door: FrontDoor::Passway,
9006 cdn_bucket: "yah-dev".into(),
9007 worker_bundle_path: None,
9008 routes: vec![DomainRoute {
9009 headers: Default::default(),
9010 path: "/*".into(),
9011 mode: RouteMode::Static {
9012 component: "yah-marketing/site".into(),
9013 },
9014 }],
9015 };
9016 dom.save(root).unwrap();
9017 let cfg = CloudConfig::load(root).unwrap();
9018 assert_eq!(
9019 cfg.domain("yah-dev").unwrap().front_door,
9020 FrontDoor::Passway
9021 );
9022 }
9023
9024 // The four manifests this repo actually ships are asserted in
9025 // `tests/live_workspace_smoke.rs` — that's the only place with a
9026 // depth-agnostic path to the live `.yah/` tree and a skip path for the
9027 // standalone mirror checkout.
9028
9029 #[test]
9030 fn cross_ref_bails_on_missing_service() {
9031 let tmp = tempfile::TempDir::new().unwrap();
9032 let root = tmp.path();
9033 // No services declared at all — component ref must fail to resolve.
9034 let dom = DomainConfig {
9035 schema_version: 1,
9036 name: "yah-dev".into(),
9037 domain: "yah.dev".into(),
9038 front_door: FrontDoor::Worker,
9039 cdn_bucket: "yah-dev".into(),
9040 worker_bundle_path: None,
9041 routes: vec![DomainRoute {
9042 headers: Default::default(),
9043 path: "/".into(),
9044 mode: RouteMode::Static {
9045 component: "yah-marketing/site".into(),
9046 },
9047 }],
9048 };
9049 dom.save(root).unwrap();
9050
9051 let err = CloudConfig::load(root).unwrap_err();
9052 let msg = format!("{err:#}");
9053 assert!(msg.contains("no such service"), "got: {msg}");
9054 assert!(msg.contains("yah-marketing"), "got: {msg}");
9055 }
9056
9057 #[test]
9058 fn cross_ref_bails_on_missing_component() {
9059 let tmp = tempfile::TempDir::new().unwrap();
9060 let root = tmp.path();
9061 write_marketing_service(root); // has component id "site", not "elsewhere"
9062
9063 let dom = DomainConfig {
9064 schema_version: 1,
9065 name: "yah-dev".into(),
9066 domain: "yah.dev".into(),
9067 front_door: FrontDoor::Worker,
9068 cdn_bucket: "yah-dev".into(),
9069 worker_bundle_path: None,
9070 routes: vec![DomainRoute {
9071 headers: Default::default(),
9072 path: "/".into(),
9073 mode: RouteMode::Static {
9074 component: "yah-marketing/elsewhere".into(),
9075 },
9076 }],
9077 };
9078 dom.save(root).unwrap();
9079
9080 let err = CloudConfig::load(root).unwrap_err();
9081 let msg = format!("{err:#}");
9082 assert!(msg.contains("no component with id"), "got: {msg}");
9083 assert!(msg.contains("elsewhere"), "got: {msg}");
9084 }
9085
9086 #[test]
9087 fn cross_ref_bails_on_malformed_ref() {
9088 let tmp = tempfile::TempDir::new().unwrap();
9089 let root = tmp.path();
9090 write_marketing_service(root);
9091
9092 let dom = DomainConfig {
9093 schema_version: 1,
9094 name: "yah-dev".into(),
9095 domain: "yah.dev".into(),
9096 front_door: FrontDoor::Worker,
9097 cdn_bucket: "yah-dev".into(),
9098 worker_bundle_path: None,
9099 routes: vec![DomainRoute {
9100 headers: Default::default(),
9101 path: "/".into(),
9102 mode: RouteMode::Static {
9103 component: "no-slash-here".into(),
9104 },
9105 }],
9106 };
9107 dom.save(root).unwrap();
9108
9109 let err = CloudConfig::load(root).unwrap_err();
9110 let msg = format!("{err:#}");
9111 assert!(msg.contains("expected"), "got: {msg}");
9112 }
9113
9114 #[test]
9115 fn redirect_routes_skip_component_validation() {
9116 let tmp = tempfile::TempDir::new().unwrap();
9117 let root = tmp.path();
9118 // No services at all — redirect must still load cleanly because it
9119 // references nothing.
9120 let dom = DomainConfig {
9121 schema_version: 1,
9122 name: "yah-dev".into(),
9123 domain: "yah.dev".into(),
9124 front_door: FrontDoor::Worker,
9125 cdn_bucket: "yah-dev".into(),
9126 worker_bundle_path: None,
9127 routes: vec![DomainRoute {
9128 headers: Default::default(),
9129 path: "/old".into(),
9130 mode: RouteMode::Redirect {
9131 target: "https://yah.dev/blog".into(),
9132 status: 308,
9133 },
9134 }],
9135 };
9136 dom.save(root).unwrap();
9137
9138 let cfg = CloudConfig::load(root).unwrap();
9139 assert!(cfg.domain("yah-dev").is_some());
9140 }
9141
9142 #[test]
9143 fn name_must_match_file_stem() {
9144 let tmp = tempfile::TempDir::new().unwrap();
9145 let root = tmp.path();
9146 // Hand-write a file whose stem disagrees with its `name`.
9147 let dir = root.join(".yah").join("domains");
9148 std::fs::create_dir_all(&dir).unwrap();
9149 std::fs::write(
9150 dir.join("yah-dev.toml"),
9151 r#"schema_version = 1
9152name = "different-name"
9153domain = "yah.dev"
9154front_door = "bucket-direct"
9155cdn_bucket = "yah-dev"
9156"#,
9157 )
9158 .unwrap();
9159
9160 let err = CloudConfig::load(root).unwrap_err();
9161 let msg = format!("{err:#}");
9162 assert!(msg.contains("must match the file stem"), "got: {msg}");
9163 }
9164
9165 #[test]
9166 fn net_alias_tier_subdomain_manifest_loads_and_cross_refs() {
9167 // R561-F2: a per-tenant subdomain manifest on the net.yah.dev wildcard
9168 // alias tier is just a DomainConfig whose `domain` is `<name>.net.yah.dev`
9169 // and whose static route cross-refs the tenant's service component.
9170 // This is exactly the shape .yah/domains/scrabcake-net-yah-dev.toml ships.
9171 let tmp = tempfile::TempDir::new().unwrap();
9172 let root = tmp.path();
9173 write_marketing_service(root); // service "yah-marketing", component "site"
9174
9175 let dom = DomainConfig {
9176 schema_version: 1,
9177 name: "tenant-net-yah-dev".into(),
9178 domain: "tenant.net.yah.dev".into(),
9179 front_door: FrontDoor::Worker,
9180 cdn_bucket: "net-yah-dev".into(), // shared per-tier bucket
9181 worker_bundle_path: None,
9182 routes: vec![DomainRoute {
9183 headers: Default::default(),
9184 path: "/*".into(),
9185 mode: RouteMode::Static {
9186 component: "yah-marketing/site".into(),
9187 },
9188 }],
9189 };
9190 dom.save(root).unwrap();
9191
9192 let cfg = CloudConfig::load(root).unwrap();
9193 let dom = cfg
9194 .domain("tenant-net-yah-dev")
9195 .expect("net-tier subdomain manifest should load");
9196 assert_eq!(dom.domain, "tenant.net.yah.dev");
9197 assert_eq!(dom.cdn_bucket, "net-yah-dev");
9198 }
9199
9200 // ─── R572-F3: NodeAllocatable + taints ──────────────────────────────────
9201
9202 #[test]
9203 fn machine_allocatable_round_trips() {
9204 let toml_src = r#"
9205name = "us-west-001"
9206provider = "static"
9207mesh_tags = ["tag:cloud-runner"]
9208[allocatable]
9209memory_mb = 3800
9210cpu_millis = 2000
9211"#;
9212 let m: MachineConfig = toml::from_str(toml_src).unwrap();
9213 let a = m.allocatable.as_ref().expect("allocatable should parse");
9214 assert_eq!(a.memory_mb, 3800);
9215 assert_eq!(a.cpu_millis, 2000);
9216
9217 let s = toml::to_string(&m).unwrap();
9218 let back: MachineConfig = toml::from_str(&s).unwrap();
9219 let a2 = back.allocatable.as_ref().unwrap();
9220 assert_eq!(a2.memory_mb, 3800);
9221 assert_eq!(a2.cpu_millis, 2000);
9222 }
9223
9224 #[test]
9225 fn machine_taints_round_trips() {
9226 let toml_src = r#"
9227name = "us-south-001"
9228provider = "static"
9229mesh_tags = ["tag:cloud-runner"]
9230taints = ["no-appliance"]
9231"#;
9232 let m: MachineConfig = toml::from_str(toml_src).unwrap();
9233 assert_eq!(m.taints, vec!["no-appliance"]);
9234
9235 let s = toml::to_string(&m).unwrap();
9236 let back: MachineConfig = toml::from_str(&s).unwrap();
9237 assert_eq!(back.taints, vec!["no-appliance"]);
9238 }
9239
9240 #[test]
9241 fn machine_allocatable_absent_is_none() {
9242 let toml_src = "name = \"node\"\nprovider = \"static\"\nmesh_tags = []\n";
9243 let m: MachineConfig = toml::from_str(toml_src).unwrap();
9244 assert!(m.allocatable.is_none());
9245 assert!(m.taints.is_empty());
9246 }
9247
9248 #[test]
9249 fn machine_allocatable_skipped_when_none() {
9250 let m = make_machine("node", vec![]);
9251 let s = toml::to_string(&m).unwrap();
9252 assert!(
9253 !s.contains("allocatable"),
9254 "None allocatable must be omitted: {s}"
9255 );
9256 assert!(!s.contains("taints"), "empty taints must be omitted: {s}");
9257 }
9258
9259 #[test]
9260 fn machine_multiple_taints_round_trip() {
9261 let toml_src = r#"
9262name = "quarantined"
9263provider = "static"
9264mesh_tags = ["tag:build-worker"]
9265taints = ["no-server", "no-appliance", "no-job"]
9266"#;
9267 let m: MachineConfig = toml::from_str(toml_src).unwrap();
9268 assert_eq!(m.taints.len(), 3);
9269 assert!(m.taints.contains(&"no-server".to_string()));
9270 assert!(m.taints.contains(&"no-appliance".to_string()));
9271 assert!(m.taints.contains(&"no-job".to_string()));
9272 // R742-T4: every key here is one the scheduler reads. This fixture
9273 // used to carry `no-voter`, which none of them is.
9274 assert!(m.inert_taints().is_empty());
9275 }
9276
9277 // ─── R742-T4 (W305): inert-taint classification ─────────────────────────
9278
9279 #[test]
9280 fn every_archetype_repel_key_is_live() {
9281 for arch in LifecycleArchetype::ALL {
9282 let key = format!("no-{}", arch.taint_key());
9283 assert_eq!(
9284 taint_effect(&key),
9285 TaintEffect::Repels(arch),
9286 "{key} must repel {arch:?}"
9287 );
9288 }
9289 }
9290
9291 #[test]
9292 fn public_ip_is_an_affinity_key_not_an_inert_one() {
9293 assert_eq!(
9294 taint_effect(workload_spec::PUBLIC_IP_TAINT),
9295 TaintEffect::Attracts
9296 );
9297 }
9298
9299 #[test]
9300 fn a_free_form_taint_is_inert_and_says_so() {
9301 // W305's headline example: `taints = ["qa"]` parsed clean and did
9302 // nothing. Environment is not expressible as a taint.
9303 assert_eq!(taint_effect("qa"), TaintEffect::Inert);
9304 // And the one that actually cost fleet state: `no-voter` reads as an
9305 // exclusion and excludes nothing — "voter" is not an archetype.
9306 assert_eq!(taint_effect("no-voter"), TaintEffect::Inert);
9307 // A near-miss on a real key is inert too, not silently forgiven.
9308 assert_eq!(taint_effect("no-servers"), TaintEffect::Inert);
9309
9310 let m = make_machine_with_capacity(
9311 "dev-pi",
9312 8192,
9313 4000,
9314 vec!["no-appliance", "no-voter", "qa"],
9315 );
9316 assert_eq!(m.inert_taints(), vec!["no-voter", "qa"]);
9317 }
9318
9319 // ─── R876-B7: repel-by-default + declarable toleration ──────────────────
9320
9321 /// The headline inversion. A bare `RequiredSpec` — which is exactly what
9322 /// deserializing a mirror's `required = { regions, mesh_tags }` produces,
9323 /// since no TOML in the tree writes `tolerates` — is now repelled by a
9324 /// repelling taint. Before B7 it matched, because repulsion was conditional
9325 /// on a `#[serde(skip)]` field that this path could never fill.
9326 #[test]
9327 fn an_undeclared_spec_is_repelled_by_a_repelling_taint() {
9328 let tainted = make_machine_with_capacity("n", 8192, 4000, vec!["no-server"]);
9329 assert!(!RequiredSpec::default().matches(&tainted));
9330
9331 // And it is the DESERIALIZED shape that matters, not a hand-built one:
9332 // this is the mirror path reproduced exactly.
9333 let from_toml: RequiredSpec =
9334 toml::from_str("regions = [\"us-east\"]\n").expect("a mirror-shaped required parses");
9335 assert!(from_toml.tolerates.is_empty());
9336 let mut in_region = tainted.clone();
9337 in_region.region = Some("us-east".to_string());
9338 assert!(
9339 !from_toml.matches(&in_region),
9340 "a mirror-declared placement must now read machine.taints"
9341 );
9342 }
9343
9344 /// The opt-back-in half, and the one an operator writes by hand.
9345 #[test]
9346 fn an_explicit_toleration_admits_the_tainted_machine_again() {
9347 let tainted = make_machine_with_capacity("n", 8192, 4000, vec!["no-server"]);
9348 let spec = RequiredSpec {
9349 tolerates: vec!["no-server".to_string()],
9350 ..Default::default()
9351 };
9352 assert!(spec.matches(&tainted));
9353
9354 // Per-key, not a blanket pass: tolerating one repelling key says nothing
9355 // about another.
9356 let both = make_machine_with_capacity("n", 8192, 4000, vec!["no-server", "no-appliance"]);
9357 assert!(!spec.matches(&both));
9358
9359 // And it deserializes — the whole point of replacing a `#[serde(skip)]`
9360 // field is that a mirror can now declare this.
9361 let from_toml: RequiredSpec = toml::from_str("tolerates = [\"no-server\"]\n")
9362 .expect("a slot can declare a toleration");
9363 assert!(from_toml.matches(&tainted));
9364 }
9365
9366 /// An untainted machine is unaffected, which is what makes the migration
9367 /// bounded: six of the nine fleet machines carry no repelling taint at all.
9368 #[test]
9369 fn an_untainted_machine_matches_exactly_as_before() {
9370 let clean = make_machine_with_capacity("n", 8192, 4000, vec![]);
9371 assert!(RequiredSpec::default().matches(&clean));
9372 assert!(RequiredSpec {
9373 tolerates: vec!["no-server".to_string()],
9374 ..Default::default()
9375 }
9376 .matches(&clean));
9377 }
9378
9379 /// THE MIGRATION'S LOAD-BEARING FACT. `public-ip` is on three fleet nodes
9380 /// including us-east-001, the only origin serving the yah.dev apex. It is an
9381 /// *affinity* key, so repel-by-default must not touch it — reading every
9382 /// taint as repulsion would evict the apex on the next apply.
9383 #[test]
9384 fn an_affinity_taint_does_not_repel() {
9385 let public = make_machine_with_capacity("us-east-001", 8192, 4000, vec!["public-ip"]);
9386 assert!(
9387 RequiredSpec::default().matches(&public),
9388 "public-ip attracts; it must never be read as repulsion"
9389 );
9390 }
9391
9392 /// `select_matching` filters on the same predicate, so a tainted machine
9393 /// leaves the candidate set rather than being silently placed onto.
9394 #[test]
9395 fn select_matching_drops_a_tainted_candidate_and_keeps_the_rest() {
9396 let drained = make_machine_with_capacity("drained", 8192, 4000, vec!["no-server"]);
9397 let healthy = make_machine_with_capacity("healthy", 8192, 4000, vec![]);
9398 let pool = [&drained, &healthy];
9399
9400 let picked = select_matching(&pool, &RequiredSpec::default(), 1, "test pool", "empty")
9401 .expect("one candidate remains");
9402 assert_eq!(
9403 picked.iter().map(|m| m.name.as_str()).collect::<Vec<_>>(),
9404 vec!["healthy"],
9405 "tainting the first candidate moves the placement to the second"
9406 );
9407
9408 // Asking for both is a shortfall, not a half-placement.
9409 let err = select_matching(&pool, &RequiredSpec::default(), 2, "test pool", "empty")
9410 .expect_err("only one of two matches");
9411 assert!(format!("{err:#}").contains("only 1 of 2 machines match"));
9412 }
9413
9414 /// The `admit_workload` path must be behaviourally unchanged: its spec is
9415 /// built by `admission_spec`, which now emits the complementary tolerations.
9416 #[test]
9417 fn admission_preserves_archetype_scoped_repulsion_across_the_inversion() {
9418 let ws = minimal_spec("srv", 1); // a Server
9419 let req = admission_spec(&ws, &[]);
9420 assert_eq!(ws.effective_archetype(), LifecycleArchetype::Server);
9421
9422 let no_server = make_machine_with_capacity("n", 8192, 4000, vec!["no-server"]);
9423 let no_appliance = make_machine_with_capacity("n", 8192, 4000, vec!["no-appliance"]);
9424 assert!(!req.matches(&no_server), "its own class still repels it");
9425 assert!(
9426 req.matches(&no_appliance),
9427 "another class's taint still does not — this is the pre-B7 answer"
9428 );
9429 }
9430
9431 #[test]
9432 fn describe_names_the_toleration_so_a_refusal_is_readable() {
9433 let spec = RequiredSpec {
9434 regions: vec!["us-west".to_string()],
9435 tolerates: vec!["no-appliance".to_string()],
9436 ..Default::default()
9437 };
9438 assert_eq!(
9439 spec.describe(),
9440 "required.regions=[us-west] + required.tolerates=[no-appliance]"
9441 );
9442 }
9443
9444 /// A toleration widens; it must not make an underspecified slot look
9445 /// specified, or the deploy side stops refusing one.
9446 #[test]
9447 fn a_toleration_alone_is_still_an_unconstrained_spec() {
9448 assert!(RequiredSpec {
9449 tolerates: vec!["no-server".to_string()],
9450 ..Default::default()
9451 }
9452 .is_unconstrained());
9453 }
9454
9455 #[test]
9456 fn an_inert_taint_changes_no_placement_decision() {
9457 // The reason this is a lint and not a behaviour change: the guard's
9458 // whole premise is that these keys are invisible to `matches`.
9459 let clean = make_machine_with_capacity("n", 8192, 4000, vec![]);
9460 let noisy = make_machine_with_capacity("n", 8192, 4000, vec!["no-voter", "qa"]);
9461 // R876-B7: still true under repel-by-default, and for a sharper reason —
9462 // `matches` now walks `machine.taints` itself, so an unclassifiable key
9463 // is skipped by `taint_effect` rather than merely never looked up.
9464 for arch in LifecycleArchetype::ALL {
9465 let req = RequiredSpec {
9466 tolerates: tolerations_excluding(&[arch]),
9467 ..Default::default()
9468 };
9469 assert_eq!(req.matches(&clean), req.matches(&noisy));
9470 assert!(req.matches(&noisy), "neither key repels");
9471 }
9472 }
9473
9474 /// The toleration set [`admission_spec`] derives for a group of `archetypes`
9475 /// — every repelling key that is not the group's own class.
9476 fn tolerations_excluding(archetypes: &[LifecycleArchetype]) -> Vec<String> {
9477 LifecycleArchetype::ALL
9478 .into_iter()
9479 .filter(|a| !archetypes.contains(a))
9480 .map(|a| format!("no-{}", a.taint_key()))
9481 .collect()
9482 }
9483
9484 #[test]
9485 fn live_taint_keys_lists_the_whole_legal_vocabulary() {
9486 assert_eq!(
9487 live_taint_keys(),
9488 vec!["no-appliance", "no-job", "no-server", "public-ip"]
9489 );
9490 }
9491
9492 // ─── R742-F1 (W305): sovereign groups ───────────────────────────────────
9493
9494 /// A machine in `group`, with the role left unwritten — which is the state
9495 /// of every machine TOML that predates R605-F12 and resolves to `voter`.
9496 fn in_group(name: &str, group: Option<&str>) -> MachineConfig {
9497 MachineConfig {
9498 sovereign_group: group.map(String::from),
9499 ..make_machine(name, vec![])
9500 }
9501 }
9502
9503 /// A machine in `group` with its quorum eligibility stated (R605-F12).
9504 fn in_group_as(name: &str, group: &str, role: SovereignRole) -> MachineConfig {
9505 MachineConfig {
9506 sovereign_group: Some(group.to_string()),
9507 sovereign_role: Some(role),
9508 ..make_machine(name, vec![])
9509 }
9510 }
9511
9512 #[test]
9513 fn a_join_within_one_sovereign_group_is_permitted() {
9514 assert_eq!(
9515 judge_join(
9516 &in_group("us-west-013", Some("dev")),
9517 &in_group("us-west-011", Some("dev")),
9518 ),
9519 JoinVerdict::Permit
9520 );
9521 }
9522
9523 /// The case the field exists for: before it, the only thing standing
9524 /// between a dev Pi and the prod quorum was a comment in a TOML.
9525 #[test]
9526 fn a_cross_group_join_is_refused_naming_both_groups() {
9527 let verdict = judge_join(
9528 &in_group("us-west-011", Some("dev")),
9529 &in_group("us-west-001", Some("prod")),
9530 );
9531 let JoinVerdict::Refuse(msg) = verdict else {
9532 panic!("a dev node joining prod must be refused: {verdict:?}");
9533 };
9534 // A refusal that does not name what it saw is one the operator has to
9535 // go and reconstruct, so it gets worked around instead of fixed.
9536 assert!(msg.contains("us-west-011") && msg.contains("us-west-001"), "{msg}");
9537 assert!(msg.contains("dev") && msg.contains("prod"), "{msg}");
9538 }
9539
9540 /// `None` is a declaration ("standalone, in no group"), not a gap — so
9541 /// growing prod with an unstamped box is a cross-group join too, and the
9542 /// refusal has to say which file makes it legal.
9543 #[test]
9544 fn an_undeclared_node_cannot_join_a_declared_group() {
9545 let verdict = judge_join(
9546 &in_group("us-west-002", None),
9547 &in_group("us-west-001", Some("prod")),
9548 );
9549 let JoinVerdict::Refuse(msg) = verdict else {
9550 panic!("an unstamped node joining prod must be refused: {verdict:?}");
9551 };
9552 assert!(
9553 msg.contains(".yah/infra/machines/us-west-002.toml"),
9554 "the refusal must name the file to stamp: {msg}"
9555 );
9556 }
9557
9558 #[test]
9559 fn a_declared_node_cannot_join_a_standalone_target() {
9560 // us-west-003 is `mode: standalone` on purpose; it is not a group of
9561 // one waiting to be grown.
9562 let verdict = judge_join(
9563 &in_group("us-west-001", Some("prod")),
9564 &in_group("us-west-003", None),
9565 );
9566 assert!(matches!(verdict, JoinVerdict::Refuse(msg) if msg.contains("us-west-003")));
9567 }
9568
9569 #[test]
9570 fn two_undeclared_nodes_cannot_form_an_undeclared_group() {
9571 let verdict = judge_join(
9572 &in_group("us-west-002", None),
9573 &in_group("us-west-015", None),
9574 );
9575 assert!(
9576 matches!(&verdict, JoinVerdict::Refuse(msg) if msg.contains("us-west-002")
9577 && msg.contains("us-west-015")),
9578 "forming a group nobody declared must be refused, naming both: {verdict:?}"
9579 );
9580 }
9581
9582 // ─── R605-F12: the voting axis ──────────────────────────────────────────
9583
9584 /// The whole ticket in one assertion. us-west-003 is a member of prod —
9585 /// same secrets, same upgrade cadence, same destruction — and must never
9586 /// hold a prod raft seat. Before the role axis, the only thing refusing it
9587 /// was its *absent* group stamp, so writing down the truth above would have
9588 /// removed the guard.
9589 #[test]
9590 fn a_non_voting_member_is_refused_into_its_own_group() {
9591 let verdict = judge_join(
9592 &in_group_as("us-west-003", "prod", SovereignRole::NonVoter),
9593 &in_group_as("us-west-001", "prod", SovereignRole::Voter),
9594 );
9595 let JoinVerdict::Refuse(msg) = verdict else {
9596 panic!("a non-voting prod member must not join the prod quorum: {verdict:?}");
9597 };
9598 assert!(msg.contains("us-west-003") && msg.contains("NON-VOTING"), "{msg}");
9599 // The refusal must not blame the group: both sides say "prod", and a
9600 // cross-group message here would read as a bug in the check itself.
9601 assert!(!msg.contains("cross-group"), "{msg}");
9602 assert!(
9603 msg.contains(".yah/infra/machines/us-west-003.toml"),
9604 "the refusal must name the file that decides it: {msg}"
9605 );
9606 }
9607
9608 /// Read from the other end: a box declared non-voting has no quorum seat to
9609 /// be grown, so it cannot be a join target either.
9610 #[test]
9611 fn a_non_voting_target_has_no_quorum_to_grow() {
9612 let verdict = judge_join(
9613 &in_group_as("us-west-001", "prod", SovereignRole::Voter),
9614 &in_group_as("us-west-003", "prod", SovereignRole::NonVoter),
9615 );
9616 assert!(
9617 matches!(&verdict, JoinVerdict::Refuse(msg) if msg.contains("the target")
9618 && msg.contains("us-west-003")),
9619 "{verdict:?}"
9620 );
9621 }
9622
9623 /// A non-voter joining a *standalone* target is refused for two reasons at
9624 /// once, and the message must pick the one whose fix would actually work.
9625 /// Naming the role here would send the operator to flip `sovereign_role`
9626 /// and come back to the same refusal.
9627 #[test]
9628 fn a_refusal_names_the_group_when_fixing_the_role_would_not_help() {
9629 let verdict = judge_join(
9630 &in_group_as("us-west-003", "prod", SovereignRole::NonVoter),
9631 &in_group("us-west-002", None),
9632 );
9633 let JoinVerdict::Refuse(msg) = verdict else {
9634 panic!("a standalone target has no group to join: {verdict:?}");
9635 };
9636 assert!(
9637 msg.contains(".yah/infra/machines/us-west-002.toml"),
9638 "the refusal must point at the target's missing group stamp: {msg}"
9639 );
9640 assert!(!msg.contains("NON-VOTING"), "{msg}");
9641 }
9642
9643 /// The back-compat seam, pinned: the six nodes stamped before R605-F12
9644 /// write no role, and an absent role means what declaring a group has
9645 /// always meant. If this flips, the live prod and dev quorums stop being
9646 /// growable on a config the operator never edited.
9647 #[test]
9648 fn an_unwritten_role_still_joins_its_group() {
9649 let joiner = in_group("us-west-013", Some("dev"));
9650 assert_eq!(joiner.sovereign_role, None);
9651 assert_eq!(
9652 judge_join(&joiner, &in_group("us-west-011", Some("dev"))),
9653 JoinVerdict::Permit
9654 );
9655 assert_eq!(
9656 judge_join(
9657 &joiner,
9658 &in_group_as("us-west-011", "dev", SovereignRole::Voter)
9659 ),
9660 JoinVerdict::Permit
9661 );
9662 }
9663
9664 /// A non-voting member is still a *member*, and the two claims must not be
9665 /// conflated: `sovereign_membership()` reports the group either way, so a
9666 /// consumer asking "is this box in prod's blast radius" gets yes.
9667 #[test]
9668 fn a_non_voter_is_still_in_the_group_it_names() {
9669 let m = in_group_as("us-west-003", "prod", SovereignRole::NonVoter);
9670 assert_eq!(m.sovereign_membership().group, Some("prod"));
9671 assert!(!m.sovereign_membership().role.is_voter());
9672
9673 // …and the group-membership query the fleet reads is unaffected by it.
9674 let cfg = make_empty_cfg(vec![
9675 m,
9676 in_group_as("us-west-001", "prod", SovereignRole::Voter),
9677 ]);
9678 assert_eq!(
9679 cfg.machines_in_group("prod")
9680 .iter()
9681 .map(|m| m.name.as_str())
9682 .collect::<Vec<_>>(),
9683 vec!["us-west-003", "us-west-001"]
9684 );
9685 }
9686
9687 /// The role travels through TOML in one spelling, and an absent one stays
9688 /// absent on the way back out — otherwise every machine file would grow a
9689 /// `sovereign_role = "voter"` line the operator never wrote, and the
9690 /// unroled-member lint would have nothing left to find.
9691 #[test]
9692 fn sovereign_role_round_trips_and_is_omitted_when_unwritten() {
9693 let m: MachineConfig = toml::from_str(
9694 r#"
9695name = "us-west-003"
9696provider = "static"
9697region = "us-west"
9698arch = "x86_64"
9699mesh_tags = []
9700sovereign_group = "prod"
9701sovereign_role = "non-voter"
9702"#,
9703 )
9704 .unwrap();
9705 assert_eq!(m.sovereign_role, Some(SovereignRole::NonVoter));
9706 assert!(toml::to_string(&m)
9707 .unwrap()
9708 .contains(r#"sovereign_role = "non-voter""#));
9709
9710 let unwritten = MachineConfig {
9711 sovereign_role: None,
9712 ..m
9713 };
9714 assert!(!toml::to_string(&unwritten)
9715 .unwrap()
9716 .contains("sovereign_role"));
9717 }
9718
9719 /// The invariant the ticket is most explicit about: a sovereign group is a
9720 /// blast radius, not a filter. If this ever fails, `matches` has grown an
9721 /// axis it must not have and dev-mode workloads have silently become
9722 /// unschedulable on the dev group.
9723 #[test]
9724 fn sovereign_group_is_not_a_placement_input() {
9725 let standalone = in_group("n", None);
9726 let grouped = in_group("n", Some("dev"));
9727 let other = in_group("n", Some("prod"));
9728
9729 for spec in [
9730 RequiredSpec::default(),
9731 RequiredSpec {
9732 regions: vec!["us-west".into()],
9733 ..Default::default()
9734 },
9735 RequiredSpec {
9736 tolerates: tolerations_excluding(&[LifecycleArchetype::Appliance]),
9737 ..Default::default()
9738 },
9739 ] {
9740 let baseline = spec.matches(&standalone);
9741 assert_eq!(spec.matches(&grouped), baseline);
9742 assert_eq!(spec.matches(&other), baseline);
9743 }
9744 }
9745
9746 // ─── R742-F3 (W305): group → machine set, and group-scoped admission ────
9747
9748 /// The primitive `migrate --to` needs and `rollout plan` still lacks
9749 /// (W314 gap 1): a group exists only as the set of machines naming it, so
9750 /// membership has to be derived rather than declared anywhere.
9751 #[test]
9752 fn machines_in_group_derives_membership_from_the_declarations() {
9753 let cfg = make_empty_cfg(vec![
9754 in_group("us-west-001", Some("prod")),
9755 in_group("us-west-011", Some("dev")),
9756 in_group("us-west-013", Some("dev")),
9757 in_group("us-west-002", None),
9758 ]);
9759
9760 let dev: Vec<&str> = cfg
9761 .machines_in_group("dev")
9762 .iter()
9763 .map(|m| m.name.as_str())
9764 .collect();
9765 assert_eq!(dev, vec!["us-west-011", "us-west-013"]);
9766 assert_eq!(cfg.machines_in_group("prod").len(), 1);
9767
9768 // Standalone is "in no group", not "in a group called none" — so an
9769 // unstamped box is never swept into a migration target.
9770 assert!(cfg.machines_in_group("").is_empty());
9771 assert!(cfg.machines_in_group("staging").is_empty());
9772 }
9773
9774 #[test]
9775 fn declared_sovereign_groups_is_the_vocabulary_a_bad_target_is_named_against() {
9776 let cfg = make_empty_cfg(vec![
9777 in_group("a", Some("prod")),
9778 in_group("b", Some("dev")),
9779 in_group("c", Some("prod")),
9780 in_group("d", None),
9781 ]);
9782 // Sorted + deduped, and standalone contributes nothing.
9783 assert_eq!(cfg.declared_sovereign_groups(), vec!["dev", "prod"]);
9784 assert!(make_empty_cfg(vec![in_group("a", None)])
9785 .declared_sovereign_groups()
9786 .is_empty());
9787 }
9788
9789 /// Group-scoped admission must be the SAME predicate as unscoped
9790 /// admission, only over fewer candidates. If it ever diverges, `migrate`
9791 /// becomes a way to place a workload somewhere `yah cloud apply` would
9792 /// refuse — which is exactly the silent routing-around W305 exists to stop.
9793 #[test]
9794 fn admit_workload_in_group_narrows_candidates_without_changing_the_predicate() {
9795 let mut prod = in_group("us-west-001", Some("prod"));
9796 prod.mesh_tags = vec!["tag:cloud-runner".into()];
9797 let mut dev_repels = in_group("us-west-011", Some("dev"));
9798 dev_repels.taints = vec!["no-appliance".into()];
9799 let mut dev_ok = in_group("us-west-013", Some("dev"));
9800 dev_ok.mesh_tags = vec!["tag:cloud-runner".into()];
9801
9802 let cfg = make_empty_cfg(vec![prod, dev_repels, dev_ok]);
9803
9804 let mut ws = ws_with_selector(None);
9805 ws.archetype = Some(LifecycleArchetype::Appliance);
9806
9807 // Unscoped picks the first match in declaration order.
9808 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "us-west-001");
9809 // Scoped skips the repelling dev node and lands on the other one —
9810 // the taint is honoured, not bypassed.
9811 assert_eq!(
9812 cfg.admit_workload_in_group(&ws, "dev").unwrap().name,
9813 "us-west-013"
9814 );
9815 }
9816
9817 #[test]
9818 fn admit_workload_in_group_distinguishes_an_empty_group_from_a_repelling_one() {
9819 let mut dev = in_group("us-west-011", Some("dev"));
9820 dev.taints = vec!["no-appliance".into()];
9821 let cfg = make_empty_cfg(vec![in_group("us-west-001", Some("prod")), dev]);
9822
9823 let mut ws = ws_with_selector(None);
9824 ws.archetype = Some(LifecycleArchetype::Appliance);
9825
9826 // A group nobody declares names the legal vocabulary, because a typo
9827 // is the realistic cause and "no candidates" would send the operator
9828 // hunting for a placement problem that does not exist.
9829 let missing = cfg.admit_workload_in_group(&ws, "stagng").unwrap_err().to_string();
9830 assert!(missing.contains("no machine declares"), "{missing}");
9831 assert!(missing.contains("dev") && missing.contains("prod"), "{missing}");
9832
9833 // A group that exists but refuses names the machines it tried.
9834 let repelled = cfg.admit_workload_in_group(&ws, "dev").unwrap_err().to_string();
9835 assert!(repelled.contains("us-west-011"), "{repelled}");
9836 }
9837
9838 #[test]
9839 fn sovereign_group_round_trips_and_is_omitted_when_standalone() {
9840 let src = r#"
9841name = "us-west-011"
9842provider = "static"
9843mesh_tags = []
9844sovereign_group = "dev"
9845"#;
9846 let m: MachineConfig = toml::from_str(src).unwrap();
9847 assert_eq!(m.sovereign_group.as_deref(), Some("dev"));
9848 assert!(toml::to_string(&m).unwrap().contains("sovereign_group"));
9849
9850 // A machine that predates the field parses as standalone and does not
9851 // grow the key back on write.
9852 let legacy: MachineConfig =
9853 toml::from_str("name = \"us-west-002\"\nprovider = \"static\"\nmesh_tags = []\n")
9854 .unwrap();
9855 assert_eq!(legacy.sovereign_group, None);
9856 assert!(!toml::to_string(&legacy).unwrap().contains("sovereign_group"));
9857 }
9858
9859 // ─── R572-F5: capacity floor + absolute (untolerable) taints ────────────
9860
9861 fn make_machine_with_capacity(
9862 name: &str,
9863 memory_mb: u32,
9864 cpu_millis: u32,
9865 taints: Vec<&str>,
9866 ) -> MachineConfig {
9867 MachineConfig {
9868 allocatable: Some(NodeAllocatable {
9869 memory_mb,
9870 cpu_millis,
9871 }),
9872 taints: taints.into_iter().map(String::from).collect(),
9873 ..make_machine(name, vec![])
9874 }
9875 }
9876
9877 fn server_spec(memory_mb: u32, cpu_millis: u32) -> WorkloadSpec {
9878 use workload_spec::{ImageRef, LifecycleArchetype, ResourceLimits, TierTag};
9879 let mut ws = WorkloadSpec::for_forge(
9880 "f5-test",
9881 ImageRef {
9882 registry: "localhost".into(),
9883 repository: "test".into(),
9884 tag: "latest".into(),
9885 digest: workload_spec::testing::test_digest(),
9886 },
9887 TierTag("infra".into()),
9888 vec![],
9889 );
9890 ws.archetype = Some(LifecycleArchetype::Server);
9891 ws.resources = ResourceLimits {
9892 memory_mb,
9893 cpu_millis,
9894 ephemeral_storage_mb: 0,
9895 };
9896 // These are SERVER specs that borrow `for_forge` as a constructor
9897 // shortcut, so drop the forge memory request it stamps on — otherwise
9898 // every spec here silently requests the forge default instead of the
9899 // `memory_mb` the caller passed, and the capacity-floor tests below
9900 // stop testing their own argument. A server workload declares no
9901 // request, which is the documented fall-back-to-`resources.memory_mb`
9902 // path (`WorkloadSpec::memory_request_mb`).
9903 ws.annotations
9904 .remove(workload_spec::MEMORY_REQUEST_ANNOTATION);
9905 ws
9906 }
9907
9908 fn appliance_spec_ws(memory_mb: u32, cpu_millis: u32) -> WorkloadSpec {
9909 use workload_spec::LifecycleArchetype;
9910 let mut ws = server_spec(memory_mb, cpu_millis);
9911 ws.archetype = Some(LifecycleArchetype::Appliance);
9912 ws
9913 }
9914
9915 #[test]
9916 fn capacity_floor_rejects_undersized_node() {
9917 let cfg = make_empty_cfg(vec![make_machine_with_capacity("small", 256, 500, vec![])]);
9918 let ws = server_spec(512, 1000); // demands more than available
9919 assert!(cfg.admit_workload(&ws).is_err());
9920 }
9921
9922 #[test]
9923 fn capacity_floor_accepts_exact_fit() {
9924 let cfg = make_empty_cfg(vec![make_machine_with_capacity("exact", 512, 1000, vec![])]);
9925 let ws = server_spec(512, 1000);
9926 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "exact");
9927 }
9928
9929 #[test]
9930 fn capacity_floor_passes_when_allocatable_absent() {
9931 // A machine with no allocatable block skips the capacity check (no data).
9932 let cfg = make_empty_cfg(vec![make_machine("no-alloc", vec![])]);
9933 let ws = server_spec(99999, 99999); // would exceed any real node
9934 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "no-alloc");
9935 }
9936
9937 #[test]
9938 fn taint_repulsion_blocks_appliance_on_no_appliance_node() {
9939 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
9940 "south",
9941 1024,
9942 2000,
9943 vec!["no-appliance"],
9944 )]);
9945 let ws = appliance_spec_ws(256, 500);
9946 assert!(
9947 cfg.admit_workload(&ws).is_err(),
9948 "appliance must be repelled by no-appliance taint"
9949 );
9950 }
9951
9952 #[test]
9953 fn taint_repulsion_allows_server_on_no_appliance_node() {
9954 // "no-appliance" only repels Appliance workloads; servers are unaffected.
9955 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
9956 "south",
9957 1024,
9958 2000,
9959 vec!["no-appliance"],
9960 )]);
9961 let ws = server_spec(256, 500);
9962 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "south");
9963 }
9964
9965 #[test]
9966 fn taint_repulsion_job_not_blocked_by_no_server() {
9967 use workload_spec::LifecycleArchetype;
9968 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
9969 "build-box",
9970 8192,
9971 4000,
9972 vec!["no-server", "no-appliance"],
9973 )]);
9974 let mut ws = server_spec(256, 500);
9975 ws.archetype = Some(LifecycleArchetype::Job);
9976 // Job only repelled by "no-job"; "no-server" and "no-appliance" don't affect it.
9977 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "build-box");
9978 }
9979
9980 #[test]
9981 fn requires_taint_affinity_blocks_placement_without_it() {
9982 use workload_spec::{LifecycleArchetype, PUBLIC_IP_TAINT, REQUIRES_TAINT_ANNOTATION};
9983 // Simulate the passway ingress appliance: requires "public-ip" taint.
9984 let mut ws = appliance_spec_ws(256, 512);
9985 ws.archetype = Some(LifecycleArchetype::Appliance);
9986 ws.annotations
9987 .insert(REQUIRES_TAINT_ANNOTATION.into(), PUBLIC_IP_TAINT.into());
9988
9989 // Node without the taint: rejected.
9990 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
9991 "no-pip",
9992 2048,
9993 2000,
9994 vec![],
9995 )]);
9996 assert!(cfg.admit_workload(&ws).is_err());
9997
9998 // Node with the taint: accepted.
9999 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
10000 "pub-node",
10001 2048,
10002 2000,
10003 vec!["public-ip"],
10004 )]);
10005 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "pub-node");
10006 }
10007
10008 #[test]
10009 fn w244_fleet_scenario_appliance_rejected_from_south_and_west002() {
10010 // Full W244 fleet table scenario:
10011 // us-west-001/east-001: no taints, large capacity → appliance lands here
10012 // us-south-001: no-appliance taint → appliance rejected
10013 // us-west-002: no-server, no-appliance → appliance rejected
10014 let cfg = make_empty_cfg(vec![
10015 make_machine_with_capacity("us-south-001", 512, 1000, vec!["no-appliance"]),
10016 make_machine_with_capacity(
10017 "us-west-002",
10018 16384,
10019 8000,
10020 vec!["no-server", "no-appliance"],
10021 ),
10022 make_machine_with_capacity("us-west-001", 4096, 4000, vec![]),
10023 ]);
10024 let ws = appliance_spec_ws(256, 500);
10025 // Skips south (no-appliance) and west-002 (no-appliance), lands on west-001.
10026 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "us-west-001");
10027 }
10028
10029 #[test]
10030 fn w244_fleet_scenario_job_lands_on_west002_first() {
10031 use workload_spec::LifecycleArchetype;
10032 // Jobs should prefer (or at least land on) the job-only box.
10033 let cfg = make_empty_cfg(vec![
10034 make_machine_with_capacity("us-west-001", 4096, 4000, vec![]),
10035 make_machine_with_capacity(
10036 "us-west-002",
10037 16384,
10038 8000,
10039 vec!["no-server", "no-appliance"],
10040 ),
10041 ]);
10042 let mut ws = server_spec(256, 500);
10043 ws.archetype = Some(LifecycleArchetype::Job);
10044 // No fleet node declares `no-job`, so a Job is repelled by nothing;
10045 // west-001 comes first in declaration order (greedy, no preference),
10046 // which is the expected tie-break. Note this is *absence of a repel
10047 // key*, not toleration — no workload can tolerate a taint (W305).
10048 let picked = cfg.admit_workload(&ws).unwrap();
10049 // Both are eligible.
10050 assert!(
10051 picked.name == "us-west-001" || picked.name == "us-west-002",
10052 "job must land on an eligible node, got {}",
10053 picked.name
10054 );
10055 }
10056
10057 #[test]
10058 fn r569_f4_macos_node_taints_keep_cloud_critical_off_but_admit_build_jobs() {
10059 use workload_spec::LifecycleArchetype;
10060 // R569-F4: the headless M2 (us-west-015) joins the fleet as a
10061 // build-worker but must never take cloud-critical load. It carries the
10062 // same repel set as the x86 build-worker (`no-server, no-appliance` —
10063 // see .yah/infra/machines/us-west-015.toml). This pins that intent:
10064 // with a plain cloud node available beside the Mac, every
10065 // cloud-critical archetype lands on the cloud node and never the Mac;
10066 // build Jobs (the Mac's actual purpose) remain eligible on it.
10067 //
10068 // R742-T4: `no-voter` used to sit in this set and in the TOML. It was
10069 // never read here — there is no "voter" workload archetype — and
10070 // R569-F3's learner-only join is what actually keeps the box out of
10071 // quorum. It is now rejected by `yah cloud validate` as inert.
10072 let mac_taints = vec!["no-server", "no-appliance"];
10073 let fleet = || {
10074 make_empty_cfg(vec![
10075 make_machine_with_capacity("us-west-015", 24576, 8000, mac_taints.clone()),
10076 make_machine_with_capacity("us-west-001", 4096, 4000, vec![]),
10077 ])
10078 };
10079
10080 // A cloud-critical Server workload is repelled from the Mac and lands
10081 // on the untainted cloud node.
10082 let cfg = fleet();
10083 assert_eq!(
10084 cfg.admit_workload(&server_spec(256, 500)).unwrap().name,
10085 "us-west-001",
10086 "a Server workload must never land on the no-server Mac node"
10087 );
10088
10089 // Same for an Appliance (pinned/stateful cloud-critical) workload.
10090 let cfg = fleet();
10091 assert_eq!(
10092 cfg.admit_workload(&appliance_spec_ws(256, 500))
10093 .unwrap()
10094 .name,
10095 "us-west-001",
10096 "an Appliance workload must never land on the no-appliance Mac node"
10097 );
10098
10099 // Sharpest repulsion proof: with ONLY the Mac in the fleet, a
10100 // cloud-critical Server workload is rejected outright — the taint keeps
10101 // it off even when that means nowhere to run.
10102 let mac_only = make_empty_cfg(vec![make_machine_with_capacity(
10103 "us-west-015",
10104 24576,
10105 8000,
10106 mac_taints.clone(),
10107 )]);
10108 assert!(
10109 mac_only.admit_workload(&server_spec(256, 500)).is_err(),
10110 "a Server workload must be repelled from a Mac-only fleet, not admitted"
10111 );
10112
10113 // But the Mac's real job — build/forge workloads — IS admitted on it:
10114 // it tolerates every fleet taint (there is no `no-job`).
10115 let mut job = server_spec(256, 500);
10116 job.archetype = Some(LifecycleArchetype::Job);
10117 assert_eq!(
10118 mac_only.admit_workload(&job).unwrap().name,
10119 "us-west-015",
10120 "a build Job must still be admitted on the Mac build-worker"
10121 );
10122 }
10123
10124 // ─── R615-F1: linked infra sources (`.yah/infra/sources.toml`) ─────────
10125
10126 #[test]
10127 fn sources_load_is_empty_when_the_file_is_absent() {
10128 // "Every camp without linked infra has none" — which today is every
10129 // camp — must not be an error.
10130 let tmp = tempfile::TempDir::new().unwrap();
10131 let cfg = SourcesConfig::load(tmp.path()).unwrap();
10132 assert_eq!(cfg, SourcesConfig::default());
10133 assert!(cfg.source.is_empty());
10134 assert_eq!(cfg.schema_version, 1);
10135 }
10136
10137 #[test]
10138 fn sources_parses_a_path_kind_exactly_like_w274s_example() {
10139 let tmp = tempfile::TempDir::new().unwrap();
10140 std::fs::write(
10141 tmp.path().join("sources.toml"),
10142 r#"
10143schema_version = 1
10144
10145[[source]]
10146owner = "yah"
10147kind = "path"
10148path = "../yah"
10149mode = "read-only"
10150"#,
10151 )
10152 .unwrap();
10153 let cfg = SourcesConfig::load(tmp.path()).unwrap();
10154 assert_eq!(cfg.source.len(), 1);
10155 let s = &cfg.source[0];
10156 assert_eq!(s.owner, "yah");
10157 assert_eq!(s.mode, SourceMode::ReadOnly);
10158 assert!(s.select.is_empty());
10159 match &s.kind {
10160 InfraSourceKind::Path { path } => assert_eq!(path, "../yah"),
10161 other => panic!("expected Path, got {other:?}"),
10162 }
10163 }
10164
10165 #[test]
10166 fn sources_parses_a_git_kind_reusing_gitsource_verbatim() {
10167 let tmp = tempfile::TempDir::new().unwrap();
10168 std::fs::write(
10169 tmp.path().join("sources.toml"),
10170 r#"
10171schema_version = 1
10172
10173[[source]]
10174owner = "yah"
10175kind = "git"
10176repo = "git@github.com:yah-ai/infra.git"
10177ref = "main"
10178subdir = "infra"
10179select = ["tag:cloud-runner"]
10180mode = "read-only"
10181"#,
10182 )
10183 .unwrap();
10184 let cfg = SourcesConfig::load(tmp.path()).unwrap();
10185 assert_eq!(cfg.source.len(), 1);
10186 let s = &cfg.source[0];
10187 assert_eq!(s.select, vec!["tag:cloud-runner".to_string()]);
10188 match &s.kind {
10189 InfraSourceKind::Git(git) => {
10190 assert_eq!(git.repo, "git@github.com:yah-ai/infra.git");
10191 assert_eq!(git.r#ref, "main");
10192 assert_eq!(git.subdir.as_deref(), Some("infra"));
10193 }
10194 other => panic!("expected Git, got {other:?}"),
10195 }
10196 }
10197
10198 #[test]
10199 fn sources_mode_defaults_to_read_only_and_manage_is_explicit() {
10200 let tmp = tempfile::TempDir::new().unwrap();
10201 std::fs::write(
10202 tmp.path().join("sources.toml"),
10203 r#"
10204schema_version = 1
10205
10206[[source]]
10207owner = "a"
10208kind = "path"
10209path = "../a"
10210
10211[[source]]
10212owner = "b"
10213kind = "path"
10214path = "../b"
10215mode = "manage"
10216"#,
10217 )
10218 .unwrap();
10219 let cfg = SourcesConfig::load(tmp.path()).unwrap();
10220 assert_eq!(cfg.source[0].mode, SourceMode::ReadOnly, "omitted mode = read-only");
10221 assert_eq!(cfg.source[1].mode, SourceMode::Manage);
10222 }
10223
10224 #[test]
10225 fn sources_preserves_declaration_order() {
10226 // Overlay order matters (R615-F2) when two sources name the same
10227 // machine — the list must round-trip in file order, not be reordered
10228 // by owner or kind.
10229 let tmp = tempfile::TempDir::new().unwrap();
10230 std::fs::write(
10231 tmp.path().join("sources.toml"),
10232 r#"
10233schema_version = 1
10234
10235[[source]]
10236owner = "second"
10237kind = "path"
10238path = "../second"
10239
10240[[source]]
10241owner = "first"
10242kind = "path"
10243path = "../first"
10244"#,
10245 )
10246 .unwrap();
10247 let cfg = SourcesConfig::load(tmp.path()).unwrap();
10248 let owners: Vec<&str> = cfg.source.iter().map(|s| s.owner.as_str()).collect();
10249 assert_eq!(owners, vec!["second", "first"]);
10250 }
10251
10252 #[test]
10253 fn sources_round_trips_through_serialize() {
10254 let cfg = SourcesConfig {
10255 schema_version: 1,
10256 source: vec![
10257 InfraSource {
10258 owner: "yah".into(),
10259 kind: InfraSourceKind::Path {
10260 path: "../yah".into(),
10261 },
10262 mode: SourceMode::ReadOnly,
10263 select: vec![],
10264 },
10265 InfraSource {
10266 owner: "yah".into(),
10267 kind: InfraSourceKind::Git(GitSource {
10268 repo: "git@github.com:yah-ai/infra.git".into(),
10269 r#ref: "main".into(),
10270 subdir: Some("infra".into()),
10271 }),
10272 mode: SourceMode::Manage,
10273 select: vec!["tag:cloud-runner".into()],
10274 },
10275 ],
10276 };
10277 let toml_str = toml::to_string_pretty(&cfg).unwrap();
10278 let reloaded: SourcesConfig = toml::from_str(&toml_str).unwrap();
10279 assert_eq!(reloaded, cfg, "round-trip through TOML must be lossless:\n{toml_str}");
10280 }
10281
10282 // ─── R615-F2: overlay loader in CloudConfig::load ───────────────────────
10283
10284 fn write_min_machine(dir: &Path, name: &str, extra_toml: &str) {
10285 std::fs::create_dir_all(dir).unwrap();
10286 // `extra_toml` supplies `mesh_tags` when the caller cares about it;
10287 // otherwise default to the empty list. Never hardcode `mesh_tags`
10288 // here as well as in `extra_toml` -- TOML rejects a duplicate key.
10289 let mesh_tags = if extra_toml.contains("mesh_tags") {
10290 String::new()
10291 } else {
10292 "mesh_tags = []\n".to_string()
10293 };
10294 std::fs::write(
10295 dir.join(format!("{name}.toml")),
10296 format!("name = \"{name}\"\nprovider = \"static\"\n{mesh_tags}{extra_toml}"),
10297 )
10298 .unwrap();
10299 }
10300
10301 fn write_min_provider(dir: &Path, id: &str) {
10302 std::fs::create_dir_all(dir).unwrap();
10303 std::fs::write(
10304 dir.join(format!("{id}.toml")),
10305 format!("schema_version = 1\nid = \"{id}\"\nkind = \"static\"\n"),
10306 )
10307 .unwrap();
10308 }
10309
10310 fn write_sources_toml(camp_root: &Path, body: &str) {
10311 let dir = camp_root.join(".yah/infra");
10312 std::fs::create_dir_all(&dir).unwrap();
10313 std::fs::write(dir.join("sources.toml"), body).unwrap();
10314 }
10315
10316 #[test]
10317 fn load_with_no_sources_toml_is_unchanged() {
10318 let tmp = tempfile::TempDir::new().unwrap();
10319 write_min_machine(&tmp.path().join(".yah/infra/machines"), "local-1", "");
10320 let cfg = CloudConfig::load(tmp.path()).unwrap();
10321 assert_eq!(cfg.machines.len(), 1);
10322 assert!(cfg.machine_origins.is_empty());
10323 assert!(cfg.provider_origins.is_empty());
10324 }
10325
10326 #[test]
10327 fn path_source_overlays_machines_and_providers_tagged_with_origin() {
10328 let camp = tempfile::TempDir::new().unwrap();
10329 let other = tempfile::TempDir::new().unwrap();
10330 write_min_machine(&other.path().join(".yah/infra/machines"), "borrowed-1", "");
10331 write_min_provider(&other.path().join(".yah/infra/providers"), "borrowed-provider");
10332 write_sources_toml(
10333 camp.path(),
10334 &format!(
10335 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
10336 other.path().display()
10337 ),
10338 );
10339
10340 let cfg = CloudConfig::load(camp.path()).unwrap();
10341 assert_eq!(cfg.machines.len(), 1);
10342 assert_eq!(cfg.machines[0].name, "borrowed-1");
10343 assert_eq!(cfg.providers.len(), 1);
10344 assert_eq!(cfg.providers[0].id, "borrowed-provider");
10345
10346 let origin = cfg.machine_origins.get("borrowed-1").expect("origin recorded");
10347 assert_eq!(origin.owner, "other");
10348 assert_eq!(origin.mode, SourceMode::ReadOnly);
10349 assert!(origin.source.starts_with("path:"));
10350 assert_eq!(
10351 cfg.provider_origins.get("borrowed-provider").unwrap().owner,
10352 "other"
10353 );
10354 }
10355
10356 #[test]
10357 fn camp_local_wins_on_name_collision_and_carries_no_origin() {
10358 let camp = tempfile::TempDir::new().unwrap();
10359 let other = tempfile::TempDir::new().unwrap();
10360 // Both declare a machine named "shared" -- camp-local's copy must win,
10361 // and it must never gain an origin tag.
10362 write_min_machine(&camp.path().join(".yah/infra/machines"), "shared", "");
10363 write_min_machine(
10364 &other.path().join(".yah/infra/machines"),
10365 "shared",
10366 "nickname = \"the borrowed one\"\n",
10367 );
10368 write_sources_toml(
10369 camp.path(),
10370 &format!(
10371 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
10372 other.path().display()
10373 ),
10374 );
10375
10376 let cfg = CloudConfig::load(camp.path()).unwrap();
10377 assert_eq!(cfg.machines.len(), 1, "the name collides, so exactly one entry");
10378 assert_eq!(cfg.machines[0].nickname, None, "camp-local's copy, not the borrowed one");
10379 assert!(
10380 !cfg.machine_origins.contains_key("shared"),
10381 "camp-local entries never carry an origin tag"
10382 );
10383 }
10384
10385 #[test]
10386 fn an_earlier_source_wins_over_a_later_one_on_collision() {
10387 let camp = tempfile::TempDir::new().unwrap();
10388 let first = tempfile::TempDir::new().unwrap();
10389 let second = tempfile::TempDir::new().unwrap();
10390 write_min_machine(&first.path().join(".yah/infra/machines"), "dup", "");
10391 write_min_machine(&second.path().join(".yah/infra/machines"), "dup", "");
10392 write_sources_toml(
10393 camp.path(),
10394 &format!(
10395 "schema_version = 1\n\n[[source]]\nowner = \"first\"\nkind = \"path\"\npath = \"{}\"\n\n[[source]]\nowner = \"second\"\nkind = \"path\"\npath = \"{}\"\n",
10396 first.path().display(),
10397 second.path().display()
10398 ),
10399 );
10400
10401 let cfg = CloudConfig::load(camp.path()).unwrap();
10402 assert_eq!(cfg.machines.len(), 1);
10403 assert_eq!(cfg.machine_origins.get("dup").unwrap().owner, "first");
10404 }
10405
10406 #[test]
10407 fn select_filters_borrowed_machines_by_name_or_mesh_tag() {
10408 let camp = tempfile::TempDir::new().unwrap();
10409 let other = tempfile::TempDir::new().unwrap();
10410 write_min_machine(&other.path().join(".yah/infra/machines"), "runner-1", "mesh_tags = [\"tag:cloud-runner\"]\n");
10411 write_min_machine(&other.path().join(".yah/infra/machines"), "excluded-1", "");
10412 write_sources_toml(
10413 camp.path(),
10414 &format!(
10415 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\nselect = [\"tag:cloud-runner\"]\n",
10416 other.path().display()
10417 ),
10418 );
10419
10420 let cfg = CloudConfig::load(camp.path()).unwrap();
10421 assert_eq!(cfg.machines.len(), 1);
10422 assert_eq!(cfg.machines[0].name, "runner-1");
10423 }
10424
10425 #[test]
10426 fn one_unparseable_foreign_machine_does_not_sink_the_rest_of_the_directory_or_the_load() {
10427 let camp = tempfile::TempDir::new().unwrap();
10428 let other = tempfile::TempDir::new().unwrap();
10429 let dir = other.path().join(".yah/infra/machines");
10430 write_min_machine(&dir, "good", "");
10431 // Schema-skew gotcha: a foreign machine this binary's MachineConfig
10432 // can't parse at all (not just an unknown field -- MachineConfig has
10433 // no deny_unknown_fields, so this has to fail on a TYPE, not a name).
10434 std::fs::write(dir.join("bad.toml"), "name = 1\nprovider = 2\n").unwrap();
10435 write_sources_toml(
10436 camp.path(),
10437 &format!(
10438 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
10439 other.path().display()
10440 ),
10441 );
10442
10443 // Must not error at all -- camp-local load must never fail because a
10444 // source it doesn't own has one bad file.
10445 let cfg = CloudConfig::load(camp.path()).unwrap();
10446 assert_eq!(cfg.machines.len(), 1, "the good entry still loads");
10447 assert_eq!(cfg.machines[0].name, "good");
10448 }
10449
10450 #[test]
10451 fn an_unsynced_git_source_overlays_nothing_and_is_not_an_error() {
10452 // No `yah infra sync` (R615-T3) has ever run, so the cache dir this
10453 // resolves to doesn't exist. Must be silent, not fatal.
10454 let camp = tempfile::TempDir::new().unwrap();
10455 write_sources_toml(
10456 camp.path(),
10457 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"git\"\nrepo = \"git@github.com:yah-ai/infra.git\"\nref = \"main\"\n",
10458 );
10459 let cfg = CloudConfig::load(camp.path()).unwrap();
10460 assert!(cfg.machines.is_empty());
10461 assert!(cfg.machine_origins.is_empty());
10462 }
10463
10464 #[test]
10465 fn a_synced_git_source_reads_from_the_cache_dir_not_the_repo_path() {
10466 // No `subdir` declared -- the checkout ROOT is the infra root.
10467 let camp = tempfile::TempDir::new().unwrap();
10468 let cache = crate::paths::infra_source_cache_dir(camp.path(), "yah");
10469 write_min_machine(&cache.join("machines"), "synced-1", "");
10470 write_sources_toml(
10471 camp.path(),
10472 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"git\"\nrepo = \"git@github.com:yah-ai/infra.git\"\nref = \"main\"\n",
10473 );
10474 let cfg = CloudConfig::load(camp.path()).unwrap();
10475 assert_eq!(cfg.machines.len(), 1);
10476 assert_eq!(cfg.machines[0].name, "synced-1");
10477 assert!(cfg.machine_origins.get("synced-1").unwrap().source.starts_with("git:"));
10478 }
10479
10480 #[test]
10481 fn a_git_sources_subdir_is_honoured_like_the_component_case() {
10482 // W274's own example declares `subdir = "infra"` for a monorepo whose
10483 // registry lives under a subdirectory of the clone rather than at its
10484 // root -- prove `infra_root` actually reads it, not just `.subdir` on
10485 // GitSource parsing (R615-F1 already covers that half).
10486 let camp = tempfile::TempDir::new().unwrap();
10487 let cache = crate::paths::infra_source_cache_dir(camp.path(), "yah");
10488 write_min_machine(&cache.join("infra").join("machines"), "subdir-1", "");
10489 // Also plant a decoy at the checkout root to prove the root itself is
10490 // NOT read when a subdir is declared.
10491 write_min_machine(&cache.join("machines"), "root-decoy", "");
10492 write_sources_toml(
10493 camp.path(),
10494 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"git\"\nrepo = \"git@github.com:yah-ai/infra.git\"\nref = \"main\"\nsubdir = \"infra\"\n",
10495 );
10496 let cfg = CloudConfig::load(camp.path()).unwrap();
10497 assert_eq!(cfg.machines.len(), 1);
10498 assert_eq!(cfg.machines[0].name, "subdir-1");
10499 }
10500
10501 #[test]
10502 fn load_from_config_dir_never_applies_sources_overlay() {
10503 // R615-F2's explicit decision: multi-root sibling trees don't inherit
10504 // the classic .yah/infra/sources.toml. Prove it rather than assert it
10505 // silently -- a sources.toml sitting at workspace_root/.yah/infra/
10506 // must NOT leak into a load_from_config_dir call even though both
10507 // share the same workspace_root.
10508 let camp = tempfile::TempDir::new().unwrap();
10509 let other = tempfile::TempDir::new().unwrap();
10510 write_min_machine(&other.path().join(".yah/infra/machines"), "borrowed-1", "");
10511 write_sources_toml(
10512 camp.path(),
10513 &format!(
10514 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
10515 other.path().display()
10516 ),
10517 );
10518 let sibling_config_dir = camp.path().join(".noisetable");
10519 std::fs::create_dir_all(&sibling_config_dir).unwrap();
10520
10521 let cfg = CloudConfig::load_from_config_dir(&sibling_config_dir, camp.path()).unwrap();
10522 assert!(cfg.machines.is_empty(), "sources.toml must not apply here");
10523 assert!(cfg.machine_origins.is_empty());
10524 }
10525
10526 // ─── R615-T5: `inherit_machines` retirement — cutover proof ────────────
10527
10528 /// The successor to R615-T5's parity proof. That earlier pair of tests
10529 /// asserted the legacy `[infra].inherit_machines` redirect and an
10530 /// equivalent `kind = "path"` source resolved the same machine set, and
10531 /// that the two coexisted without duplicating rows. Both claims were about
10532 /// a mechanism that no longer exists, so they retired with it — what has
10533 /// to hold *now* is the other half of the same guarantee: a camp that
10534 /// declares only `sources.toml` resolves the shared root exactly as the
10535 /// redirect used to, and a stale `inherit_machines` key left behind in
10536 /// `camp.toml` changes nothing.
10537 ///
10538 /// That stale-key case is not hypothetical: it is precisely the state a
10539 /// camp is in between the code cutover and someone tidying its
10540 /// `camp.toml`, and a silent re-resolution there would double-count the
10541 /// borrowed nodes or hide their origin badge.
10542 #[test]
10543 fn a_stale_inherit_machines_key_does_not_change_what_sources_toml_resolves() {
10544 let shared = tempfile::TempDir::new().unwrap();
10545 write_min_machine(&shared.path().join(".yah/infra/machines"), "shared-node-1", "");
10546 write_min_machine(&shared.path().join(".yah/infra/machines"), "shared-node-2", "");
10547
10548 let sources_toml = format!(
10549 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"path\"\npath = \"{}\"\nmode = \"read-only\"\n",
10550 shared.path().display()
10551 );
10552
10553 // Camp A: migrated cleanly — sources.toml only.
10554 let clean = tempfile::TempDir::new().unwrap();
10555 write_sources_toml(clean.path(), &sources_toml);
10556
10557 // Camp B: mid-migration — same source, plus the retired key still
10558 // sitting in camp.toml pointing at the same root.
10559 let stale = tempfile::TempDir::new().unwrap();
10560 std::fs::create_dir_all(stale.path().join(".yah")).unwrap();
10561 std::fs::write(
10562 stale.path().join(".yah/camp.toml"),
10563 format!(
10564 "[infra]\ninherit_machines = \"{}\"\n",
10565 shared.path().display()
10566 ),
10567 )
10568 .unwrap();
10569 write_sources_toml(stale.path(), &sources_toml);
10570
10571 let via_clean = CloudConfig::load(clean.path()).unwrap();
10572 let via_stale = CloudConfig::load(stale.path()).unwrap();
10573
10574 let names = |cfg: &CloudConfig| {
10575 let mut v: Vec<String> = cfg.machines.iter().map(|m| m.name.clone()).collect();
10576 v.sort();
10577 v
10578 };
10579 assert_eq!(
10580 names(&via_clean),
10581 names(&via_stale),
10582 "a leftover inherit_machines key must be inert — the retired redirect is gone"
10583 );
10584 assert_eq!(names(&via_clean), vec!["shared-node-1", "shared-node-2"]);
10585
10586 // And both are *borrowed*, not camp-local. This is the operator-facing
10587 // win the stopgap could never deliver: under the old redirect these
10588 // resolved with no origin at all, indistinguishable from locally-owned
10589 // nodes.
10590 assert_eq!(via_clean.machine_origins.len(), 2);
10591 assert_eq!(via_stale.machine_origins.len(), 2);
10592 for origin in via_stale.machine_origins.values() {
10593 assert_eq!(origin.owner, "yah");
10594 assert_eq!(origin.mode, SourceMode::ReadOnly);
10595 }
10596 }
10597
10598 // ─── R860-T4 (W338): placement groups ───────────────────────────────────
10599
10600 /// One requirement edge, written the way a spec author writes it.
10601 fn requirement(ident: &str, locality: Locality) -> workload_spec::Requirement {
10602 workload_spec::Requirement {
10603 ident: workload_spec::MeshIdent(ident.into()),
10604 locality,
10605 supply: workload_spec::Supply::Wait,
10606 provides: None,
10607 }
10608 }
10609
10610 /// A `minimal_spec` (256 MiB / 250 millicores, Server by inference) that
10611 /// requires the given edges.
10612 fn spec_requiring(name: &str, requires: Vec<workload_spec::Requirement>) -> WorkloadSpec {
10613 WorkloadSpec {
10614 requires,
10615 ..minimal_spec(name, 1)
10616 }
10617 }
10618
10619 /// The declared inventory an ident is resolved against — `.yah/infra/workloads/`.
10620 fn declared(specs: Vec<WorkloadSpec>) -> Vec<WorkloadConfig> {
10621 specs.into_iter().map(|spec| WorkloadConfig { spec }).collect()
10622 }
10623
10624 fn member_names(group: &[WorkloadSpec]) -> Vec<&str> {
10625 group.iter().map(|s| s.name.as_str()).collect()
10626 }
10627
10628 /// The headline case: `local` means "same node", so the two specs are one
10629 /// placement unit and admission has to reason about both.
10630 #[test]
10631 fn a_local_edge_binds_the_provider_into_the_placement_group() {
10632 let inventory = declared(vec![minimal_spec("headscale-replicator", 1)]);
10633 let requirer = spec_requiring(
10634 "headscale",
10635 vec![requirement("headscale-replicator", Locality::Local)],
10636 );
10637
10638 assert_eq!(
10639 member_names(&placement_group(&requirer, &inventory)),
10640 vec!["headscale", "headscale-replicator"]
10641 );
10642 }
10643
10644 /// The edge that must NOT bind. `prefer-local` "never blocks placement"
10645 /// (W338's locality table), and `anywhere` — which is what every legacy
10646 /// `depends_on` folds into — is an ordinary service dependency. Binding
10647 /// either would silently make every dependency in the tree a co-scheduling
10648 /// constraint and start summing unrelated workloads into the capacity floor.
10649 #[test]
10650 fn prefer_local_and_anywhere_edges_do_not_bind_the_group() {
10651 let inventory = declared(vec![
10652 minimal_spec("headscale-db", 1),
10653 minimal_spec("metrics", 1),
10654 minimal_spec("legacy-dep", 1),
10655 ]);
10656
10657 let requirer = WorkloadSpec {
10658 depends_on: vec![workload_spec::MeshIdent("legacy-dep".into())],
10659 ..spec_requiring(
10660 "headscale",
10661 vec![
10662 requirement("headscale-db", Locality::PreferLocal),
10663 requirement("metrics", Locality::Anywhere),
10664 ],
10665 )
10666 };
10667
10668 assert_eq!(
10669 member_names(&placement_group(&requirer, &inventory)),
10670 vec!["headscale"]
10671 );
10672 }
10673
10674 /// Transitive, and via the inline spec a `supply = "self"` requirement
10675 /// carries rather than via an ident lookup — the sidecar shape W338's
10676 /// worked example is built on.
10677 #[test]
10678 fn the_group_is_the_transitive_closure_and_traverses_inline_provides() {
10679 let inline = workload_spec::Requirement {
10680 supply: workload_spec::Supply::SelfProvision,
10681 provides: Some(Box::new(minimal_spec("headscale-restore", 1))),
10682 ..requirement("headscale-restore", Locality::Local)
10683 };
10684 let middle = WorkloadSpec {
10685 requires: vec![requirement("wal-shipper", Locality::Local)],
10686 ..minimal_spec("headscale-replicator", 1)
10687 };
10688 let inventory = declared(vec![middle, minimal_spec("wal-shipper", 1)]);
10689
10690 let requirer = spec_requiring(
10691 "headscale",
10692 vec![
10693 inline,
10694 requirement("headscale-replicator", Locality::Local),
10695 ],
10696 );
10697
10698 assert_eq!(
10699 member_names(&placement_group(&requirer, &inventory)),
10700 vec![
10701 "headscale",
10702 "headscale-restore",
10703 "headscale-replicator",
10704 "wal-shipper"
10705 ]
10706 );
10707 }
10708
10709 /// `validate::check_requires` bounds `provides` nesting to depth 1 but
10710 /// cannot stop two separately-declared specs from naming each other. Without
10711 /// the visited set this closure never terminates, so admission would hang
10712 /// rather than refuse — the worst failure shape for a deploy gate.
10713 #[test]
10714 fn an_ident_cycle_closes_the_group_instead_of_looping_forever() {
10715 let b = spec_requiring("b", vec![requirement("a", Locality::Local)]);
10716 let a = spec_requiring("a", vec![requirement("b", Locality::Local)]);
10717 let inventory = declared(vec![a.clone(), b]);
10718
10719 assert_eq!(member_names(&placement_group(&a, &inventory)), vec!["a", "b"]);
10720 }
10721
10722 /// An unresolvable ident is skipped, not fatal: admission is a pure function
10723 /// of the declared inventory, and refusing every deploy whose provider is
10724 /// not yet declared would make `requires` unusable before R860-T6 lands.
10725 #[test]
10726 fn an_unresolvable_local_ident_is_skipped_rather_than_refused() {
10727 let requirer = spec_requiring("headscale", vec![requirement("not-declared", Locality::Local)]);
10728 assert_eq!(
10729 member_names(&placement_group(&requirer, &[])),
10730 vec!["headscale"]
10731 );
10732 }
10733
10734 /// W338 §Placement consequences 1: the capacity floor is the group's sum.
10735 /// A node that fits the requirer alone must refuse the group — placing it
10736 /// there would oversubscribe the node the moment the provider follows.
10737 #[test]
10738 fn the_capacity_floor_is_the_sum_of_the_group_not_the_requirer_alone() {
10739 let provider = minimal_spec("headscale-replicator", 1);
10740 let requirer = spec_requiring(
10741 "headscale",
10742 vec![requirement("headscale-replicator", Locality::Local)],
10743 );
10744 // Two `minimal_spec`s: 256 MiB + 250 millicores each.
10745 let inventory = declared(vec![provider]);
10746
10747 let too_small = CloudConfig {
10748 workloads: inventory.clone(),
10749 ..make_empty_cfg(vec![make_machine_with_capacity("small", 300, 4000, vec![])])
10750 };
10751 let err = too_small.admit_workload(&requirer).unwrap_err().to_string();
10752 assert!(
10753 err.contains("memory_mb>=512"),
10754 "the floor must name the group's summed demand, got: {err}"
10755 );
10756
10757 let big_enough = CloudConfig {
10758 workloads: inventory,
10759 ..make_empty_cfg(vec![make_machine_with_capacity("roomy", 512, 4000, vec![])])
10760 };
10761 assert_eq!(
10762 big_enough.admit_workload(&requirer).unwrap().name,
10763 "roomy",
10764 "a node covering the sum must still admit the group"
10765 );
10766 }
10767
10768 /// W338 §Placement consequences 2, and the reason repulsion is computed over
10769 /// a set at all: the requirer is a `Server`, so the pre-R860 axis would have
10770 /// let it onto a `no-appliance` dev Pi and dragged its Appliance provider
10771 /// there with it.
10772 #[test]
10773 fn a_server_requiring_an_appliance_locally_is_repelled_by_no_appliance() {
10774 let appliance = WorkloadSpec {
10775 archetype: Some(LifecycleArchetype::Appliance),
10776 ..minimal_spec("headscale", 1)
10777 };
10778 let requirer = spec_requiring("headscale-ui", vec![requirement("headscale", Locality::Local)]);
10779 assert_eq!(
10780 requirer.effective_archetype(),
10781 LifecycleArchetype::Server,
10782 "precondition: the requirer itself must not be an Appliance"
10783 );
10784
10785 let cfg = CloudConfig {
10786 workloads: declared(vec![appliance]),
10787 ..make_empty_cfg(vec![
10788 make_machine_with_capacity("dev-pi", 8192, 4000, vec!["no-appliance"]),
10789 make_machine_with_capacity("us-west-001", 8192, 4000, vec![]),
10790 ])
10791 };
10792
10793 assert_eq!(
10794 cfg.admit_workload(&requirer).unwrap().name,
10795 "us-west-001",
10796 "the dev Pi repels the group's Appliance member"
10797 );
10798
10799 // And with the Appliance gone from the group, the same requirer is
10800 // admissible on the same Pi — proving the repulsion came from the edge.
10801 let alone = minimal_spec("headscale-ui", 1);
10802 assert_eq!(cfg.admit_workload(&alone).unwrap().name, "dev-pi");
10803 }
10804
10805 /// The set-valued form of the per-workload drain skip the node makes in
10806 /// `drain_workloads`: one Appliance member pins the whole group.
10807 #[test]
10808 fn a_group_containing_an_appliance_is_not_drainable() {
10809 let server = minimal_spec("headscale-ui", 1);
10810 let appliance = WorkloadSpec {
10811 archetype: Some(LifecycleArchetype::Appliance),
10812 ..minimal_spec("headscale", 1)
10813 };
10814
10815 assert!(group_is_drainable(std::slice::from_ref(&server)));
10816 assert!(!group_is_drainable(&[server, appliance]));
10817 }
10818
10819 /// The regression that matters most: nothing in the tree declares
10820 /// `requires` yet, so every existing spec's group is exactly itself and its
10821 /// admission axes must be bit-identical to the pre-R860 derivation.
10822 #[test]
10823 fn a_spec_with_no_local_edges_admits_exactly_as_it_did_before() {
10824 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
10825 let req = admission_spec(&ws, &[]);
10826
10827 assert_eq!(req.mesh_tags, vec!["tag:build-worker", "arch:x86"]);
10828 assert_eq!(req.memory_mb, ws.memory_request_mb());
10829 assert_eq!(req.cpu_millis, ws.resources.cpu_millis);
10830 // R876-B7: the axis is now the complement — every repelling key EXCEPT
10831 // this spec's own class, which is the same predicate stated from the
10832 // other side. Asserted against the derivation rather than a literal so
10833 // it stays true if a fourth archetype is added.
10834 assert_eq!(
10835 req.tolerates,
10836 tolerations_excluding(&[ws.effective_archetype()])
10837 );
10838 let own = format!("no-{}", ws.effective_archetype().taint_key());
10839 assert!(
10840 !req.tolerates.contains(&own),
10841 "a spec never tolerates the taint aimed at its own class"
10842 );
10843 }
10844
10845 // ─── R860-T5 (W338 §Placement consequences 3): native-exec capability ────
10846
10847 /// A `minimal_spec` carrying the `yah.exec = native` marker — the only way
10848 /// a workload says "fork+exec me on the host" (`WorkloadSpec::
10849 /// wants_native_exec`). It stays a Container workload on the wire; the
10850 /// marker is the whole difference.
10851 fn native_spec(name: &str) -> WorkloadSpec {
10852 let mut ws = minimal_spec(name, 1);
10853 ws.annotations.insert(
10854 workload_spec::NATIVE_EXEC_ANNOTATION.to_string(),
10855 workload_spec::NATIVE_EXEC_VALUE.to_string(),
10856 );
10857 assert!(ws.wants_native_exec(), "precondition: the marker must read back");
10858 ws
10859 }
10860
10861 /// The R858 failure, now caught at placement instead of at dispatch: a node
10862 /// whose kamaji has no `--native-exec-dir` accepted the election and then
10863 /// refused the deploy, and nothing upstream could see it coming.
10864 #[test]
10865 fn a_node_without_the_native_exec_capability_cannot_host_a_native_workload() {
10866 let native = native_spec("headscale");
10867
10868 let incapable = make_empty_cfg(vec![make_machine("us-south-001", vec![])]);
10869 let err = incapable.admit_workload(&native).unwrap_err().to_string();
10870 assert!(
10871 err.contains(NATIVE_EXEC_MESH_TAG),
10872 "the refusal must name the missing capability, got: {err}"
10873 );
10874
10875 let capable = make_empty_cfg(vec![
10876 make_machine("us-south-001", vec![]),
10877 make_machine("us-west-001", vec![NATIVE_EXEC_MESH_TAG]),
10878 ]);
10879 assert_eq!(
10880 capable.admit_workload(&native).unwrap().name,
10881 "us-west-001",
10882 "a node declaring the capability admits the native workload"
10883 );
10884 }
10885
10886 /// W338's actual sentence: a `supply = "self"` spec "must be placeable where
10887 /// its requirer lands". The requirer here is an ordinary container workload
10888 /// — it is the *provider* reached by a `local` edge that needs the host
10889 /// backend, so the capability has to be required of the group, not of the
10890 /// spec being deployed.
10891 #[test]
10892 fn a_local_edge_to_a_native_provider_makes_the_requirer_need_the_capability() {
10893 let requirer = spec_requiring(
10894 "headscale-ui",
10895 vec![requirement("headscale", Locality::Local)],
10896 );
10897 assert!(
10898 !requirer.wants_native_exec(),
10899 "precondition: the requirer itself is an ordinary container workload"
10900 );
10901
10902 let cfg = CloudConfig {
10903 workloads: declared(vec![native_spec("headscale")]),
10904 ..make_empty_cfg(vec![
10905 make_machine("plain", vec![]),
10906 make_machine("us-west-001", vec![NATIVE_EXEC_MESH_TAG]),
10907 ])
10908 };
10909
10910 assert_eq!(
10911 cfg.admit_workload(&requirer).unwrap().name,
10912 "us-west-001",
10913 "the group's native member pulls the requirer onto a capable node"
10914 );
10915
10916 // Without the edge the same requirer is admissible on the plain node,
10917 // so the constraint provably came from the group and not from the spec.
10918 let alone = minimal_spec("headscale-ui", 1);
10919 assert_eq!(cfg.admit_workload(&alone).unwrap().name, "plain");
10920 }
10921
10922 /// The regression guard: nothing in the tree is native-marked today, so
10923 /// every existing spec's axes must be untouched by this ticket.
10924 #[test]
10925 fn a_group_with_no_native_member_does_not_require_the_capability() {
10926 let inventory = declared(vec![minimal_spec("headscale-replicator", 1)]);
10927 let requirer = spec_requiring(
10928 "headscale",
10929 vec![requirement("headscale-replicator", Locality::Local)],
10930 );
10931
10932 let req = admission_spec(&requirer, &inventory);
10933 assert!(
10934 !req.mesh_tags.iter().any(|t| t == NATIVE_EXEC_MESH_TAG),
10935 "no native member ⇒ no capability axis, got: {:?}",
10936 req.mesh_tags
10937 );
10938
10939 // And it still lands on a node that declares nothing at all.
10940 let cfg = CloudConfig {
10941 workloads: inventory,
10942 ..make_empty_cfg(vec![make_machine("plain", vec![])])
10943 };
10944 assert_eq!(cfg.admit_workload(&requirer).unwrap().name, "plain");
10945 }
10946}