cloud/config.rs
1//! @yah:ticket(R040-F16, "pg-on-mesh service recipe: bind tailscale0 + pg_hba.conf snippet + ufw rules")
2//! @yah:at(2026-05-05T00:32:34Z)
3//! @yah:assignee(agent:claude)
4//! @yah:status(review)
5//! @yah:parent(R040)
6//! @yah:handoff("Companion to R040-F15. Inter-node TCP (Postgres primary↔replica, NATS clusters, anything raw-protocol) lives on the Headscale mesh, not on Hetzner public IPs. Each node has a stable 100.64.x.x mesh IP that survives replacement of the underlying box, so DNS / config / pg_hba never churn when a CPX-11 is rebuilt. WireGuard already encrypts the wire — TLS becomes defense-in-depth, not load-bearing. This ticket carries the concrete pg-shaped recipe so the first stateful service deploy doesn't have to re-derive the pattern; subsequent services (redis, NATS, etc.) cargo-cult from it.")
7//! @yah:next("ServiceConfig gains a `bind_interface: Option<String>` field (e.g. `Some(\"tailscale0\")` for mesh-only services). The cloud-init/podman compose renderer translates this into either `--network host` + `pg listen_addresses = '<mesh-ip>'` OR a podman macvlan/host-binding pattern that achieves the same.")
8//! @yah:next("Generated pg_hba.conf snippet: allow the mesh subnet (100.64.0.0/10) for replication + app users. Postgres binds to the node's tailscale0 mesh IP only — `listen_addresses` is templated from the node's `tailscale ip --4` at first boot.")
9//! @yah:next("Generated ufw rules: `ufw allow in on tailscale0 to any port 5432; ufw deny 5432` — mirrors the existing yah-yubaba 7443 pattern in mirror.yml. Same shape works for any mesh-only port.")
10//! @yah:next("Replica connection string uses primary's mesh IP, NOT its public IP. Stable across box replacement.")
11//! @yah:next("Out of scope: pg_basebackup orchestration, failover, WAL archiving — those belong in noisetable's domain; this ticket only standardizes the binding/firewall/auth shape so noisetable's pg deployment doesn't reinvent it.")
12//!
13//!
14//! @yah:ticket(R323-F9, "Add sync-wave ordering to ServiceComponent (deploy-panel wave order)")
15//! @yah:assignee(agent:claude)
16//! @yah:at(2026-05-26T15:20:25Z)
17//! @yah:status(review)
18//! @yah:phase(P2)
19//! @yah:parent(R323)
20//! @yah:next("ServiceComponent gains a wave/order field (or depends_on between components) so the deploy panel (R323-F4) can group workload rollout rows into sync waves (wave 0 parallel, wait healthy, wave 1, …). Today all components are implicitly wave 0.")
21//! @yah:next("compute_service/compute_cell in reconciler/sync_status.rs surface the wave per workload so F4 doesn't re-derive it.")
22//! @yah:gotcha("Until this lands, F4 should render every workload as wave 0 (no ordering).")
23//! @yah:handoff("Added wave: u32 (serde default=0, skip_serializing_if zero) to ServiceComponent in config.rs. Added is_zero_u32 helper. Fixed the three struct literal call-sites that now need wave: 0 (config.rs test, local_sim.rs x2, mesofact_static.rs). Added wave?: number to the TS ServiceComponent interface with a doc comment. Deploy panel now reads c.wave ?? 0 for each WorkloadRow instead of hardcoded 0. SyncFooter computes maxWave from the components array and renders 'wave 0' (all-zero case) or 'waves 0–N' (multi-wave). All 218 cloud lib tests pass; bun run typecheck clean.")
24//! @yah:verify("cargo test -p cloud --lib # 218 passed")
25//! @yah:verify("cd packages/yah/ui && bun run typecheck # no new errors")
26//! @yah:verify("In service.toml: add wave = 1 to a component, rebuild, open the deploy panel — that workload row shows 'w1' badge; SyncFooter shows 'waves 0–1'")
27//! @yah:verify("Component with no wave field in TOML deserializes as wave=0 (default). Saving a wave=0 component omits the field from the output TOML (skip_serializing_if).")
28//!
29//! @arch:see(.yah/docs/working/W142-pond.md)
30//!
31//! @yah:relay(R615, "Linked infra sources: sources.toml overlay so a camp can borrow another camp's substrate")
32//! @yah:at(2026-07-20T18:18:05Z)
33//! @yah:status(open)
34//! @arch:see(.yah/docs/working/W274-linked-infra-sources.md)
35//!
36//! @yah:ticket(R615-F1, "InfraSource types + SourcesConfig::load(infra_dir) parsing .yah/infra/sources.toml")
37//! @yah:status(review)
38//! @yah:assignee(agent:bundle-anthropic-miravel)
39//! @yah:at(2026-08-08T19:55:57Z)
40//! @yah:phase(P1)
41//! @yah:parent(R615)
42//! @yah:next("Add InfraSourceKind { Path { path }, Git(GitSource) } + InfraSource { owner, kind, mode, select } to cloud/src/config.rs. Reuse the existing GitSource (config.rs:1205, { repo, ref, subdir }) verbatim — do not invent a second git-source shape.")
43//! @yah:next("SourcesConfig::load(infra_dir) reads .yah/infra/sources.toml (schema_version = 1, ordered [[source]] array). Absent file = empty list, never an error — every existing camp has no sources.toml.")
44//! @yah:next("mode is the write-gate: read-only (borrower cannot mutate) vs owner-manages. Model it as an enum, not a bool, so a future read-write-with-approval tier is additive.")
45//! @yah:verify("cargo check -p cloud && cargo test -p cloud")
46//! @arch:see(.yah/docs/working/W274-linked-infra-sources.md)
47//! @yah:tier(Cleric)
48//! @yah:handoff("InfraSourceKind{Path{path},Git(GitSource)} + SourceMode{ReadOnly,Manage} + InfraSource{owner,kind,mode,select} + SourcesConfig{schema_version,source} all landed in oss/yubaba/crates/cloud/src/config.rs (after default_git_ref, ~line 1550). GitSource reused verbatim -- Git(GitSource) wraps the existing R561 type unchanged, no second git-source shape. InfraSourceKind is internally tagged (#[serde(tag=\"kind\", rename_all=\"kebab-case\")]) and flattened into InfraSource so a [[source]] table reads exactly like W274's example: owner/kind/path-or-repo+ref+subdir/mode/select all at one table level. mode: SourceMode defaults ReadOnly via #[serde(default)] on the field (enum, not bool, per the ticket's own instruction -- Manage is the explicit escape hatch). SourcesConfig::load(infra_dir) returns Ok(default()) -- schema_version=1, empty source list -- when sources.toml is absent; only parses+errors when the file exists and is malformed.")
49//! @yah:handoff("Tree anchor 85801e7f. Pathspec: oss/yubaba/crates/cloud/src/config.rs (only file touched). Tests: cargo test -p yah-cloud --lib (from oss/yubaba) 710 passed / 0 failed / 4 ignored, +6 new over the 704 baseline your R707-T6 verification recorded (sources_load_is_empty_when_the_file_is_absent, sources_parses_a_path_kind_exactly_like_w274s_example, sources_parses_a_git_kind_reusing_gitsource_verbatim, sources_mode_defaults_to_read_only_and_manage_is_explicit, sources_preserves_declaration_order, sources_round_trips_through_serialize). cargo check -p cloud also green (implied by the test build).")
50//! @yah:handoff("Tree anchor at handoff: 85801e7f6b76b369c0c8ecd2e5c7874990cd9286 — the shared tree as I left it. Diff against it (`git diff 85801e7f6b76b369c0c8ecd2e5c7874990cd9286..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
51//! @yah:next("R615-F2 picks this straight up: overlay these sources into CloudConfig::load, tagging origin{owner,source} and merging camp-local-wins-on-collision.")
52//! @yah:handoff("Verified pre-existing work: InfraSourceKind{Path,Git(GitSource)} + SourceMode + InfraSource + SourcesConfig all present in oss/yubaba/crates/cloud/src/config.rs at tree anchor 871fde1c, matching the inline @yah:handoff notes already on this ticket. GitSource reused verbatim, no second git-source shape. This session added no new code -- only ran verification and closed the board state, which a prior session left stuck in `open` despite the work being done (code + handoff notes landed, but board.review/handoff was never called).")
53//! @yah:verify("cargo check -p yah-cloud -- clean (2 pre-existing unrelated warnings)")
54//! @yah:verify("cargo test -p yah-cloud --lib -- 723 passed; 0 failed; 4 ignored (from oss/yubaba)")
55//!
56//! @yah:ticket(R615-F2, "Overlay loader: resolve sources in CloudConfig::load, tag origin, camp-local wins on collision")
57//! @yah:status(review)
58//! @yah:assignee(agent:bundle-anthropic-miravel)
59//! @yah:at(2026-08-08T19:56:05Z)
60//! @yah:phase(P1)
61//! @yah:parent(R615)
62//! @yah:next("In CloudConfig::load, after loading camp-local machines/providers/rules, resolve each source to an infra root (git sources read from the .yah/cache/infra/ sync cache — load stays offline), load that root's machines/providers/rules, tag each entry with origin { owner, source }, and overlay UNDER camp-local. Camp-local wins on name collision.")
63//! @yah:next("The machine load site is config.rs:533 (load_dir::<MachineConfig>(paths::machines_dir(...))). Note config.rs:575 load_from_config_dir is a SECOND machine load site that deliberately skips the inherit_machines redirect for multi-root/sibling trees (W206) — decide explicitly whether sources overlay applies there too, and document the answer either way.")
64//! @yah:verify("cargo check -p cloud && cargo test -p cloud")
65//! @yah:verify("A camp with sources.toml [[source]] kind=path to a sibling camp sees that camp's machines in CloudConfig::load, each tagged with the source owner")
66//! @yah:gotcha("Cross-camp MachineConfig schema skew is real: noisetable ships an older machine schema (location/server_type/hosts_mirrors) while yah's use region/arch/[connect]. A borrowed source can carry fields the borrower's binary predates. Overlay load MUST tolerate/skip unparseable foreign entries per-file and warn — never fail the whole load.")
67//! @arch:see(.yah/docs/working/W274-linked-infra-sources.md)
68//! @yah:depends_on(R615-F1)
69//! @yah:tier(Warrior)
70//! @yah:handoff("Overlay landed in CloudConfig::load (oss/yubaba/crates/cloud/src/config.rs). After camp-local machines/providers/legacy-merge finish, SourcesConfig::load(paths::infra_dir(workspace_root)) resolves + overlay_infra_sources() merges each source's machines/providers UNDER what's already there -- camp-local wins any name collision, and among sources themselves the earlier-declared one wins (both proven by dedicated tests). Provenance is NOT a field on MachineConfig/ProviderConfig: added CloudConfig.machine_origins/provider_origins: BTreeMap<String, InfraOrigin> instead, keyed by name/id. Reason recorded in a doc comment on InfraOrigin -- MachineConfig/ProviderConfig are constructed by struct literal in test helpers across several crates (including crates/yah/agent-tools/src/cloud_tools.rs, which is fenced/live-owned this session), so widening either shape would have forced an edit there for zero semantic gain; origin is a property of the LOAD, not the machine.")
71//! @yah:handoff("GOTCHA closed: added load_dir_tolerant<T>() -- a per-file-tolerant sibling of the existing (strict) load_dir -- so one unparseable foreign machine/provider (schema skew) skips-with-a-tracing::warn! and never sinks the rest of that source's directory or this camp's own load. Proven by one_unparseable_foreign_machine_does_not_sink_the_rest_of_the_directory_or_the_load. load_dir itself is untouched -- camp-local files still hard-fail on a bad TOML, which is correct, only borrowed roots get the tolerant path.")
72//! @yah:handoff("Git sources: InfraSource::infra_root() resolves kind=path to <workspace_root>/<path>/.yah/infra (live tree, no I/O beyond building the path) and kind=git to paths::infra_source_cache_dir(workspace_root, owner)/infra -- a NEW path helper in paths.rs, also what R615-T3's `yah infra sync` target directory must be so the two line up. An unsynced git source (cache dir absent) overlays nothing and is explicitly NOT an error (test: an_unsynced_git_source_overlays_nothing_and_is_not_an_error) -- load() stays fully offline as W274 §3 requires.")
73//! @yah:handoff("select filtering implemented for machines only (name exact-match or literal mesh_tags membership -- not a glob engine, matches W274's own example verbatim) via machine_matches_select(); does NOT apply to providers -- documented as a deliberate choice, nothing in W274 or the ticket describes a provider-scoped filter.")
74//! @yah:handoff("EXPLICIT DECISION on the config.rs:575-equivalent gotcha (now load_from_config_dir): sources overlay does NOT apply there. Multi-root sibling config dirs (W206 layout (b)) are a second config root INSIDE the same camp, not a second camp -- .yah/infra/sources.toml is tied to paths::infra_dir(workspace_root) specifically, which has no well-defined meaning for an arbitrary config_dir. Documented in the function's doc comment and proven by load_from_config_dir_never_applies_sources_overlay (a sources.toml at the real workspace root does NOT leak into a load_from_config_dir call against a sibling .noisetable/ dir under that same root).")
75//! @yah:handoff("Tree anchor 85801e7f. Pathspec: oss/yubaba/crates/cloud/src/config.rs, oss/yubaba/crates/cloud/src/paths.rs (added infra_source_cache_dir + 1 test), oss/yubaba/crates/cloud/src/reconciler/mesofact_bundle.rs (CloudConfig test-literal fixed for the 2 new fields), app/yah/cli/src/cloud.rs (3 CloudConfig test-literal sites fixed, same reason). Tests: cargo test -p yah-cloud --lib (from oss/yubaba) 720 passed / 0 failed / 4 ignored, +10 over R615-F1's 710 baseline (9 overlay tests in config.rs + 1 in paths.rs). cargo build -p yah --lib (repo root) green -- confirms nothing downstream (agent-tools, cloud.rs, hub) broke from CloudConfig's two new fields.")
76//! @yah:handoff("Tree anchor at handoff: 85801e7f6b76b369c0c8ecd2e5c7874990cd9286 — the shared tree as I left it. Diff against it (`git diff 85801e7f6b76b369c0c8ecd2e5c7874990cd9286..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
77//! @yah:next("R615-T3 (yah infra sync) is unblocked and has everything it needs: paths::infra_source_cache_dir(workspace_root, owner) is the exact target directory to clone/pull git sources into, already matching what F2's overlay reads from.")
78//! @yah:next("R615-F4 (Infra tab origin badge, not in my assigned lane) can read CloudConfig.machine_origins/provider_origins directly -- no further backend plumbing needed for the badge itself.")
79//! @yah:handoff("Verified pre-existing work: overlay landed in CloudConfig::load (oss/yubaba/crates/cloud/src/config.rs) at tree anchor 871fde1c -- SourcesConfig::load resolves sources, overlay_infra_sources() merges under camp-local with camp-local-wins and earlier-source-wins collision rules, machine_origins/provider_origins BTreeMaps added to CloudConfig, load_dir_tolerant() added for per-file-tolerant foreign schema skew, InfraSource::infra_root() resolves path/git kinds, load_from_config_dir explicitly does NOT get the overlay (documented). Matches this ticket's own inline @yah:handoff notes. This session added no new code -- only ran verification and closed board state that a prior session left stuck in `open` despite the work being done.")
80//! @yah:verify("cargo check -p yah-cloud -- clean (2 pre-existing unrelated warnings)")
81//! @yah:verify("cargo test -p yah-cloud --lib -- 723 passed; 0 failed; 4 ignored (from oss/yubaba), includes overlay tests + load_dir_tolerant test + infra_source_cache_dir test in paths.rs")
82//!
83//! @yah:ticket(R605-F12, "Sovereign groups have no voting axis, so non-voting membership is inexpressible and the raft guard is enforced by an absent field")
84//! @yah:status(review)
85//! @yah:at(2026-08-20T05:15:30Z)
86//! @yah:assignee(agent:bundle-anthropic-ashguard)
87//! @yah:parent(R605)
88//! @arch:see(.yah/docs/working/W325-isolated-x86-build-capacity.md)
89//! @yah:next("OPERATOR INTENT (2026-08-19) that the model cannot currently record: us-west-003 is a NON-VOTING member of the us-west-001-based (prod) sovereign group, and us-west-011 is a DIFFERENT sovereign (dev) from 001/003. The dev/prod split is already declared correctly. The non-voting membership is not — us-west-003.toml declares no sovereign_group at all.")
90//! @yah:next("THE GAP: MachineConfig::sovereign_group is a single Option<String>, so membership is binary, and judge_join (oss/yubaba/crates/cloud/src/config.rs:459) permits a join IFF both sides declare the same non-None group. There is no way to say 'in this blast radius, but not quorum-eligible'.")
91//! @yah:next("WHY THAT IS ACTIVELY BAD, not just missing: today the ONLY thing refusing us-west-003 into the prod raft at the join gate is its ABSENT stamp. Its own file is emphatic it must never hold a raft node id ('a home-internet partition should never be able to stall the raft'), and that guarantee currently rests on a field nobody wrote. Stamping it prod to record the operator's real intent would REMOVE the guard. This is precisely the W305 failure mode that produced R742-T4: `no-voter` sat inert on three nodes asserting something nothing enforced.")
92//! @yah:next("PROPOSED SHAPE (recommended): a second axis, e.g. sovereign_role = voter | non-voter (default voter for back-compat, or make it required), with judge_join permitting a same-group join only for voters. Then us-west-003 stamps prod + non-voter, the intent is machine-readable, and the raft guard stops depending on omission. us-west-004 (R605-T7) would take the same shape.")
93//! @yah:next("TOUCHES TWO COPIES OF THE PREDICATE, do not fix only one: cloud::judge_join renders the camp-side refusal, but the predicate itself lives in workload_spec::sovereign::join_permitted because yubaba's POST /raft/add-learner gate asks the same question and there is deliberately no yubaba -> cloud edge. Also re-read `yubaba serve --sovereign-group`, whose node-side gate is narrower on purpose (an unset flag means 'declared nothing', not 'declared standalone').")
94//! @yah:gotcha("THE CODE AND THE OPERATOR CURRENTLY DISAGREE ABOUT 003, and a reader should know which is which before editing. judge_join's own doc comment asserts 'prod and dev are both stamped, and us-west-002/003/015 are deliberately not raft members' — i.e. R742-F1 modelled 003 as STANDALONE. The operator's model is that it is a NON-VOTING MEMBER of prod. Those are different claims, not a wording difference: standalone means no blast-radius relationship to 001 at all. Do not silently 'correct' either side; this ticket is the reconciliation.")
95//! @yah:gotcha("FLEET STATE AS DECLARED (2026-08-19): prod = us-west-001, us-south-001, us-east-001. dev = us-west-011, us-west-013, us-west-014. NO sovereign_group declared = us-west-002, us-west-003, us-west-015. Verify against the files rather than trusting this list — xtask/tests/fleet_sovereign_groups.rs pins the roster and will need updating in the same change (it also asserts the stamp parses as a TOP-LEVEL key, which matters because 003 has a long comment block before [allocatable] where a stamp would silently become a member of that table).")
96//! @yah:gotcha("SEPARATE AXIS, DO NOT ENTANGLE: mesh membership is not sovereign membership. The standing rule is ONE mesh for the entire fleet regardless of group (operator, 2026-08-19), so us-west-003 and us-west-011 enrolling in headscale is unrelated work with no design question in it — see R605-T10. A voting axis on sovereign_group must not become a reason to keep any node off the mesh.")
97//! @yah:gotcha("SHARED-TREE COLLISION, live 2026-08-20: R772 (Miravel:spade, session:ce6d74a9) is refactoring oss/yubaba/crates/cloud/src/validate.rs at the same time and the file is currently RED - error[E0425] cannot find function load_machines at validate.rs:753, a half-landed extraction of the machine-loading walk that check_inert_taints / check_retired_arch_tags / the new check_unroled_sovereign_members all duplicate. That error is NOT from this ticket. Told them by party.chat and asked them to absorb check_unroled_sovereign_members into load_machines rather than leave one holdout. Do not hand-fight the file.")
98//! @yah:gotcha("R772 ALSO BROKE THREE PRE-EXISTING INGRESS TESTS, again not this ticket: two_services_fronting_one_node_collate_into_one_front_door, a_cross_service_hostname_clash_is_reported_with_both_declarations, one_mirrors_broken_declaration_does_not_hide_the_rest - all failing with 'providers.compute.use = hetzner - no such provider'. Cause is their new CloudConfig::load(workspace_root) at validate.rs:750 inside collate_workspace_ingress; the fronted_mirror fixture declares the slot but never writes infra/providers/hetzner.toml, and CloudConfig::load runs cross_ref_validate. Left alone deliberately - peer-owned.")
99//! @yah:gotcha("TRAP THAT MADE THREE OF MY OWN TESTS PASS FOR THE WRONG REASON: the machine-lint sweeps SKIP unparseable TOMLs by design (a peer's half-written scaffold must not sink the sweep). So a test fixture missing a REQUIRED MachineConfig field - mesh_tags is the one that bites - is silently skipped, the lint finds nothing, and every assert-empty test passes vacuously. Only the one test asserting found.len() == 1 noticed. write_sovereign_machine now always writes mesh_tags = [] and carries a comment saying why. Check this before trusting any new test in cloud::validate.")
100//! @yah:verify("cargo test -p yah-workload-spec --lib sovereign (from oss/yah-base) -- 9 passed, 0 failed. Covers both new refusals (a_non_voting_member_does_not_join_its_own_group, a_non_voting_target_has_no_quorum_to_join), the back-compat pin (the_default_role_is_the_pre_r605_f12_meaning), and the one-spelling round-trip across TOML/CLI/JSON.")
101//! @yah:verify("cargo test -p yubaba --lib sovereign (from oss/yubaba) -- 13 passed, 0 failed. Includes a_non_voting_joiner_is_refused_by_role_not_by_group, a_non_voting_target_refuses_every_joiner, a_group_without_a_role_key_is_a_voter_not_a_refusal (the deployed-fleet back-compat seam), a_peer_reports_its_role_in_the_toml_spelling.")
102//! @yah:verify("cargo test -p yubaba --test raft_sovereign_group (from oss/yubaba) -- 11 passed, 0 failed, up from 8. Three new end-to-end against real single-node rafts: a_non_voting_member_of_the_same_group_is_refused, a_non_voting_leader_refuses_to_grow_its_quorum, a_node_publishes_its_role_and_the_leader_reads_it_there (which also proves the request body cannot vote a non-voter in - the leader dials the joiner).")
103//! @yah:verify("cargo test -p xtask --test fleet_sovereign_groups (from repo root) -- 2 passed, 0 failed. THE DECISIVE ONE: parses the real .yah/infra/machines/*.toml through the actual MachineConfig deserializer. Confirms us-west-003 = prod + non-voter on disk, all six pre-existing voters now stamped sovereign_role = voter explicitly, and neither key swallowed by a table header.")
104//! @yah:verify("cargo test -p yah-cloud --lib (from oss/yubaba) -- 891 passed, 3 failed, where all 3 failures were R772's ingress-collate tests and none were mine. A clean re-run is BLOCKED, not failing: R555's in-flight AdmissionGrant.secrets field breaks velveteen-exec, and yah-cloud is not a root workspace member so its dev-deps can only resolve from the oss/yubaba workspace. Re-run once R555 lands.")
105//! @yah:handoff("LANDED, operator chose the second-axis shape (Call 1 = A, 2026-08-20). sovereign_role = voter | non-voter now sits beside sovereign_group, and ONE predicate judges both: workload_spec::sovereign::join_permitted(Membership, Membership) where Membership { group: Option<&str>, role: SovereignRole }. Permitted iff same non-None group AND both sides Voter. Both copies of the predicate call it - cloud::judge_join (camp-side) and yubaba::sovereign_group::judge (node-side) - so the rule itself cannot drift; only the prose differs, which was already the R742-F1 split.")
106//! @yah:handoff("WHY THE ROLE IS CHECKED ON BOTH SIDES, since only the joiner half was asked for: a join grows a quorum and it takes two nodes. Refusing a non-voting JOINER is the us-west-003 case. Refusing a non-voting TARGET is the same assertion read from the other end - a box declared non-voting that is serving add-learner is already holding a raft seat its own declaration forbids, and permitting there would paper over the contradiction. Both refusals name the role rather than the group when the groups match, because a message reading 'cross-group join refused: prod and prod' reads as a bug in the check.")
107//! @yah:handoff("THE DEFAULT IS THE LOAD-BEARING DECISION AND IT IS DELIBERATELY PERMISSIVE. An absent sovereign_role resolves to Voter (MachineConfig::sovereign_membership, the ONE place the Option is resolved). Reason: before this field, declaring a group WAS declaring quorum eligibility, so absence has to keep meaning that or the change silently retires six live voters. The permissiveness is bounded at the other end by cloud::validate::check_unroled_sovereign_members, which makes `yah cloud validate` FAIL on a group stamp with no role beside it - so the default can be reached by choice but not by silence. MachineConfig::sovereign_role stays Option<SovereignRole> (not a defaulted plain field) precisely so that lint can tell 'chose voter' from 'never considered it'.")
108//! @yah:handoff("NODE-SIDE BACK-COMPAT SEAM, pinned by a test because it is a decision and not an oversight: a peer answering GET /raft/status with a sovereign_group but NO sovereign_role key - every yubaba built between R742-F1 and R605-F12, which today is the entire prod raft - is read as Voter, not refused. Refusing would freeze a stamped cluster's growth until every member was rolled, strictly worse than what the role guards against, and it is the same degrade-toward-prior-behaviour stance the module already took for the group. Residue, named rather than hidden in read_group's doc: a box whose machine.toml says non-voter but whose daemon predates the flag answers 'voter' and the node gate admits it. judge_join refuses it camp-side, which is where operator-driven joins go. Window closes per-group as its nodes carry the flag.")
109//! @yah:handoff("FILES: workload-spec/src/sovereign.rs (SovereignRole + Membership + role-aware join_permitted, +227). cloud/src/config.rs (sovereign_role field, sovereign_membership(), judge_join same-group role branch, SovereignRole re-exported from cloud::config). cloud/src/validate.rs (check_unroled_sovereign_members + UnroledSovereignMember). app/yah/cli/src/cloud.rs (lint wired: ERROR in `yah cloud validate`, WARNING in the apply preflight - same split as inert-taint/retired-arch-tag, because an unwritten role changes no placement decision and the machine may be declared in a tree this camp does not own). yubaba/src/{sovereign_group,lib,main}.rs (--sovereign-role flag, ServerState.sovereign_role, /raft/status publishes it always-never-null, gate both directions). yubaba-test-harness/src/solo_node.rs (solo_node_with_sovereign_role). .yah/infra/machines/*.toml (7 files). xtask/tests/fleet_sovereign_groups.rs + fleet_build_placement.rs. W325 section 3d.")
110//! @yah:handoff("ONE BEHAVIOUR CHANGE WORTH A SECOND OPINION: a node started with --sovereign-role non-voter AND a --raft-node-id now refuses EVERY add-learner. I judged that correct - it is a contradiction the operator should see loudly - but the symptom is 'joins mysteriously stop working' rather than a startup refusal. main.rs warns loudly at boot when that pair is present; I did NOT make it fatal, because refusing to start could brick a node mid-roll. Reconsider if it bites.")
111//! @yah:handoff("NOT DONE, and it is a HARD GATE: .yah/schema/machine.toml.schema.json has NOT been regenerated, so sovereign_role is absent from it and schema-drift-guard (scripts/check-schema-drift.sh, a step in .yah/qed/check.toml, run by CI on every push) WILL FAIL. Fix is `cargo run -p xtask -- emit-schemas` from the repo root - it was queued behind ~7 concurrent peer cargo builds for the whole session. Nothing else is required to make this pushable.")
112//! @yah:handoff("ALSO NOT RE-CONFIRMED: `cargo test -p yah-cloud --lib` needs a clean run. Its last real run was 891 passed / 3 failed with all three failures belonging to R772's ingress-collate work and none to this ticket. The re-run is BLOCKED not failing - R555's in-flight AdmissionGrant.secrets field breaks velveteen-exec, and yah-cloud is not a root workspace member so its dev-deps only resolve from the oss/yubaba workspace where that break lives. Re-run from oss/yubaba once R555 lands.")
113//! @yah:verify("cargo run -p xtask -- emit-schemas (from repo root) -- wrote 8 files, exit 0 after an 18m24s build queued behind ~7 concurrent peer cargo jobs. .yah/schema/machine.toml.schema.json now carries the sovereign_role property (anyOf SovereignRole | null, with the full doc comment) and the SovereignRole definition as a oneOf over the two string enums voter / non-voter. The schema-drift-guard gate for THIS ticket is closed.")
114//! @yah:gotcha("emit-schemas IS ALL-OR-NOTHING AND WILL PICK UP A PEER'S UNCOMMITTED WORK. Running it to close this ticket's machine-schema drift also regenerated .yah/schema/secret.toml.schema.json (+34) from R555-F5's in-flight SecretAccess::Recipes / RecipeMatch source. That output is CORRECT for the tree as it stands and was not hand-edited, but it means the schema diff in the working tree is not purely R605-F12's: machine.toml.schema.json (+32) is this ticket, secret.toml.schema.json (+34) is R555. Told Ashguard:spade by party.chat so they carry it with their commit rather than regenerating on top. Anyone splitting these commits needs to split the schema diff too.")
115//! @yah:handoff("ALL GATES CLOSED as of 2026-08-20. Both items listed as outstanding in the earlier handoff notes are done: emit-schemas ran (machine.toml.schema.json carries sovereign_role + the SovereignRole voter/non-voter enum, drift guard satisfied), and cargo test -p yah-cloud --lib is 896 passed / 0 failed once R555 and R772 settled. 45 tests green across workload-spec (9), yubaba lib (13), yubaba raft integration (11), yah-cloud lib (10 of this ticket's, within 896), xtask fleet (2). Ready for review. NOTE for whoever commits: the working tree's schema diff is not purely this ticket - .yah/schema/machine.toml.schema.json (+32) is R605-F12, .yah/schema/secret.toml.schema.json (+34) is R555-F5, both correct generated output from one emit-schemas run. Ashguard:spade has agreed to carry theirs.")
116//! @yah:verify("cargo test -p yah-cloud --lib (from oss/yubaba) -- 896 passed, 0 FAILED, 4 ignored. The blocked check from earlier is now clean: R555 landed the velveteen-exec and TransformRecipe.secrets fixes, R772's ingress-collate work settled (they replaced the CloudConfig::load in collate_workspace_ingress with a narrower machines-only loader, so cross_ref_validate can no longer fail the collate over an unrelated provider typo). All 45 R605-F12 tests across the four crates are green simultaneously on one tree.")
117//! @yah:verify("Confirmed by NAME rather than by total, since a passing count proves nothing about which tests ran: cargo test -p yah-cloud --lib -- role voter voting lists all ten of this ticket's cloud tests green - a_non_voting_member_is_refused_into_its_own_group, a_non_voting_target_has_no_quorum_to_grow, a_refusal_names_the_group_when_fixing_the_role_would_not_help, an_unwritten_role_still_joins_its_group, a_non_voter_is_still_in_the_group_it_names, sovereign_role_round_trips_and_is_omitted_when_unwritten, a_group_with_no_role_is_reported_with_the_declaring_file, either_stated_role_is_clean, a_machine_in_no_group_is_not_asked_for_a_role, unroled_findings_are_ordered_by_file_so_output_is_stable.")
118//!
119//! @yah:ticket(R876-B7, "Node taints are structurally inert for mirror-declared placements: you cannot drain a node, and it fails silently")
120//! @yah:at(2026-09-09T09:05:55Z)
121//! @yah:status(review)
122//! @yah:assignee(agent:bundle-anthropic-ashguard)
123//! @yah:parent(R876)
124//! @yah:severity(high)
125//! @yah:next("SECOND HALF, and it is what makes the relay's headline question answerable: a working taint must produce a MOVE, not a refusal. Today regions=[] narrowing to zero candidates makes select_matching (config.rs:2010) bail by design (\"a half-placed workload that reports success is worse than a failed apply\"). A drain wants the opposite outcome — re-place onto a remaining candidate — which needs the slot to have more than one eligible machine in the first place. Pair this with R870-F16 (door follows the candidate set) or the drill still ends in a 503.")
126//! @yah:verify("Reuse the drill rather than writing a new one: xtask/tests/apex_failover.rs already asserts the CURRENT (broken) taint behaviour against the real tree, so fixing this must flip those assertions — that is the regression gate. Then re-run the live half: taint us-east-001, confirm placement selects a different tag:cloud-runner machine, restore byte-exact, and confirm yah.dev stays 200 throughout.")
127//! @yah:gotcha("IT FAILS SILENTLY, WHICH IS THE SHARP EDGE. \"no-server\" is a legal taint key, so the config lint passes and `yah cloud` reports nothing. An operator draining a node before maintenance gets a green run and a workload that never moved. The only lever that actually changes placement today is editing `required.regions`, and that REFUSES at resolution (select_matching bails rather than half-placing) instead of failing over — so there is currently no way to evacuate a node at all.")
128//! @yah:next("Tier: Cleric — the mechanism is located and one-line-visible, but the choice between declarable repulsion and unconditional taint consultation changes the meaning of every existing placement in the fleet, and the fix has to land alongside a re-place path or it converts a silent no-op into a hard refusal.")
129//! @yah:gotcha("MEASURED, NOT INFERRED — R876-S2's drill, 2026-09-09. `taints = [\"public-ip\", \"no-server\"]` was written onto the REAL .yah/infra/machines/us-east-001.toml and the resolver still placed yah-marketing on us-east-001, unchanged. Restored byte-exact (diff empty, sha256 back to 17dd15e2..., git clean against blob d66ab6d8); yah.dev stayed 200 throughout and no mutating apply was run.")
130//! @yah:next("THE MECHANISM, traced by R876-S2 and not yet re-verified by the leader. Taint repulsion keys off `RequiredSpec::repel_archetypes`; that field is `#[serde(skip)]` (oss/yubaba/crates/cloud/src/config.rs:4067), so a slot declared in a mirror's `required = {...}` ALWAYS deserializes with it empty. `matches` (config.rs:4182) consequently never reads `machine.taints` at all. Confirm both line anchors before editing — the shared tree moves.")
131//! @yah:handoff("SEMANTICS LANDED — repel-by-default + declarable toleration. `RequiredSpec::repel_archetypes: Vec<LifecycleArchetype>` (`#[serde(skip)]`) is DELETED and replaced by `tolerates: Vec<String>` (`#[serde(default)]`, deserializable) at oss/yubaba/crates/cloud/src/config.rs:4319. `matches` (config.rs:4397) no longer iterates a field of `self`: it walks `machine.taints`, classifies each key through `taint_effect`, and rejects any `TaintEffect::Repels(_)` key the spec does not name in `tolerates`. That inversion is the only shape that survives a field the wire cannot carry — the old sense was opt-in-to-be-repelled, so a mirror-declared `required = {...}` always deserialized with an empty archetype set and `machine.taints` was never read at all. Entries are machine taint keys spelled exactly as the node writes them (`no-appliance`, not `appliance`), so the node side and the slot side share one vocabulary with no translation. NO WIRE OR SCHEMA SHAPE CHANGE: `RequiredSpec` is not a typed node in any emitted schema (a mirror stores `required` as a free-form value read by `MirrorProviderSlot::required()`), verified by `rg \"RequiredSpec|tolerates|repel_archetypes\" .yah/schema/*.json` — the only hits are prose inside a doc-comment description.")
132//! @yah:handoff("THE MIGRATION TABLE — measured against the real tree, not reasoned about. FLEET TAINTS, all nine machines (`grep -rE \"^\\s*taints\\s*=\" .yah/infra/machines/*.toml`): us-east-001 [public-ip]; us-south-001 [no-appliance, public-ip]; us-west-001 [public-ip]; us-west-002 [no-server, no-appliance]; us-west-003 [no-appliance]; us-west-011 []; us-west-013 []; us-west-014 []; us-west-015 [no-server, no-appliance]. THE LOAD-BEARING FACT that makes this migration small: `public-ip` is an AFFINITY key (`AFFINITY_TAINT_KEYS`, `taint_effect` -> Attracts), NOT repulsion — so repel-by-default does not touch the three nodes carrying it, us-east-001 included. Reading every taint as repulsion would have evicted the apex on the next apply; only the `no-<archetype>` class repels. Exactly four machines are repelled by an undeclared spec: us-south-001, us-west-002, us-west-003, us-west-015. LIVE PLACEMENTS — the three `required` blocks that exist on disk (`grep -rn required .yah/services/*/mirrors/*.toml`): (1) yah-marketing providers.bundle, cloud.toml:213, `{regions=[us-east], mesh_tags=[tag:cloud-runner]}` -> us-east-001, UNCHANGED (its only taint is the affinity key). (2) yah-cloud providers.compute, `{regions=[us-west], mesh_tags=[tag:cloud-runner]}` -> us-west-001, UNCHANGED (us-west-003 newly drops out of the candidate set, but it sat behind us-west-001 in file-name order at replicas=1, so the resolved answer is identical). (3) yah-cloud-admin providers.compute, same constraint -> us-west-001, UNCHANGED. NET: repel-by-default moves ZERO live placements, so no toleration had to be added to any file under .yah/services/ or .yah/infra/ and none was. No file under .yah/infra/machines/ or .yah/services/ was written by this ticket at all.")
133//! @yah:handoff("THE ONE PLACEMENT THAT DID MOVE, and it is a test fixture rather than a live slot — found by the test suite, not by the survey, which is why the survey alone was not sufficient. `xtask/tests/mirror_ingress.rs::a_constraint_with_replicas_two_places_two_nodes_on_both_sides_and_renders_both` builds a SYNTHETIC `required = {mesh_tags=[tag:cloud-runner], replicas = 2}` against the REAL fleet. Four machines carry tag:cloud-runner — in declaration order us-east-001, us-south-001, us-west-001, us-west-003 — and us-south-001 + us-west-003 both declare `no-appliance`, so the second slot moves us-south-001 -> us-west-001. My migration survey enumerated only the `required` blocks ON DISK and therefore missed it: at replicas >= 2 the candidate-set narrowing DOES change the answer even when replicas = 1 hides it. Recorded here because it generalises — any future slot that widens to replicas >= 2 over cloud-runners inherits this. Fixed at the site that caught it (mirror_ingress.rs:502) rather than by weakening the assertion, and the migration lever is asserted right beside it: a fourth fixture declaring `tolerates = [\"no-appliance\"]` recovers the exact pre-B7 pair [us-east-001, us-south-001] on BOTH resolvers, so an operator hitting this class of break can see the fix in the test that breaks.")
134//! @yah:verify("BASELINE MEASURED BEFORE EDITING, then re-measured after. `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` = 1129 passed / 0 failed / 4 ignored, exit 0 (the run completed and printed its result line before my first Edit; a deferred W298 skew advisory later named config.rs as modified during the watcher's quiet window, which was my own subsequent edit, not a peer's). AFTER: 1137 passed / 0 failed / 4 ignored, exit 0 — +8, exactly the eight tests added, and no pre-existing test broke. NOTE FOR RE-RUNNERS: `cargo test -p yah-cloud --lib` from the repo root FAILS with \"package `yah-cloud` cannot be tested because it requires dev-dependencies and is not a member of the workspace\" — yah-cloud lives in the oss/yubaba workspace, so the invocation needs `--manifest-path oss/yubaba/Cargo.toml`. Also `cargo check --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --all-targets` exit 0 and `-p yubaba --all-targets` exit 0 (yubaba consumes cloud, so it is where the field removal would have surfaced). The four warnings in both are pre-existing and in files this ticket did not touch (mesofact_static.rs unused imports, app_manifest.rs dead field, pond_door.rs unused fn, reconciler/mod.rs non-snake-case).")
135//! @yah:verify("EIGHT NEW UNIT TESTS in config.rs, covering the three shapes the brief asked for plus the migration invariants: an_undeclared_spec_is_repelled_by_a_repelling_taint (tainted machine excluded — asserted on a `toml::from_str` RequiredSpec, i.e. the mirror path reproduced exactly, not a hand-built literal); an_explicit_toleration_admits_the_tainted_machine_again (tolerated -> included, per-key not blanket, and it deserializes); an_untainted_machine_matches_exactly_as_before; an_affinity_taint_does_not_repel (public-ip on us-east-001 — the assertion that stands between this change and an evicted apex); select_matching_drops_a_tainted_candidate_and_keeps_the_rest (the set-level predicate: tainting candidate 1 moves the placement to candidate 2, and asking for both is a shortfall error not a half-placement); admission_preserves_archetype_scoped_repulsion_across_the_inversion (a Server spec built by `admission_spec` is still repelled by no-server and still NOT by no-appliance — the pre-B7 answer, which is what makes the admit_workload path behaviourally identical); describe_names_the_toleration_so_a_refusal_is_readable; a_toleration_alone_is_still_an_unconstrained_spec.")
136//! @yah:verify("REGRESSION GATE FLIPPED, not deleted. `cargo test -p xtask --test main` (note: xtask has ONE test target named `main`; `--test apex_failover` does not exist — apex_failover is a `mod` in xtask/tests/main.rs). Result 65 passed / 1 failed. xtask/tests/apex_failover.rs: the drill's finding-1 test was inverted and renamed every_repelling_taint_at_once_leaves_the_apex_bundle_exactly_where_it_was -> ..._now_makes_the_apex_node_ineligible; it now asserts that ONE repelling key is enough (checked before the all-three case so a regression handling only the union is still caught), that all three refuse, and that restoring us-east-001's real taint list [\"public-ip\"] puts the placement straight back. The module header was rewritten to say the hole is closed. ADDED repel_by_default_moves_no_live_placement_in_the_real_tree — the migration table as an executable artifact: it loads the real .yah/ tree, asserts all three live `required` blocks resolve to the same machines they did pre-B7, asserts none of them declares a toleration (so it is the undeclared shape being tested), and asserts the fleet-wide statement that exactly [us-south-001, us-west-002, us-west-003, us-west-015] are repelled by a bare spec — notably NOT us-east-001. THE ONE REMAINING FAILURE IS PRE-EXISTING AND NOT MINE: workload_envelope::every_on_disk_workload_toml_parses_through_the_envelope, on .yah/infra/state/sources/scrabcake/site/site/workload.toml (`unknown field routes`). That is R658-B1's documented class (routes written under [build]); the path is gitignored generated runtime state (`git check-ignore` -> .yah/.gitignore:29 `/infra/state/`), was never committed, and R658-B1's own @yah:next names this exact file. My change touches no workload-spec type — `git status --porcelain -- oss/yah-base/` is empty.")
137//! @yah:handoff("SCOPE BOUNDARY HELD, deliberately. yah-marketing's candidate set was NOT widened: `.yah/services/yah-marketing/mirrors/cloud.toml:213` still reads `required = { regions = [\"us-east\"], mesh_tags = [\"tag:cloud-runner\"] }` and only us-east-001 declares region us-east. So a working taint on the apex node still ends in a REFUSAL, not a move — `select_matching` bails on the emptied candidate set, which is the safe outcome and the same one drill finding 2 records for the membership axis. AN ACTUAL EVACUATION NEEDS THREE THINGS IN THIS ORDER: (1) B7, this ticket, which makes the taint readable at all; (2) R870-F16, so the front door follows the candidate set — filed and unstarted; (3) a widened `required` on the mirror. Doing (3) before (2) buys a workload that relocates and a yah.dev that 503s, which is why it was not done here. Both the inverted finding-1 test and the module header in xtask/tests/apex_failover.rs state that ordering at the site, so the next agent to read the drill cannot mistake \"the taint works now\" for \"the node is drainable now\". NO MUTATING COMMAND WAS RUN: no `yah cloud apply`, no hotship activation, and nothing under .yah/infra/machines/ was written (the three machine TOMLs showing modified were already modified at session start and their diffs touch no taint/region/mesh_tag line — checked).")
138//! @yah:handoff("GENERATED ARTIFACTS REGENERATED, and one of them is a peer's. `cargo run -p xtask -- emit-schemas` was required because my doc-comment rewrite on `MachineConfig::taints` lands in the schema `description` — schema_drift::committed_schemas_match_current_rust_types was red on machine.toml.schema.json. The regen also swept in mirror.toml.schema.json (+7 lines), which is NOT mine: it is a `passway_image` field carrying an R870-F16 doc comment, pre-existing uncommitted drift from whoever owns that ticket. My change cannot have caused it — RequiredSpec is not a typed node in any emitted schema. Regenerated per CLAUDE.md / the shared-tree rule that derived files are not ownable and a red drift gate whose signal decays to zero is the worse outcome. @Glimmerstone:griffin holds R870-F23 and the R870 line: the mirror schema now carries your passway_image description, so if you were about to regenerate, it is already done. Both schema files are the only two under .yah/schema/ that changed.")
139//! @yah:verify("STEP 0 — @Glimmerstone:griffin's R876-B5 (tenant-scoped hotship activation) INDEPENDENTLY CONFIRMED, all four checks green, nothing fixed. (1) `bash -n scripts/hotship.sh` clean. (2) `./scripts/hotship.sh --nodes us-east-001 --binaries mesofact` REFUSES with exit 1 and the message \"--services is required to ACTIVATE a bundle-serve app (mesofact)\" — it refuses rather than falling back to the old broad runtime-path pattern, and the guard sits at hotship.sh:507 ahead of the version stamp and every remote call. (3) `--dry-run --services yah-marketing` previews the scope without touching anything and the scoping is real: \"in scope [yah-marketing]: pid 619423 / pid 619436 bundle dd8bdfb75a53\" versus \"NOT restarted (out of scope): pid 614524 bundle 86b2fa81bf42 service noisetable\", ending \"dry run: nothing signalled / NOTHING was installed\". (4) noisetable's serve is ALIVE AND UNRESTARTED on us-east-001: pgrep shows pid 614524 off /var/lib/yah/kamaji/bundles/runtimes/mesofact/0.8.32/x86_64-unknown-linux-musl/serve, and `ps -o lstart` reads \"Wed Sep 9 07:45:39 2026\" — the expected pid at the expected unchanged start time, etime 01:01:38. `curl -sS -o /dev/null -w %{http_code} https://yah.dev/` = 200. No real hotship activation was run.")
140//! @yah:verify("BUILDS. `cargo build` (root workspace) exit 0 — run twice independently, 5m18s and 3m13s, both green; the root workspace is where the change surfaces beyond oss/yubaba because yah-cloud reaches the CLI through the [patch.crates-io] bridge. `cargo check --manifest-path oss/yubaba/Cargo.toml -p yubaba --all-targets` exit 0. Clean re-measure of `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` after all annotation writes: 1137 passed / 0 failed / 4 ignored, exit 0 — identical to the first post-change measurement, so the earlier W298 skew advisory naming config.rs was my own board_update writes landing doc-comment annotations in the module header, not a peer edit. A later advisory on the root build named app/yah/cli/src/cloud.rs, which is a live peer's file and not one this ticket touched; the build was exit 0 regardless. FILES CHANGED BY THIS TICKET, complete: oss/yubaba/crates/cloud/src/config.rs, xtask/tests/apex_failover.rs, xtask/tests/mirror_ingress.rs, .yah/schema/machine.toml.schema.json, .yah/schema/mirror.toml.schema.json. Nothing under .yah/infra/ or .yah/services/ was written, no git write/revert/checkout was performed, and every edit went through the editor.")
141//! @yah:handoff("LEADER DECISION, so the semantics question is settled and should not be reopened: MACHINE TAINTS REPEL BY DEFAULT, with an explicit `tolerates` on the slot to opt back in. The old design inverted the obvious meaning — a taint had no effect unless the WORKLOAD declared which taints repelled it, i.e. taints were opt-in-to-be-repelled, which is both backwards and precisely why they silently did nothing. `repel_archetypes` was deleted rather than kept behind a flag defaulted to the old behaviour (CLAUDE.md, \"break it, don't tape it\").")
142//! @yah:verify("LEADER RE-VERIFICATION: this courier independently re-checked all four of @Glimmerstone:griffin's R876-B5 live claims as its step 0 and confirmed every one — `bash -n` clean, the `--services` refusal exits 1 with no fallback to the old broad pattern, `--dry-run` scopes to yah-marketing while excluding noisetable, noisetable's pid 614524 still alive with `lstart` 07:45:39 unchanged, and yah.dev 200. Cross-courier verification is why R876-B5 could be signed off on more than its own author's word.")
143//! @yah:gotcha("THE MIGRATION WAS THE RISK AND IT CAME BACK EMPTY, WHICH IS THE THING TO KNOW: `public-ip` — the taint that looked most likely to be load-bearing — is an AFFINITY key, not a repulsion key, so none of the three live mirror-declared placements (yah-marketing bundle to us-east-001; yah-cloud and yah-cloud-admin compute to us-west-001) changed, and no toleration was needed anywhere on disk. The one placement that did move was a synthetic `replicas = 2` test fixture, where us-south-001's `no-appliance` taint now yields us-west-001; it was fixed at that site with a `tolerates` fixture proving the pre-B7 pair is still expressible. Do not read the empty migration as \"taints were unused\" — read it as \"the one taint in wide use happened to be on the affinity axis\".")
144//!
145//! @yah:ticket(R870-F23, "Render and supervise the inner door: the service.toml + domain-manifest join that feeds passway's PathRouter config")
146//! @yah:at(2026-09-09T08:23:48Z)
147//! @yah:status(open)
148//! @yah:assignee(agent:bundle-anthropic-glimmerstone)
149//! @yah:parent(R870)
150//! @yah:next("THE CONSUMER SIDE IS DONE AND ITS FORMAT IS FIXED (R870-T18, in review). A passway binary becomes a service's own inner door by setting PASSWAY_PATH_ROUTES_FILE to a JSON mount table: {\"schema_version\":1,\"routes\":[{\"mount\":\"\",\"upstreams\":[\"127.0.0.1:8081\"]},{\"mount\":\"/app\",\"upstreams\":[\"127.0.0.1:8082\"],\"headers\":{\"cross-origin-opener-policy\":\"same-origin\"}}]}. Parser + validation: oss/passway/crates/passway/src/path_routes_file.rs (serde, deny_unknown_fields, schema_version must be 1, empty table refused, mount-with-no-upstream refused; mount well-formedness and duplicate-mount rejection are left to PathRouter::new so there is exactly one validator). Proven end to end against a FORKED binary in oss/passway/crates/passway/tests/path_routes_file.rs. This ticket is the producer: write that file.")
151//! @yah:next("WHY THIS IS A SEPARATE TICKET AND NOT HALF OF R870-T18. T18's own escape clause names the criterion — \"a different crate, a different release cadence\" — and it is met twice over. (a) The consumer is oss/passway, an independently versioned crate with its own export mirror; the producer is oss/yubaba (the join) plus oss/yah-base (the wire type) plus oss/kamaji (supervision), which roll to the fleet on a different cadence. (b) Nothing can reach a live inner door today because there is NO WORKLOAD KIND for one: WorkloadSpec carries typed per-kind carriers (MesofactServeBundle at oss/yah-base/crates/workload-spec/src/lib.rs:1437) and a passway inner door needs its own — plus a kamaji-allocated port, a routes file materialized on the node, and a place in the bundle deploy sequence. Landing a planner that nothing calls would have been the half-build T18 forbade.")
152//! @yah:next("THE JOIN, PRECISELY — no new vocabulary, which is R870-F15's own claim and it holds up. Inputs: .yah/services/<svc>/service.toml (ServiceComponent { id, kind, mount, ... }, config.rs:3427) and .yah/domains/<zone>.toml (DomainRoute { path, headers, mode }, config.rs:4558, where front_door = passway). Per mount: mount = path_route::mount_from_component(component.mount) — that function already exists and is already the ONE place the \"app\"/None to \"/app\"/\"\" translation happens; headers = the DomainRoute whose route_path_prefix(path) equals normalize_mount(component.mount) (cross_ref_validate already PROVES those two agree, config.rs:1688-1725, so the join cannot silently mismatch); upstreams = the address of the deployed unit serving that mount. Only the last one is placement-time and is why this needs the workload kind above. Group by DEPLOYED UNIT, not by component: every bundle-tier component of a service shares ONE bundle workload (that is config 1, R870-B11), so config-1 mounts collapse to a single root upstream and only independently-deployed units earn their own mount.")
153//! @yah:next("THE TWO ADMISSION RULES, and where each one goes. Both belong to the GENERATOR, never to passway — passway proxies whatever PathRouter it is handed and has no view of how many components a service declares. (1) A service with ONE independently-deployed unit gets NO inner tier at all — enforce by construction: the planner returns Option<InnerDoorPlan> and answers None below two units, so there is no config to write and no process to supervise, and the negative is assertable on the ABSENCE of the plan rather than on a site staying up. (2) A component cannot be both bundle-staged (config 1) and its own workload. R870-B11 landed the config-1-internal half in CloudConfig::cross_ref_validate (config.rs:1621-1657, two bundle components at one mount are refused); put this half in the SAME loop rather than a parallel one. NOTE, checked not assumed: the second half is NOT EXPRESSIBLE TODAY — [providers.bundle] is a per-MIRROR slot, not per-component, so there is no way to say \"give this one component its own workload\" at all. The rule becomes writable in the same commit that introduces that vocabulary, which is this ticket. Do not invent the vocabulary separately.")
154//! @yah:gotcha("DESIGN WRINKLE FOUND WHILE BUILDING R870-T18, and it is an operator call, not a coding one. passway ALWAYS terminates TLS on its listener: TlsMode has exactly two variants, Manual and Acme (oss/passway/crates/passway/src/tls.rs:215), and main() unconditionally calls proxy_service.add_tls_with_settings(&listen, None, tls_settings). So an inner door on loopback still needs a cert on disk, and the outer door still needs PASSWAY_UPSTREAM_TLS=true plus an SNI to reach it. That works — T18's binary-level test does exactly this with an rcgen self-signed leaf — but it means the \"cheap inner tier\" costs a cert, a renewal story, and an upstream TLS handshake per request on loopback. The obvious fix is a plaintext listener mode, and it was deliberately NOT taken in T18: adding a way for a public-facing trust-boundary door to serve cleartext is a security decision with a blast radius past this relay. Decide it before building the supervisor, because it changes what the workload spec has to carry.")
155//! @yah:verify("A two-component service whose components deploy INDEPENDENTLY gets an inner door: one yah cloud apply leaves both https://<host>/ and https://<host>/app/ at 200, and curl -sI on /app/ carries cross-origin-opener-policy: same-origin AND cross-origin-embedder-policy: require-corp from the /app/* route in the domain manifest, while / carries neither.")
156//! @yah:verify("THE NEGATIVE, asserted on absence rather than on uptime: a single-component service (yah-marketing) produces NO inner-door config and NO inner-door process — no routes file materialized on the node, no extra supervised workload in kamaji's table, and a byte-identical workload spec to today. A unit test on the planner returning None is the cheap half; the node-side absence check is the half that matters.")
157//! @yah:gotcha("OPERATOR CALL ASKED AND NOT ANSWERED (R870 relay leader, session:abde2cbb, 2026-09-09). The TLS question in this ticket first gotcha was put to the operator as a three-way choice and the prompt timed out unanswered after 30 minutes, so it remains genuinely open — it was not skipped and not decided by default. The three options as framed, so whoever picks this up does not have to re-derive them: (A) add a plaintext listener mode gated so it is structurally impossible to combine with a public bind — refuse at config load unless the bind is loopback, keep it mutually exclusive with ACME/cert paths; this was the leader recommendation, on the grounds that it makes the inner tier actually cheap as R870-F15 design claimed while keeping the risk a bounded testable invariant rather than an operator remembering not to misconfigure it. (B) keep TLS everywhere and have F23 carry a cert-issuance plus renewal story for every inner door, which is safest by construction and already proven working in R870-T18 binary-level test with an rcgen self-signed leaf, but makes every service with 2+ independently-deployed components pay a cert, a renewal and a loopback handshake per request. (C) park the tier — nothing regresses, because config 1 (bundle staging, R870-B11, in review) already covers the deploy-together case, which is the one noisetable actually needs. THIS IS THE ONLY THING BLOCKING F23 DESIGN; the join itself, both admission rules and the workload-kind vocabulary are all specified in this ticket next entries and need no further decisions.")
158
159use anyhow::{bail, Context, Result};
160use serde::{Deserialize, Serialize};
161use std::collections::{BTreeMap, HashMap};
162use std::path::Path;
163use thiserror::Error;
164use workload_spec::secrets::SecretAccess;
165use workload_spec::sovereign::Membership;
166pub use workload_spec::sovereign::SovereignRole;
167use workload_spec::{validate, LifecycleArchetype, Locality, TenantId, WorkloadSpec};
168
169/// Static node capacity declaration on `machine.toml` (R572-F3).
170///
171/// `memory_mb` and `cpu_millis` express the node's *total* hardware budget.
172/// F5's bin-packer subtracts the sum of committed workload requests from
173/// this floor to determine available headroom; an absent `allocatable`
174/// block means no capacity constraint is enforced (any workload fits).
175#[derive(Debug, Clone, Serialize, Deserialize)]
176#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
177pub struct NodeAllocatable {
178 /// Total physical RAM in mebibytes (e.g. 512 for a 512 MB node).
179 pub memory_mb: u32,
180 /// Total CPU in k8s millicores (1000 = 1 core, 250 = 0.25 CPU).
181 pub cpu_millis: u32,
182}
183
184/// `[registration]` — facts **observed** about a running box, written by the
185/// fleet rather than declared by an operator (R707-T1).
186///
187/// The rest of `machine.toml` is *declaration*: intent, operator-authored,
188/// reviewed and diffed like any other source. This block is the other half —
189/// what the box turned out to be once it booted and joined. Keeping the two
190/// apart is what lets the published fleet index (R707-F3) say which half it is
191/// carrying; publishing them under one schema would bake the confusion into a
192/// permanent record.
193///
194/// The split is a **provenance** boundary, not a trust or reach one:
195/// - *Declaration* answers "what did we ask for" — `name`, `region`, `arch`,
196/// `mesh_tags`, `[allocatable]`, and the declared reach in [`ConnectSpec`].
197/// - *Registration* answers "what did we observe" — the hostkey TOFU'd at
198/// attach, the mesh address headscale assigned at join.
199///
200/// It stays in the git-tracked TOML on purpose. Registration is not local
201/// scratch state: every consumer needs the mesh address to dial a node, so it
202/// has to travel with the declaration. (`.yah/infra/state/machines/<name>.json`
203/// — [`crate::state::MachineState`] — remains the *gitignored* sidecar for
204/// provider-side derivatives that nobody but this camp needs.)
205#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq, Eq)]
206#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
207pub struct MachineRegistration {
208 /// Yubaba's ed25519 `/identity` fingerprint, TOFU-recorded by
209 /// `yah cloud machine attach` on first contact (`SHA256:…`). An observed
210 /// property of a running process — not the operator's intent — which is
211 /// why it moved out of the top level here.
212 #[serde(default, skip_serializing_if = "Option::is_none")]
213 pub hostkey_fingerprint: Option<String>,
214 /// Mesh (headscale/tailnet) IPv4 assigned at join, e.g. `"100.64.0.1"`.
215 /// Bare address, not a URL: the *port* is declared reach and lives on
216 /// [`ConnectSpec::yubaba_port`]. [`MachineConfig::yubaba_url`] composes the
217 /// two. Absent until the node has joined the mesh.
218 #[serde(default, skip_serializing_if = "Option::is_none")]
219 pub mesh_ipv4: Option<String>,
220 /// RFC3339 timestamp of the mesh join that produced `mesh_ipv4`. Free-form
221 /// audit; nothing keys off it.
222 #[serde(default, skip_serializing_if = "Option::is_none")]
223 pub joined_at: Option<String>,
224}
225
226impl MachineRegistration {
227 /// True when nothing has been observed yet — used to omit the whole
228 /// `[registration]` table from a serialized machine TOML.
229 pub fn is_empty(&self) -> bool {
230 self.hostkey_fingerprint.is_none() && self.mesh_ipv4.is_none() && self.joined_at.is_none()
231 }
232}
233
234/// Per-machine TOML from `.yah/infra/machines/<name>.toml`.
235///
236/// Two halves, split by provenance (R707-T1): everything here is *declaration*
237/// — operator intent under review and blame — except [`registration`], which
238/// carries what the fleet observed. See [`MachineRegistration`] for why the
239/// boundary is drawn there and what depends on it.
240///
241/// @yah:ticket(R860-T5, "Model per-node native-exec capability as an admission axis (W338 §Placement consequences 3 / R858-T4 gap)")
242/// @yah:status(review)
243/// @yah:phase(P1)
244/// @yah:at(2026-09-05T18:29:19Z)
245/// @yah:assignee(agent:bundle-anthropic-ashguard)
246/// @yah:parent(R860)
247/// @yah:next("Cheapest defensible shape: express it on MachineConfig, which already has the two vocabularies — `mesh_tags: Vec<String>` (config.rs:246, superset match, already carries `arch:`/`os:`/`tag:build-worker`) and `taints: Vec<String>` (config.rs:337). A `native-exec` mesh tag required by any group member whose kind is native is a one-line admission axis in `admission_spec()`. Whichever is chosen, it must be declared in .yah/infra/machines/*.toml for the nodes that actually run kamaji with --native-exec-dir, and `check_inert_taints` (config.rs:703) lints unread taint keys dead — so a taint nobody reads will be flagged.")
248/// @yah:verify("cargo test -p cloud --lib config")
249/// @arch:see(.yah/docs/working/W338-workload-dependencies-and-appliance-composition.md)
250/// @yah:depends_on(R860-T4)
251/// @yah:gotcha("Verified 2026-09-04: native-exec capability is modelled NOWHERE in placement — `rg \"native\" oss/yubaba/crates/cloud/src/config.rs` returns zero hits, and the raft state machine models no member attributes, labels or taints at all (`rg \"taint|capabilit|labels|mesh_tag\"` over raft/{mod,store,network}.rs yields one unrelated comment at raft/store.rs:591). Native-exec is a node-local kamaji startup decision today: `--native-exec-dir` (oss/kamaji/crates/kamaji-bin/src/main.rs:152-156, :51-55) plus the `native-exec` cargo feature (kamaji-bin/src/server.rs:329-330). A node without it refuses the deploy at dispatch time and nothing upstream can see that in advance — which is exactly the deploy-time surprise W338 wants turned into a placement precondition.")
252/// @yah:handoff("NATIVE-EXEC IS NOW A PLACEMENT PRECONDITION, NOT A DISPATCH-TIME SURPRISE. New `pub const NATIVE_EXEC_MESH_TAG: &str = \"cap:native-exec\"` in oss/yubaba/crates/cloud/src/config.rs (declared just above `node_selector_mesh_tags`), and one axis in `admission_spec()` immediately after the R860-T4 group loop: if ANY member of `placement_group(ws, declared)` returns true from `WorkloadSpec::wants_native_exec()`, the tag is appended to the derived `RequiredSpec.mesh_tags` (deduped). No new field on `RequiredSpec`, no signature change anywhere, no wire or serde change — the mesh_tags axis is already an AND-ed superset check against `machine.mesh_tags` in `matches` and is already rendered by `describe`, so a refusal now reads `required.mesh_tags=[...,cap:native-exec]`.")
253/// @yah:handoff("ITEM 1 — HOW A NATIVE WORKLOAD IS DETECTED, settled by opening the type rather than guessing. There is no `kind` on `WorkloadSpec`: on the wire a native workload is still `Workload::Container(WorkloadSpec)`, and the ONLY difference is the annotation `yah.exec = native`, read through `WorkloadSpec::wants_native_exec()` (oss/yah-base/crates/workload-spec/src/lib.rs:2939; consts `NATIVE_EXEC_ANNOTATION` / `NATIVE_EXEC_VALUE` at :3402/:3407). That accessor is what the admission axis calls — matching kamaji, whose `deploy_container` checks the same marker first and routes to `deploy_native_exec` (oss/kamaji/crates/kamaji-bin/src/server.rs). The `yah.exec` key is a substrate selector with a second value, `microvm` (`wants_microvm`, same key, R605-F8), so per-node microVM capability is the obvious sibling axis and is NOT modelled here — see next-steps.")
254/// @yah:handoff("ITEM 2 — DECLARATIONS LANDED ON TWO NODES, FROM READINGS RECORDED IN-REPO, NOT INFERRED. `cap:native-exec` added to `mesh_tags` in .yah/infra/machines/us-west-001.toml and .yah/infra/machines/us-west-003.toml, each with a comment naming its evidence and its re-check condition. us-west-001: the R858 gotcha in its own header records a `ps` reading taken on the box 2026-09-05 — pid 515908 is `/usr/local/bin/kamaji --native-exec-dir /var/lib/yah/kamaji/native`, supervising headscale as a native child. us-west-003: its header's 'THE DEPLOYED KAMAJI PREDATES THE microVM BACKEND' note quotes the box's actual ExecStart, read over ssh 2026-09-01, carrying `--native-exec-dir /var/lib/yah/kamaji/native` (corroborated by .yah/docs/architecture/A043-yah-on-machine-daemons.md's @yah:verify for the same probe). Both comments say plainly that the capability lives in the systemd unit's ExecStart, not in the TOML, so it must be re-checked after any roll.")
255/// @yah:handoff("ITEM 2, THE NEGATIVES — TWO NODES ARE KNOWN NOT TO HAVE IT AND WERE DELIBERATELY LEFT UNSET. us-south-001: kamaji refused headscale there 2026-09-03 with 'native backend not configured — start kamaji with --native-exec-dir' (the R858 chain, quoted in .yah/infra/machines/us-west-001.toml and W267). I did NOT edit us-south-001.toml — it was already dirty in the working tree at the anchor SHA and @Ashguard:eclipse is live on R858, so I left it alone rather than race it; the mechanism fails closed there, which is the correct state. us-west-015 (the sole darwin builder): W254-darwin-build-nodes.md's own next-step records that its kamaji is built/started `--docker` only. I added a comment to us-west-015.toml explaining that the tag is deliberately absent, that this is the node where the axis changes an error message (a darwin build row is native by construction, so it is now refused at ELECTION naming cap:native-exec instead of reaching the box and being refused by kamaji), and the exact enable sequence: rebuild with `--features native-exec`, restart with `--native-exec-dir <dir>`, THEN add the tag. us-west-002/011/013/014 are unestablished from the repo and left unset. THE OPERATOR-FACING ANSWER: the file is `.yah/infra/machines/<node>.toml` and the key is `mesh_tags`; add the literal string `cap:native-exec` to that array, and only after the roll.")
256/// @yah:handoff("DECISIONS THE BRIEF LEFT OPEN, all recorded in doc comments at the site. (1) MESH TAG, NOT TAINT — as recommended, and the doc says why in the terms the brief asked for: mesh tags are positive capability with superset matching ('this node CAN'), which is the claim being made; a taint is repulsion and would have to be inverted to `no-native-exec` on every node LACKING the backend (declaration burden on the majority, and silently wrong for a node nobody has edited) AND taught to `taint_effect`, or `check_inert_taints` would correctly lint the key dead. (2) THE `cap:` NAMESPACE IS NEW. Live prefixes are `tag:` (operator-assigned role), `arch:`/`os:` (silicon and userland facts, emitted as requirements by qed::platform::build_worker_mesh_tags), and `tier:` which R763 RETIRED for architecture and reserved for the environment axis — so reusing any of them would have stated the wrong kind of fact. A capability the daemon was configured with is none of those. Nothing validates tag prefixes (only `check_retired_arch_tags` looks at one), so this costs no wiring. (3) COMPUTED OVER THE GROUP, not the requirer — that is literally W338's sentence ('supply = self specs must be placeable where their requirer lands'), and the second test proves it: an ordinary container requirer with a `local` edge to a native provider is pulled onto a capable node. (4) FAILS CLOSED, accepted deliberately: an undeclared node is simply not a candidate, so an undeclared fleet reports 'no node admits' at election rather than dispatching to a node that refuses. Nothing in `.yah/infra/workloads/` is native-marked today (only yah-cloud-admin.toml exists there), so the only live consumer is the qed darwin build row, where failing closed is strictly the better error.")
257/// @yah:handoff("BLAST RADIUS, MEASURED. `admission_spec` is private and its callers are unchanged: `admit_workload` / `admit_workload_candidates` / `admit_workload_in_group` (config.rs), reached from app/yah/cli/src/cloud.rs (deploy, rolling, topology analyzer), app/yah/cli/src/yubaba_client.rs `elect_node`, and cloud/src/migrate.rs. The headscale appliance path inside yubaba (headscale_appliance.rs) does NOT go through admission — it is node-internal — so nothing eclipse holds on R858 is touched by this. Files edited, in full: oss/yubaba/crates/cloud/src/config.rs; .yah/infra/machines/{us-west-001,us-west-003,us-west-015}.toml. Nothing in oss/yubaba/crates/yubaba/ was opened, and oss/kamaji/crates/kamaji-bin/src/server.rs was READ ONLY (to confirm the marker check), per @Ashguard:hydra's contention triage.")
258/// @yah:handoff("ONE SCOPE ADDITION, stated loudly rather than slipped in: `MachineConfig::mesh_tags` (config.rs:256) had NO doc comment at all — the operator-facing declaration key for four tag namespaces was undocumented. I gave it one enumerating `tag:` / `arch:`+`os:` / the new `cap:` / retired `tier:`, and noting that nothing validates the prefix (which is why the two lints exist). CONSEQUENCE TO KNOW: that field's doc is the source of the `mesh_tags` description in the GENERATED .yah/schema/machine.toml.schema.json, so it is schema-drift-affecting — see the gotcha.")
259/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I found it and left it. Quote this SHA rather than 'HEAD' in any revert/restore instruction; to undo a hunk, read it with `git show 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2:<path>` and put it back with Edit, never `git checkout`/`restore` (they restore whole files and would delete peers' uncommitted work).")
260/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
261/// @yah:next("MICROVM IS THE IDENTICAL UNMODELLED GAP, one line away. `yah.exec` is a substrate selector with a second value: `WorkloadSpec::wants_microvm()` (workload-spec/src/lib.rs, R605-F8), and kamaji constructs MicroVmRuntime only when started with `--microvm-dir` — A043's probe records that us-west-003's deployed kamaji has `--native-exec-dir` but NOT `--microvm-dir`, so a microvm-marked deploy is refused there by exactly the same dispatch-time surprise this ticket removed for native. The shape is `cap:microvm` alongside NATIVE_EXEC_MESH_TAG in the same `if` in `admission_spec`. Not done here because no node in the fleet can host one yet (R605-F14 must land a guest kernel + rootfs first), so declaring the tag anywhere today would be the wrong fact.")
262/// @yah:next("us-south-001 needs `cap:native-exec` DECIDED, not defaulted, and it is the R858 node. It is the one machine the repo positively records as LACKING the backend (kamaji refused headscale there 2026-09-03), so leaving the tag off is correct TODAY — but if R858's fix is 'give us-south-001 a native-capable kamaji' rather than 'stop moving headscale', then the roll and the tag must land together, in that order. I left .yah/infra/machines/us-south-001.toml untouched because it was already dirty at anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 and @Ashguard:eclipse is live on R858.")
263/// @yah:next("R860-T6 (`supply = \"self\"` provisioning) inherits this for free — `admission_spec` already requires the capability of the whole group, so a self-provisioned native member cannot be elected onto a node that cannot run it. What T6 must still not do is re-elect per member: reuse the node URL `elect_node` returned for the requirer, per R860-T4's handoff.")
264/// @yah:verify("BASELINE RECORDED BEFORE EDITING, at tree anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2: `cargo test -p yah-cloud --lib` from oss/yubaba = 1090 passed / 0 failed / 4 ignored, exit 0 — exactly the count the brief predicted. AFTER: 1093 passed / 0 failed / 4 ignored, exit 0 (+3, exactly the three tests added). `cargo check -p yah-cloud --all-targets` exit 0 and `cargo check -p yubaba --all-targets` exit 0 (yubaba consumes cloud, so it is where any signature change would surface — there is none). Every exit code echoed explicitly via an `EXIT=$?` / `${PIPESTATUS[0]}` marker and read back, never inferred from an empty grep. The four `yah-cloud` warnings are all pre-existing and in other files (object-store r2.rs, reconciler/mesofact_static.rs unused imports, app_manifest.rs, reconciler/mod.rs non_snake_case); config.rs contributes none.")
265/// @yah:verify("NEW TESTS (config.rs `mod tests`, R860-T5 section at the end, after the R860-T4 block). (1) a_node_without_the_native_exec_capability_cannot_host_a_native_workload — a `yah.exec = native` spec is refused by a bare node with an error naming `cap:native-exec`, and admitted by a node declaring it, with both nodes in the same fleet so the choice is provably the tag. (2) a_local_edge_to_a_native_provider_makes_the_requirer_need_the_capability — an ordinary container requirer (asserted `!wants_native_exec()`) with a `local` edge to a native provider lands on the capable node, while the SAME spec without the edge still lands on the plain one, so the constraint provably comes from the group. (3) a_group_with_no_native_member_does_not_require_the_capability — the regression guard: the axis is absent from `admission_spec`'s mesh_tags and a group with a local edge between two ordinary specs still admits on a node declaring nothing. Helper `native_spec()` asserts the marker reads back through `wants_native_exec()` before the test uses it, so a typo cannot make the test pass vacuously.")
266/// @yah:verify("Machine-config lints were considered and are unaffected by construction: `check_inert_taints` reads `taints` (I touched none), and `check_retired_arch_tags` flags only the `tier:` prefix. `cap:` is a new namespace and nothing validates prefixes, so no lint fires and no lint needs teaching.")
267/// @yah:gotcha("SCHEMA DRIFT IS EXPECTED FROM THIS TICKET AND WAS ALREADY RED BEFORE IT. `.yah/schema/machine.toml.schema.json` is generated from `cloud::config` by `cargo run -p xtask -- emit-schemas`, and MachineConfig's DOC COMMENT is what the generator emits as its `description` — which means (a) my new `mesh_tags` doc changes it, and (b) so does this very handoff, because R860-T5's @yah: annotation block lives inside MachineConfig's doc at config.rs:201. That file was ALSO already dirty in the working tree at anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2, before I touched anything — `scripts/check-schema-drift.sh` compares the regenerated tree against git, so it is red for any uncommitted schema edit regardless of author. Regenerate with `cargo run -p xtask -- emit-schemas` (or `scripts/check-schema-drift.sh --update`) when the root target dir is not contended; the pre-commit hook no longer does it (disabled 2026-08-15, see CLAUDE.md).")
268/// @yah:verify("SCHEMA REGENERATED IN THIS SESSION, so the drift gate is not left for the next reader: `cargo run --quiet -p xtask -- emit-schemas` exit 0, run from the repo root after the handoff was written (so it captures the annotation text too). Two files moved. `.yah/schema/machine.toml.schema.json`: MachineConfig's `description` grows by this ticket's annotation block, plus a genuinely new `mesh_tags.description` from the doc comment I added. `.yah/schema/workload.toml.schema.json`: +104 lines that are NOT mine — the `Locality` / `Requirement` / `Supply` / `WorkloadSpec.requires` types R860-T1 landed had never been emitted, so the sibling ticket's schema drift was still outstanding and my regen swept it in. Derived artifacts are not ownable (shared-tree doctrine), so this is deliberate rather than accidental; @Ashguard, whoever picks up R860-T1's review should know the schema now describes `requires`.")
269/// @yah:verify("FINAL RE-RUN AFTER THE HANDOFF ANNOTATION WAS WRITTEN INTO config.rs (the board write edits MachineConfig's doc block, so the file changed under the earlier green): `cargo test -p yah-cloud --lib` = 1093 passed / 0 failed / 4 ignored, exit 0. Unchanged. Note for anyone reading the camp build rail's skew warnings on this session: the one `SUSPECT RESULT` it emitted names `oss/yubaba/crates/cloud/src/config.rs` as modified mid-run, and that modification was MY OWN board_handoff annotation write, not a peer — the two authoritative runs (full lib test, and both cargo checks) each came back `Input closure unchanged across the whole run: no skew`.")
270/// @yah:verify("All builds were run with `CARGO_TARGET_DIR=/tmp/r860t5-target` rather than the shared oss/yubaba/target, following R860-T4's recorded gotcha — a peer (session:83093d9d) held the shared target lock for the entire session (20+ minutes of `cargo check -p yubaba --lib`). Costs one cold dep build, then every subsequent run is seconds. Worth reaching for immediately when the queue message says you are behind someone.")
271/// @yah:handoff("LEADER RE-VERIFIED (session:69b18855, independent of the courier's self-report). `cargo test -p yah-cloud --lib` from oss/yubaba: 1093 passed / 0 failed / 4 ignored, exit 0, against the 1090/0/4 baseline this relay's own T4 established — +3 = exactly its new tests. Axis confirmed by content: `NATIVE_EXEC_MESH_TAG = \"cap:native-exec\"` at config.rs:2304, appended to the derived `RequiredSpec.mesh_tags` at :2112-2114 when any `placement_group` member returns true from `WorkloadSpec::wants_native_exec()`. No new `RequiredSpec` field, no signature change, no wire change — it rides the existing AND-ed superset check, so a refusal now reads `required.mesh_tags=[...,cap:native-exec]`.")
272/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
273/// @yah:verify("MACHINE DECLARATIONS AUDITED FOR PROVENANCE, because a wrong capability declaration is worse than an absent one. Both are traceable to measurements ALREADY RECORDED IN-REPO, not inferred: us-west-001 from the `ps` reading at us-west-001.toml:21 (pid 517125, ppid 515908 = `/usr/local/bin/kamaji --native-exec-dir /var/lib/yah/kamaji/native`, cgroup `0::/yubaba.slice/kamaji.service/native`, 2026-09-05); us-west-003 from the actual ExecStart read over ssh 2026-09-01 at us-west-003.toml:141. us-west-015 was deliberately left WITHOUT the tag and carries enable instructions at :207-217 — unknown fails closed, which is the correct direction. No node was guessed at and nothing was probed live.")
274/// @yah:handoff("THIS TICKET MODELS THE EXACT DRIFT THAT CAUSED THE 25-HOUR MESH OUTAGE, which is worth stating because it turns an abstract W338 bullet into a measured one. us-west-001.toml:8 records the root-cause chain: on 2026-09-03T06:03:03Z leadership moved to us-south-001, which tried to deploy headscale and kamaji refused — \\\"workload requests native host execution (yah.exec=native) but no native backend is available (native backend not configured — start kamaji with --native-exec-dir)\\\" — then the systemd fallback failed too, both at WARN, and the mesh had no coordination server for 25 hours. us-west-001.toml:10 names it explicitly as \\\"a silent per-node capability drift that placement does not model\\\". After this ticket, placement models it: a group needing native exec can no longer be admitted onto a node that has not declared `cap:native-exec`.")
275/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
276/// @yah:verify("RE-VERIFIED AT HEAD 00ee20d1 (session:aa5e882d, 2026-09-05). `NATIVE_EXEC_MESH_TAG` present in oss/yubaba/crates/cloud/src/config.rs (declared above `node_selector_mesh_tags`, appended to the derived `RequiredSpec.mesh_tags` when any `placement_group` member wants native exec). `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` = 1141 passed / 0 failed / 4 ignored, exit 0.")
277/// @yah:cleanup("cap:microvm remains the identical unmodelled axis, one line from done in the same `if` in `admission_spec`. Deliberately NOT taken: no node in the fleet can host a microvm until R605-F14 lands a guest kernel + rootfs, so declaring the tag today would assert a false fact. Do it when R605-F14 lands, not before.")
278#[derive(Debug, Clone, Serialize, Deserialize)]
279#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
280pub struct MachineConfig {
281 pub name: String,
282 pub provider: String,
283 /// Who the hardware actually comes from (`"ovh"`, `"vultr"`, `"on-prem"`).
284 ///
285 /// Deliberately *not* [`provider`](Self::provider), which selects the
286 /// auto-provision driver: a box we rented by hand and brought up over SSH
287 /// is `provider = "static"` for its whole life, and writing the vendor
288 /// there instead would flip it driver-backed and make
289 /// [`validate`](Self::validate) demand `location` + `server_type` it has no
290 /// answer for. The two axes genuinely differ — vendor is who bills you,
291 /// `provider` is who yah can call an API against.
292 ///
293 /// Worth recording because vendor-scoped policy is invisible in every other
294 /// field and decides real work: outbound port 25, rDNS/PTR control, IP
295 /// reputation, egress billing. It survived only in TOML prose until now,
296 /// which made it ungreppable at exactly the moment you need it.
297 #[serde(default, skip_serializing_if = "Option::is_none")]
298 pub vendor: Option<String>,
299 /// Human label for the box (`"gamer"`, `"the GEEKOM"`). Free-form and never
300 /// matched on — [`name`](Self::name) stays the identity everywhere. This is
301 /// only so operators and agents can say which box they mean out loud.
302 #[serde(default, skip_serializing_if = "Option::is_none")]
303 pub nickname: Option<String>,
304 /// Provider DC code (e.g. Hetzner `"hil"`). **Provisioning-only**: required
305 /// iff the provider has an auto-provision driver ([`provider_has_machine_driver`]);
306 /// a BYO `static` node we brought up over SSH has no such code. Optional at
307 /// load time so static machine.tomls omit it; [`MachineConfig::validate`]
308 /// enforces presence at the right moment for driver-backed providers.
309 #[serde(default, skip_serializing_if = "Option::is_none")]
310 pub location: Option<String>,
311 /// Provider SKU/size (e.g. Hetzner `"ccx13"`). Provisioning-only, same
312 /// optionality contract as [`location`](Self::location).
313 #[serde(default, skip_serializing_if = "Option::is_none")]
314 pub server_type: Option<String>,
315 /// **Deprecated (R330-F16).** A machine should describe *itself* (region,
316 /// zone, provider, mesh_tags); *which* mirrors run on it is derived by the
317 /// reconciler from each mirror's `required` placement spec, not declared
318 /// here. Now optional + omitted-when-empty so new machine.tomls leave it
319 /// out. The legacy `resolve_mirror_machine` topology fallback still reads
320 /// it until yubaba's reverse-index supersedes the topology.toml path; once
321 /// that lands, this field and its readers are removed wholesale.
322 #[serde(default, skip_serializing_if = "Vec::is_empty")]
323 pub hosts_mirrors: Vec<String>,
324 /// Positive placement facts about this node, matched as a **superset**:
325 /// a workload is admitted only where every tag it requires is present, so
326 /// adding a tag can only ever make a machine match more, never fewer.
327 ///
328 /// Four namespaces are live, and they are not interchangeable:
329 /// - `tag:<role>` — a role the operator assigns (`tag:build-worker`,
330 /// `tag:qed`, `tag:cloud-runner`, `tag:mac-builder`);
331 /// - `arch:<x86|arm>` / `os:<linux|darwin>` — facts about the silicon and
332 /// userland, emitted as *requirements* by
333 /// [`qed::platform::build_worker_mesh_tags`];
334 /// - `cap:<capability>` — something the node's daemons were configured to
335 /// be able to do. Today just [`NATIVE_EXEC_MESH_TAG`] (R860-T5);
336 /// - `tier:` is **retired** for architecture (R763) and reserved for the
337 /// environment axis — [`crate::validate::check_retired_arch_tags`]
338 /// flags a machine still carrying `tier:<arch>`.
339 ///
340 /// Nothing validates the prefix, which is why the lint above exists: a tag
341 /// nobody requires is silently inert, and a *stale* one silently stops
342 /// matching and reports "no node" rather than "wrong tag".
343 pub mesh_tags: Vec<String>,
344 /// Canonical geo region label (latency axis), e.g. `"us-west"`. F16's three
345 /// topology axes are orthogonal: `region` = geo (latency), `zone` = failure
346 /// domain within a region (HA), `provider` = network/cost. `region` is
347 /// distinct from `location` (the provider's DC code, e.g. Hetzner `"hil"`):
348 /// `location` is provider-scoped, `region` is our provider-neutral label.
349 /// Optional for backward-compat; a machine without it never satisfies a
350 /// `required.regions` constraint.
351 #[serde(default, skip_serializing_if = "Option::is_none")]
352 pub region: Option<String>,
353 /// Failure-domain label within a region (HA axis), e.g. `"hil"`. For
354 /// single-DC Hetzner this typically mirrors `location`. F16 placement
355 /// matches `required.zones` against this. Optional for backward-compat.
356 #[serde(default, skip_serializing_if = "Option::is_none")]
357 pub zone: Option<String>,
358 /// Declared CPU architecture (`"x86_64"` / `"aarch64"`). A machine has
359 /// exactly one — it's a first-class property of the box, not a reach
360 /// detail and not a mesh tag. Drives the yubaba release triple. Optional
361 /// only because there's no provider API to probe it (static nodes declare
362 /// it; a driver-backed provider may leave it unset until known).
363 #[serde(default, skip_serializing_if = "Option::is_none")]
364 pub arch: Option<String>,
365 pub bucket: Option<BucketSpec>,
366 /// **Legacy location, superseded by `[registration].hostkey_fingerprint`**
367 /// (R707-T1). Still deserialized so machine TOMLs written before the split
368 /// keep parsing; never *read* directly — go through
369 /// [`MachineConfig::hostkey_fingerprint`], which prefers the registration
370 /// block. [`MachineConfig::normalize`] folds this into `registration`, and
371 /// [`MachineConfig::save`] normalizes before writing, so a load→save cycle
372 /// migrates the file rather than dropping the value.
373 #[serde(
374 rename = "hostkey_fingerprint",
375 default,
376 skip_serializing_if = "Option::is_none"
377 )]
378 pub legacy_hostkey_fingerprint: Option<String>,
379 /// Provider-side SSH-key IDs (Hetzner: from `GET /v1/ssh_keys`)
380 /// authorized for `root` at create time. Defaults to empty for
381 /// backwards-compat with existing machine declarations; an empty
382 /// list yields a Hetzner-emailed random root password (which the
383 /// driver currently discards). Populate this when you want pre-mesh
384 /// SSH access for bootstrap deploys or recovery.
385 #[serde(default, skip_serializing_if = "Vec::is_empty")]
386 pub ssh_keys: Vec<u64>,
387 /// Cloudflare Tunnel ID this machine joins (e.g. `abc123.cfargotunnel.com`).
388 /// `None` → no tunnel (mesh-only node, no public ingress).
389 /// When set, `yah cloud machine provision` reads `cloudflare-tunnel-token`
390 /// from the keys vault and injects the cloudflared install block into
391 /// cloud-init so the new machine connects to CF edge on first boot.
392 #[serde(default, skip_serializing_if = "Option::is_none")]
393 pub cloudflared: Option<String>,
394 /// Provider-issued floating/reserved IP that follows **public-ingress
395 /// ownership** onto this box — R859-F2 (W267 §Tier 1).
396 ///
397 /// The value is the provider's own identifier, opaque here and interpreted
398 /// only by the matching adapter: a Hetzner numeric floating-IP id as a
399 /// string, an OVH Additional-IP address (`"51.81.85.200"`), a Vultr
400 /// reserved-IP UUID. Same "the adapter is the boundary" convention
401 /// [`crate::envoy::floating_ip::FloatingIpAssignInput::ip_id`] documents.
402 ///
403 /// # Why it lives on the machine
404 ///
405 /// [`crate::envoy::floating_ip`] shipped the `floating_ip.*` verbs and
406 /// three provider adapters with no config anywhere saying *which* floating
407 /// IP is "the" ingress IP — the gap R594-F5 recorded and deliberately left.
408 /// This is that field, and it sits beside [`cloudflared`](Self::cloudflared)
409 /// on purpose: that is already the per-node "how the world reaches this
410 /// box" handle, and a floating IP is the sovereign-tier answer to the same
411 /// question. `[[ingress]]`'s
412 /// [`tunnel_id`](crate::config::IngressEdge::tunnel_id) is the *service*
413 /// side of ingress identity — which cohort a given service fronts through —
414 /// and a floating IP is not per-service: one IP moves between boxes, so it
415 /// cannot be partitioned by slot or hostname.
416 ///
417 /// # Absent means "no floating-IP path", never an error
418 ///
419 /// Most machines have none, and that is the normal case: mesh-only nodes,
420 /// boxes behind a Cloudflare tunnel, and every provider without a
421 /// floating-IP adapter. The effector skips such a machine cleanly rather
422 /// than refusing — see
423 /// [`plan_ingress_owner_effect`](crate::provider::floating_ip::plan_ingress_owner_effect).
424 ///
425 /// # The cohort has to agree
426 ///
427 /// Every machine that can hold the same ingress IP must declare the *same*
428 /// id: the IP is one resource that moves, so two ids inside one
429 /// [`sovereign_group`](Self::sovereign_group) means an ownership flip
430 /// silently reassigns a *different* IP than the one currently serving
431 /// traffic. `yah cloud validate` refuses that
432 /// ([`crate::validate::check_ingress_floating_ip`]) rather than leaving it
433 /// to be discovered during a failover.
434 #[serde(default, skip_serializing_if = "Option::is_none")]
435 pub ingress_floating_ip: Option<String>,
436 /// When `true`, this machine hosts operator-bridge workloads (Tailscale
437 /// operator access to mesh-internal services). `yah cloud machine provision`
438 /// will install tailscaled and run `tailscale up` during cloud-init via the
439 /// `{{OPERATOR_BRIDGE_BLOCK}}` placeholder. Defaults to `false` for
440 /// backward-compat with existing machine declarations.
441 #[serde(default)]
442 pub hosts_operator_bridge: bool,
443 /// BYO `static`-node reach descriptor. Static nodes have no provider API to
444 /// probe, so how the camp reaches them (SSH user@host + the yubaba URL,
445 /// which is loopback until the WireGuard mesh lands) is *declared* here.
446 /// `None` for driver-backed providers (Hetzner/Vultr), whose address is
447 /// resolved from the provider API / mesh at provision time.
448 #[serde(default, skip_serializing_if = "Option::is_none")]
449 pub connect: Option<ConnectSpec>,
450 /// Static node capacity (R572-F3). Declares the node's total hardware
451 /// budget; F5's scheduler subtracts committed workload requests from this
452 /// to check whether a new workload fits. Absent means unconstrained.
453 #[serde(default, skip_serializing_if = "Option::is_none")]
454 pub allocatable: Option<NodeAllocatable>,
455 /// Placement taint keys (R572-F3). A repelling key blocks placement by
456 /// default, and a placement opts back in by naming that exact key in
457 /// [`RequiredSpec::tolerates`] (R876-B7).
458 ///
459 /// The `unless` is real now. It was not between R742-T4 and R876-B7: the
460 /// spec side declared which archetypes it *was* rather than which taints it
461 /// tolerated, that field was `#[serde(skip)]`, and so every placement
462 /// declared as `required = {...}` in a mirror TOML read this list as empty
463 /// and could not be drained at all. See [`RequiredSpec::tolerates`].
464 ///
465 /// A key in this list influences placement in exactly one of two ways, and
466 /// [`taint_effect`] is the authority on which:
467 ///
468 /// - **repulsion** — `"no-server"` / `"no-appliance"` / `"no-job"` reject
469 /// any placement that does not tolerate them. The archetype in the key is
470 /// now vocabulary rather than a filter: `matches` does not compare it
471 /// against the workload's class, it checks the toleration list, and
472 /// [`admission_spec`] is what turns a workload's class into the
473 /// tolerations that reproduce the old archetype-scoped behaviour;
474 /// - **affinity** — a key in [`AFFINITY_TAINT_KEYS`] (today just
475 /// `"public-ip"`) that a workload names in
476 /// `yah.placement.requires-taint`, which then *requires* this node.
477 ///
478 /// Anything else is **inert**: it parses, it round-trips, and no scheduler
479 /// decision can ever read it. `yah cloud validate` rejects such keys
480 /// (`validate::check_inert_taints`) rather than letting them sit looking
481 /// load-bearing — which is how `no-voter` spent months asserting a
482 /// falsehood on three nodes. Facts about a node that are not placement
483 /// inputs belong in [`mesh_tags`](Self::mesh_tags) or a comment.
484 #[serde(default, skip_serializing_if = "Vec::is_empty")]
485 pub taints: Vec<String>,
486 /// Which consensus group this node belongs to — W305/R742-F1. `None` means
487 /// standalone: in no group at all, which is us-west-002 and us-west-015.
488 ///
489 /// Membership is not by itself quorum eligibility; that is
490 /// [`sovereign_role`](Self::sovereign_role), added by R605-F12 because
491 /// us-west-003 is in prod's blast radius *and* must never vote in it.
492 ///
493 /// **Not a placement input.** It is deliberately absent from
494 /// [`RequiredSpec::matches`], and adding it there would be a category
495 /// error: a sovereign group is a *blast radius*, not a filter. Nothing
496 /// about "which quorum does this box vote in" should decide where a
497 /// workload runs — that is what made the fleet express three unrelated
498 /// properties through one taint list and get all three wrong (W305).
499 ///
500 /// What it *is* for is refusal. [`judge_join`] answers "may this node join
501 /// that node's cluster", and the answer is no unless both declare the same
502 /// group. Before this field the only guard was a comment in three machine
503 /// TOMLs saying "never run a raft join against this box from a shell
504 /// pointed at prod" — habit, with no mechanism behind it, which is the
505 /// same class of guard W257 §8 admitted to.
506 ///
507 /// # Why `sovereign_group` and not `raft_group`
508 ///
509 /// Raft is today's mechanism (operator, 2026-08-10). A field named for the
510 /// mechanism goes stale the day the mechanism is swapped, and every
511 /// consumer that reads it inherits the lie. `sovereign` names what the
512 /// group *has* — its own authority, its own upgrade cadence, its own
513 /// destruction — which stays true under any consensus protocol.
514 ///
515 /// Note the word already appears in this tree as prose (W267's title, the
516 /// `IngressProvider::Passway` doc comment's "sovereign edge"). That is an
517 /// adjective meaning "self-hosted, not SaaS"; this is the first time it
518 /// carries structure.
519 #[serde(default, skip_serializing_if = "Option::is_none")]
520 pub sovereign_group: Option<String>,
521 /// Whether this node may hold a seat in its group's quorum — R605-F12.
522 /// Meaningless without [`sovereign_group`](Self::sovereign_group): a
523 /// standalone box has no quorum to be eligible for.
524 ///
525 /// **`None` is "not written", not a third role.** Read it through
526 /// [`sovereign_membership`](Self::sovereign_membership), which resolves the
527 /// absence to [`SovereignRole::Voter`] — what declaring a group has always
528 /// meant, so the six nodes stamped before this field keep their seats
529 /// without an edit. The distinction is kept only so
530 /// [`crate::validate::check_unroled_sovereign_members`] can tell an
531 /// operator who *chose* voter from one who never considered the question;
532 /// no join decision reads the `Option` directly.
533 ///
534 /// # Why this is not a taint
535 ///
536 /// It was, once: `no-voter` sat in [`taints`](Self::taints) on three nodes
537 /// for months, read by nothing, and R742-T4 removed it because the taint
538 /// list is a *placement* vocabulary and this is not a placement input (see
539 /// [`taint_effect`]). Nor is it a second group label. It is a modifier on
540 /// the membership this node already declares, which is why it lives beside
541 /// the group and is judged with it in one predicate,
542 /// [`workload_spec::sovereign::join_permitted`].
543 #[serde(default, skip_serializing_if = "Option::is_none")]
544 pub sovereign_role: Option<SovereignRole>,
545 /// `[registration]` — the observed half (R707-T1). Empty until the box has
546 /// been attached / mesh-joined. See [`MachineRegistration`].
547 #[serde(default, skip_serializing_if = "MachineRegistration::is_empty")]
548 pub registration: MachineRegistration,
549}
550
551/// True iff `provider` has an auto-provision driver (create/destroy via API).
552/// Driver-backed providers require `location` + `server_type`; BYO `static`
553/// nodes (brought up over SSH) do not. The cloud-vs-vps distinction the fleet
554/// cares about lives here — at the provider-capability layer — not as a
555/// separate machine type (W242 BYO Phase-0 decision).
556pub fn provider_has_machine_driver(provider: &str) -> bool {
557 matches!(provider, "hetzner" | "vultr" | "digitalocean")
558}
559
560/// Taint keys a workload may name in `yah.placement.requires-taint` to
561/// *require* a node (W305/R742-T4 affinity vocabulary).
562///
563/// This is a closed list on purpose. `WorkloadSpec::requires_taint` returns
564/// free text, but every producer in the tree is code — `passway_ingress.rs`
565/// and `cloudflared_ingress.rs`, both emitting
566/// [`workload_spec::PUBLIC_IP_TAINT`] — and no on-disk `workload.toml` sets the
567/// annotation at all. So the set of keys a node can usefully carry for
568/// affinity is knowable at compile time, which is what lets
569/// [`taint_effect`] call anything outside it inert instead of guessing.
570///
571/// **Adding an affinity key means adding it here**, in the same change that
572/// teaches a workload to require it. That coupling is the point: it makes the
573/// node side and the workload side impossible to land apart.
574pub const AFFINITY_TAINT_KEYS: &[&str] = &[workload_spec::PUBLIC_IP_TAINT];
575
576/// How a key in [`MachineConfig::taints`] can affect placement.
577///
578/// W305 finding 1: before R742-T4 nothing asked this question, so a key that
579/// no scheduler path could read — `"qa"`, `"no-voter"` — parsed, validated,
580/// and quietly did nothing. Both of the findings that cost real fleet state
581/// were invisible for exactly that reason.
582#[derive(Debug, Clone, Copy, PartialEq, Eq)]
583pub enum TaintEffect {
584 /// `"no-<archetype>"`: rejects placement outright unless the constraint
585 /// names this key in [`RequiredSpec::tolerates`]. Read by
586 /// [`RequiredSpec::matches`], which walks `machine.taints` and classifies
587 /// each key through [`taint_effect`] (R876-B7).
588 Repels(LifecycleArchetype),
589 /// A key in [`AFFINITY_TAINT_KEYS`]: a workload naming it in
590 /// `yah.placement.requires-taint` is restricted to nodes carrying it.
591 Attracts,
592 /// Neither. No placement decision can read this key.
593 Inert,
594}
595
596/// Classify one node taint key. See [`TaintEffect`].
597///
598/// The repulsion half is derived from [`LifecycleArchetype::ALL`] rather than
599/// a literal list, so a fourth archetype makes `no-<its key>` live without an
600/// edit here.
601pub fn taint_effect(key: &str) -> TaintEffect {
602 if let Some(arch) = LifecycleArchetype::ALL
603 .into_iter()
604 .find(|a| key == format!("no-{}", a.taint_key()))
605 {
606 return TaintEffect::Repels(arch);
607 }
608 if AFFINITY_TAINT_KEYS.contains(&key) {
609 return TaintEffect::Attracts;
610 }
611 TaintEffect::Inert
612}
613
614/// Every key the scheduler *can* act on, sorted — for error messages that
615/// tell the operator what the legal vocabulary actually is instead of only
616/// what was wrong.
617pub fn live_taint_keys() -> Vec<String> {
618 let mut keys: Vec<String> = LifecycleArchetype::ALL
619 .into_iter()
620 .map(|a| format!("no-{}", a.taint_key()))
621 .chain(AFFINITY_TAINT_KEYS.iter().map(|k| (*k).to_string()))
622 .collect();
623 keys.sort();
624 keys
625}
626
627/// What [`judge_join`] decided about one proposed cluster join.
628///
629/// Shaped like yubaba's `PromotionVerdict` / `GeographyVerdict` and for the
630/// same reason: the rule stays unit-testable without a live cluster, and a
631/// refusal carries its reason from the place that knows it.
632#[derive(Debug, Clone, PartialEq, Eq)]
633pub enum JoinVerdict {
634 /// Both nodes declare the same sovereign group and both are voters. The
635 /// join is within one blast radius and grows a quorum both sides are
636 /// eligible for.
637 Permit,
638 /// The join is refused. Carries an operator-readable reason naming both
639 /// declared values and the file to edit — a refusal that only says
640 /// "invalid" gets worked around rather than fixed.
641 Refuse(String),
642}
643
644/// May `joiner` join the cluster `target` belongs to? — W305/R742-F1.
645///
646/// **A join is permitted iff both nodes declare the same non-`None`
647/// [`sovereign_group`](MachineConfig::sovereign_group) and both are
648/// [`SovereignRole::Voter`].** One rule, no special cases, and it makes the
649/// declaration mandatory before any quorum grows.
650///
651/// The case this exists for is two *different* declared groups: joining a dev
652/// Pi into prod is refused rather than trusted, where today the only guard is
653/// a comment saying not to do it. But an undeclared node is refused too, and
654/// that is the deliberate half — `None` means "in no group", not "unknown", so
655/// growing prod with an unstamped box is exactly as much a cross-group join as
656/// the dev case is. Failing open there would leave the operator believing a
657/// guarantee that was never evaluated, which is the reasoning
658/// `QuorumGeography::judge` already applies to untagged voters.
659///
660/// No legitimate flow pays for that strictness: prod and dev are both stamped,
661/// and us-west-002/015 are deliberately in no group at all. Adding a real
662/// member means declaring it first, which is the point.
663///
664/// # The non-voting refusal (R605-F12)
665///
666/// Same group and still refused, when either side declares
667/// [`SovereignRole::NonVoter`]. This is the case a group label alone could not
668/// express. us-west-003 is a residential-uplink build box the operator counts
669/// as part of prod — same secrets, same upgrade cadence, same destruction — and
670/// which must never hold a prod raft seat, because a home-internet partition
671/// should not be able to stall the quorum. Until R605-F12 the only thing
672/// refusing it was its *absent* stamp, so recording the operator's real intent
673/// (`sovereign_group = "prod"`) would have removed the guard. Now the intent
674/// and the guard are the same two lines.
675///
676/// Note what this is not: the refusal here is about *voting*, and it says
677/// nothing about the mesh. One mesh spans the whole fleet regardless of group
678/// or role (operator, 2026-08-19); a non-voter is reachable, schedulable and
679/// rollable like any other node.
680///
681/// This is the **camp-side** rendering of the rule. The predicate itself lives
682/// in [`workload_spec::sovereign::join_permitted`] because yubaba's
683/// `POST /raft/add-learner` gate asks the same question and cannot see this
684/// crate (there is deliberately no yubaba → cloud edge). Only the prose is
685/// duplicated, and it has to be: a refusal here names
686/// `.yah/infra/machines/<name>.toml`, while the node-side one has no machine
687/// name in hand and must also name `yubaba serve --sovereign-group`.
688///
689/// The node-side gate is *narrower* on purpose, and the difference is worth
690/// knowing when reading either: a daemon started without `--sovereign-group`
691/// has declared nothing rather than declared standalone, so yubaba resolves
692/// that unknown before it judges, and its gate is in force only once the
693/// cluster being joined declares a group. See `yubaba::sovereign_group`.
694pub fn judge_join(joiner: &MachineConfig, target: &MachineConfig) -> JoinVerdict {
695 let stamp_hint = |m: &MachineConfig| {
696 format!(
697 "declare `sovereign_group = \"<group>\"` in .yah/infra/machines/{}.toml",
698 m.name
699 )
700 };
701 let role_hint = |m: &MachineConfig| {
702 format!(
703 "set `sovereign_role = \"voter\"` in .yah/infra/machines/{}.toml",
704 m.name
705 )
706 };
707 if workload_spec::sovereign::join_permitted(
708 joiner.sovereign_membership(),
709 target.sovereign_membership(),
710 ) {
711 return JoinVerdict::Permit;
712 }
713 let (j, t) = (
714 joiner.sovereign_group.as_deref(),
715 target.sovereign_group.as_deref(),
716 );
717 // Everything below is a refusal; the only permitted shape returned above.
718 //
719 // R605-F12: when both sides name the SAME group, the role is the only thing
720 // left that can have refused, and it gets its own message. Falling through
721 // to the arms below would print "cross-group join refused: 'us-west-003' is
722 // in "prod" and 'us-west-001' is in "prod"" — a message that reads as a bug
723 // in the check rather than a decision about the fleet.
724 //
725 // Deliberately not hoisted above the group comparison. A non-voting joiner
726 // whose target is standalone is refused for *both* reasons, and naming the
727 // role there would send the operator to fix a field that would not have
728 // made the join legal anyway.
729 if let (Some(a), Some(b)) = (j, t) {
730 if a == b {
731 for (m, side, other) in [
732 (joiner, "the joiner", &target.name),
733 (target, "the target", &joiner.name),
734 ] {
735 if m.sovereign_membership().role.is_voter() {
736 continue;
737 }
738 return JoinVerdict::Refuse(format!(
739 "join refused: {side} '{}' is a NON-VOTING member of sovereign group {a:?}, \
740 the same group as '{other}'. It is inside that blast radius — same secrets, \
741 same upgrade cadence, same destruction — but declares itself ineligible for \
742 the quorum, so this is refused by declaration rather than by omission. If it \
743 should genuinely vote, {}; if it should not, this refusal is the field doing \
744 its job and the join is the thing to reconsider.",
745 m.name,
746 role_hint(m),
747 ));
748 }
749 }
750 }
751 match (j, t) {
752 (Some(a), Some(b)) => JoinVerdict::Refuse(format!(
753 "cross-group join refused: '{}' is in sovereign group {a:?} and '{}' is in {b:?}. \
754 These are separate blast radii — separate quorums, separate upgrade cadences, \
755 separately destroyable — and merging them is not something a join can undo. If \
756 the move is genuinely intended, restamp '{}' to {b:?} first and treat it as \
757 leaving its old group.",
758 joiner.name,
759 target.name,
760 joiner.name,
761 )),
762 (None, Some(b)) => JoinVerdict::Refuse(format!(
763 "join refused: '{}' declares no sovereign_group, so it is standalone — in no \
764 group — while '{}' is in {b:?}. That is a cross-group join, not an unchecked \
765 one. To make '{}' a member of {b:?}, {}.",
766 joiner.name,
767 target.name,
768 joiner.name,
769 stamp_hint(joiner),
770 )),
771 (Some(a), None) => JoinVerdict::Refuse(format!(
772 "join refused: '{}' is in sovereign group {a:?} but '{}' declares none, so the \
773 target is standalone and has no group to join. Either {}, or found the group on \
774 '{}' rather than growing it.",
775 joiner.name,
776 target.name,
777 stamp_hint(target),
778 joiner.name,
779 )),
780 (None, None) => JoinVerdict::Refuse(format!(
781 "join refused: neither '{}' nor '{}' declares a sovereign_group, so this join \
782 would form a group nobody declared and nothing could later reason about. Name \
783 the group on both boxes first: {}, and the same for '{}'.",
784 joiner.name,
785 target.name,
786 stamp_hint(joiner),
787 target.name,
788 )),
789 }
790}
791
792impl MachineConfig {
793 /// This node's declared place in a sovereign group, as the shared join rule
794 /// wants it — R605-F12.
795 ///
796 /// The one place `sovereign_role`'s `None` is resolved. Absence means
797 /// [`SovereignRole::Voter`], which is what declaring a group meant before
798 /// the role existed; resolving it here rather than at each call site is what
799 /// keeps the camp-side and node-side gates from disagreeing about a node
800 /// that never wrote the field.
801 pub fn sovereign_membership(&self) -> Membership<'_> {
802 Membership {
803 group: self.sovereign_group.as_deref(),
804 role: self.sovereign_role.unwrap_or_default(),
805 }
806 }
807
808 /// Provider DC code, or `""` when omitted (static nodes). Most readers want
809 /// a `&str`; the driver-backed provision/status paths still go through
810 /// [`validate`](Self::validate) which guarantees presence for those.
811 pub fn location(&self) -> &str {
812 self.location.as_deref().unwrap_or("")
813 }
814
815 /// Provider SKU, or `""` when omitted (static nodes).
816 pub fn server_type(&self) -> &str {
817 self.server_type.as_deref().unwrap_or("")
818 }
819
820 /// Enforce the provisioning-only-field contract: a machine whose provider
821 /// has an auto-provision driver MUST declare `location` + `server_type`
822 /// (the driver can't create a server without them). Static nodes may omit
823 /// both. Call this before any provision/diff that assumes a driver.
824 pub fn validate(&self) -> Result<()> {
825 if provider_has_machine_driver(&self.provider) {
826 if self.location.is_none() {
827 anyhow::bail!(
828 "machine '{}' (provider '{}') has an auto-provision driver but no `location`",
829 self.name,
830 self.provider
831 );
832 }
833 if self.server_type.is_none() {
834 anyhow::bail!(
835 "machine '{}' (provider '{}') has an auto-provision driver but no `server_type`",
836 self.name,
837 self.provider
838 );
839 }
840 }
841 Ok(())
842 }
843
844 /// Declared taints that no placement decision can read (W305/R742-T4).
845 ///
846 /// Deliberately **not** folded into [`validate`](Self::validate): that
847 /// guard runs on the provision/diff hot path and answers a different
848 /// question (can the driver create this server). An inert taint is a lint
849 /// — it never breaks an operation in flight, it just means the file is
850 /// asserting something the scheduler will not honour. `yah cloud validate`
851 /// is where the operator asks for that judgement; see
852 /// [`crate::validate::check_inert_taints`].
853 pub fn inert_taints(&self) -> Vec<&str> {
854 self.taints
855 .iter()
856 .filter(|t| taint_effect(t) == TaintEffect::Inert)
857 .map(String::as_str)
858 .collect()
859 }
860
861 /// Yubaba's TOFU'd hostkey fingerprint, from `[registration]` and falling
862 /// back to the pre-R707-T1 top-level field. **The only read path** — a
863 /// caller that reaches for `legacy_hostkey_fingerprint` directly sees
864 /// `None` on every migrated machine.
865 pub fn hostkey_fingerprint(&self) -> Option<&str> {
866 self.registration
867 .hostkey_fingerprint
868 .as_deref()
869 .or(self.legacy_hostkey_fingerprint.as_deref())
870 }
871
872 /// Record (or clear) the observed hostkey fingerprint. Writes
873 /// `[registration]` and drops any pre-R707-T1 top-level value, so the two
874 /// locations can never disagree after a writeback.
875 pub fn set_hostkey_fingerprint(&mut self, fingerprint: Option<String>) {
876 self.registration.hostkey_fingerprint = fingerprint;
877 self.legacy_hostkey_fingerprint = None;
878 }
879
880 /// Mesh (tailnet) IPv4 for this node, or `None` pre-mesh.
881 ///
882 /// Prefers `[registration].mesh_ipv4`; falls back to the host of a legacy
883 /// `[connect].yubaba` URL when that host is in the `100.64.0.0/10` CGNAT
884 /// range the mesh uses. A loopback placeholder (`http://127.0.0.1:7443`,
885 /// meaning "pre-mesh, reachable only through an SSH tunnel") is *not* a
886 /// mesh address and yields `None`.
887 pub fn mesh_ipv4(&self) -> Option<&str> {
888 if let Some(ip) = self.registration.mesh_ipv4.as_deref() {
889 return Some(ip);
890 }
891 let url = self.connect.as_ref()?.yubaba.as_deref()?;
892 mesh_ipv4_from_url(url)
893 }
894
895 /// Base URL for this node's yubaba, or `None` when no reach resolves.
896 ///
897 /// Thin wrapper over [`reach`](Self::reach) for the many call sites that
898 /// only branch on presence. Prefer `reach` anywhere the operator sees the
899 /// outcome — a `None` here throws away a refusal that names exactly which
900 /// address is missing.
901 pub fn yubaba_url(&self) -> Option<String> {
902 self.reach().ok()
903 }
904
905 /// The **one** address automation dials for this node — mesh-only.
906 ///
907 /// `Err` is a *named refusal*, not an absence: a node with no mesh address
908 /// is unresolvable to every automated path, and R605-T10's whole complaint
909 /// is that this used to surface as a connect timeout against an address the
910 /// caller has no route to.
911 ///
912 /// Resolution order:
913 ///
914 /// 1. A declared `[connect].yubaba` on a **private** host (10/8,
915 /// 172.16/12, 192.168/16) is **not dialed** — see below.
916 /// 2. Any other declared `[connect].yubaba` wins verbatim. That includes
917 /// the pre-mesh loopback placeholder (`http://127.0.0.1:7443`, "I have
918 /// no mesh address; reach me through the SSH tunnel to `ssh`"), which is
919 /// a genuine declaration and stays honoured.
920 /// 3. Otherwise `[registration].mesh_ipv4` composed with
921 /// `[connect].yubaba_port`.
922 ///
923 /// **Why a LAN literal loses (R605-T10, operator 2026-08-19).** The LAN
924 /// address is an emergency break-glass route, never an official one, and
925 /// automation must ALWAYS assume the caller is not on that LAN — this camp
926 /// sits on 192.168.22.0/22 with no route to the fleet's 192.168.10.0/24 at
927 /// all. Writing one into the field every resolver dials does not sit beside
928 /// the mesh route, it *overrides* it: R707-T6 made a declared literal beat
929 /// `mesh_ipv4` outright, so us-west-011 (mesh-joined, healthy) was elected
930 /// for every aarch64 build and then dialed at an address that answers only
931 /// from inside bldg-2506.
932 ///
933 /// **What R707-T6 wanted is preserved elsewhere.** Its forcing case was
934 /// identity, not reach: the dev raft group advertises LAN addrs
935 /// (`192.168.10.11:7443`, verified live off `/raft/status` 2026-08-27), and
936 /// `rollout::yubaba::membership_to_nodes` has to map those back to declared
937 /// machines. That match now runs against [`lan_endpoint`](Self::lan_endpoint),
938 /// which is composed from the break-glass `[connect].address` metadata and
939 /// is never dialed — so the two concerns the old precedence rule fused are
940 /// split, and the literal can stop squatting a dialed field.
941 ///
942 /// The LAN address itself STAYS in the machine TOML. It is useful metadata
943 /// and the manual `ssh` path is entitled to it; it is only disconnected
944 /// from every automated process.
945 pub fn reach(&self) -> Result<String, String> {
946 let Some(connect) = self.connect.as_ref() else {
947 return Err(format!(
948 "machine {:?} declares no [connect] block, so nothing knows how to reach it \
949 \u{2192} declare one, or leave it unprovisioned and out of placement",
950 self.name
951 ));
952 };
953 let mesh = || {
954 self.registration
955 .mesh_ipv4
956 .as_deref()
957 .map(|ip| format!("http://{ip}:{}", connect.yubaba_port()))
958 };
959 if let Some(literal) = &connect.yubaba {
960 let Some(lan) = private_ipv4_from_url(literal) else {
961 return Ok(literal.clone());
962 };
963 return mesh().ok_or_else(|| {
964 format!(
965 "machine {:?} is unresolvable to automation: its only declared yubaba reach \
966 is the private literal {:?} and it has no [registration].mesh_ipv4\n\
967 \u{2192} a LAN address is an emergency break-glass route, never an official \
968 one (R605-T10) — every automated path assumes the caller is NOT on {}/24\n\
969 \u{2192} mesh-join the box and record `mesh_ipv4` under [registration], then \
970 delete `[connect].yubaba` so the port composes with it",
971 self.name,
972 literal,
973 lan.rsplit_once('.').map(|(net, _)| net).unwrap_or(lan),
974 )
975 });
976 }
977 mesh().ok_or_else(|| {
978 format!(
979 "machine {:?} has no [registration].mesh_ipv4 and declares no \
980 [connect].yubaba, so no automated path can reach it\n\
981 \u{2192} mesh-join the box and record its tailnet address, or taint it out of \
982 placement — do not point `[connect].yubaba` at a LAN address (R605-T10)",
983 self.name
984 )
985 })
986 }
987
988 /// The LAN `host:port` this node's yubaba answers on, composed from the
989 /// break-glass `[connect].address` metadata plus the declared port.
990 ///
991 /// **Identity only — never dial this.** It exists so a raft membership
992 /// entry that names a node by its LAN address can be mapped back to the
993 /// declared machine (`rollout::yubaba::membership_to_nodes`) without that
994 /// address having to live in a field a resolver reads. `None` when the
995 /// machine is unprovisioned.
996 pub fn lan_endpoint(&self) -> Option<String> {
997 let connect = self.connect.as_ref()?;
998 Some(format!("{}:{}", connect.address, connect.yubaba_port()))
999 }
1000
1001 /// Fold the pre-R707-T1 top-level `hostkey_fingerprint` into
1002 /// `[registration]`, and lift a mesh IP out of a legacy `[connect].yubaba`
1003 /// URL. Idempotent; a machine already on the split shape is untouched.
1004 ///
1005 /// [`save`](Self::save) calls this, so writing a machine TOML migrates it
1006 /// rather than round-tripping the old shape back out.
1007 pub fn normalize(&mut self) {
1008 if let Some(fp) = self.legacy_hostkey_fingerprint.take() {
1009 self.registration.hostkey_fingerprint.get_or_insert(fp);
1010 }
1011 if self.registration.mesh_ipv4.is_none() {
1012 if let Some(ip) = self
1013 .connect
1014 .as_ref()
1015 .and_then(|c| c.yubaba.as_deref())
1016 .and_then(mesh_ipv4_from_url)
1017 .map(str::to_string)
1018 {
1019 self.registration.mesh_ipv4 = Some(ip);
1020 // The URL was pure derivation from mesh IP + port; keep only
1021 // the declared half so the two can't drift apart.
1022 if let Some(c) = self.connect.as_mut() {
1023 c.yubaba = None;
1024 }
1025 }
1026 }
1027 }
1028
1029 /// Persist to `<cloud_dir>/machines/<name>.toml`, creating the dir if needed.
1030 ///
1031 /// ⚠ Serializes the struct, so **operator comments in the target file are
1032 /// lost**. Pre-existing behaviour, not introduced here, but it is why
1033 /// registration writeback (`yah cloud machine attach`) goes through
1034 /// [`crate::state::MachineState`] and the comment-preserving path in the
1035 /// CLI rather than calling this on a hand-authored inventory file.
1036 pub fn save(&self, cloud_dir: &Path) -> Result<()> {
1037 let dir = cloud_dir.join("machines");
1038 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
1039 let path = dir.join(format!("{}.toml", self.name));
1040 let mut normalized = self.clone();
1041 normalized.normalize();
1042 let s = toml::to_string_pretty(&normalized)
1043 .with_context(|| format!("serializing machine {}", self.name))?;
1044 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
1045 }
1046}
1047
1048/// Host of an `http://host:port` URL iff it is a mesh (headscale) IPv4 in the
1049/// `100.64.0.0/10` CGNAT range. String-level rather than URL-parsed: the
1050/// inventory format is stable and this crate carries no URL dependency (same
1051/// reasoning as `fleet_metrics::extract_host` and
1052/// `hub::coordinator::is_loopback_url`).
1053fn mesh_ipv4_from_url(url: &str) -> Option<&str> {
1054 let host = ipv4_host_of(url)?;
1055 let ip: std::net::Ipv4Addr = host.parse().ok()?;
1056 let [a, b, ..] = ip.octets();
1057 // 100.64.0.0/10 ⇒ first octet 100, second octet 64..=127.
1058 (a == 100 && (64..=127).contains(&b)).then_some(host)
1059}
1060
1061/// Host of an `http://host:port` URL iff it is an **RFC1918 private** IPv4 —
1062/// `10/8`, `172.16/12`, `192.168/16`. `None` for anything else, loopback and
1063/// the `100.64/10` mesh range included: neither is a LAN literal.
1064///
1065/// The judgement R605-T10 turns on. A private literal is only ever reachable
1066/// from inside one building, so it is metadata about where the box physically
1067/// sits and never an address automation may dial — see
1068/// [`MachineConfig::reach`] and [`crate::validate::check_lan_dial_targets`].
1069pub fn private_ipv4_from_url(url: &str) -> Option<&str> {
1070 let host = ipv4_host_of(url)?;
1071 is_private_ipv4(host).then_some(host)
1072}
1073
1074/// Whether a bare host string is an RFC1918 private IPv4 literal.
1075pub fn is_private_ipv4(host: &str) -> bool {
1076 let Ok(ip) = host.parse::<std::net::Ipv4Addr>() else {
1077 return false;
1078 };
1079 ip.is_private()
1080}
1081
1082/// Bare host of a `[scheme://]host[:port][/path]` string.
1083fn ipv4_host_of(url: &str) -> Option<&str> {
1084 let after_scheme = url.split("://").nth(1).unwrap_or(url);
1085 after_scheme.split(['/', ':']).next()
1086}
1087
1088/// Declared **reach** for a BYO `static` node (no provider API). Lives under
1089/// `[connect]` in the machine TOML.
1090///
1091/// Reach only — how the camp gets to the box. *Permission* is a separate axis
1092/// that belongs to cheers' scopes (W295 §"Deliberately deferred"); the two
1093/// collapse in practice today (mesh membership grants everything) and the data
1094/// model must not fuse them, so do not add an authorization field here.
1095///
1096/// `address`, `ssh` and `identity_file` stay whole, literal, operator-authored
1097/// strings even though their values often *look* derived. They are not:
1098/// us-west-001 dials SSH over its public IP while us-west-002 was deliberately
1099/// repointed at its tailnet IP (R608-F10) precisely because the LAN address is
1100/// unreachable off-LAN. Decomposing them into user + host and recomposing
1101/// would silently undo per-machine decisions like that one. `yubaba` is the
1102/// field that *was* derived — mesh IP plus a fixed port, rewritten by
1103/// mesh-join — so that is where R707-T1 cut.
1104#[derive(Debug, Clone, Serialize, Deserialize)]
1105#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
1106pub struct ConnectSpec {
1107 /// Reachable IPv4/host for the box, e.g. `"45.32.194.254"`. Declared: which
1108 /// of a machine's several addresses the camp should use is an operator
1109 /// choice (public IP vs. LAN IP vs. tailnet IP).
1110 pub address: String,
1111 /// SSH target the camp dials for bootstrap + (pre-mesh) tunneled deploys,
1112 /// e.g. `"root@45.32.194.254"` or `"struc@100.64.0.4"`. Declared, whole —
1113 /// see the type doc. Pair with `identity_file` for a copy-pasteable
1114 /// `ssh -i <identity_file> <ssh>`.
1115 pub ssh: String,
1116 /// Private key path the camp uses to authenticate `ssh`, e.g.
1117 /// `"~/.ssh/yah"`. Every node in the fleet uses the same operator key
1118 /// today, but this is declared per-machine rather than assumed globally
1119 /// for the same reason `ssh` is whole rather than decomposed: a future
1120 /// node with a different key should not have to fight a hardcoded
1121 /// default. `~` is not shell-expanded by this crate — callers that shell
1122 /// out to `ssh`/`scp` pass it through `-i`, which expands it itself.
1123 pub identity_file: String,
1124 /// Port yubaba listens on. Declared reach; defaults to 7443 when omitted,
1125 /// which is every machine in the fleet today. Composed with the *observed*
1126 /// [`MachineRegistration::mesh_ipv4`] by [`MachineConfig::yubaba_url`].
1127 #[serde(default, skip_serializing_if = "Option::is_none")]
1128 pub yubaba_port: Option<u16>,
1129 /// Explicit yubaba base URL, overriding the composed form.
1130 ///
1131 /// Two live uses, both genuine declarations: a pre-mesh node saying
1132 /// `"http://127.0.0.1:7443"` — "I have no mesh address; reach me through
1133 /// the SSH tunnel to `ssh`" — and any node whose yubaba is not at
1134 /// `mesh_ipv4:port`. A URL here whose host *is* a mesh IP is the
1135 /// pre-R707-T1 shape; [`MachineConfig::normalize`] lifts it into
1136 /// `[registration].mesh_ipv4` and clears this field so the two cannot
1137 /// drift apart.
1138 #[serde(default, skip_serializing_if = "Option::is_none")]
1139 pub yubaba: Option<String>,
1140}
1141
1142/// Default yubaba listen port, used when `[connect].yubaba_port` is omitted.
1143pub const DEFAULT_YUBABA_PORT: u16 = 7443;
1144
1145impl ConnectSpec {
1146 /// Declared yubaba port, defaulting to [`DEFAULT_YUBABA_PORT`].
1147 pub fn yubaba_port(&self) -> u16 {
1148 self.yubaba_port.unwrap_or(DEFAULT_YUBABA_PORT)
1149 }
1150}
1151
1152#[derive(Debug, Clone, Serialize, Deserialize)]
1153#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
1154pub struct BucketSpec {
1155 pub name: String,
1156 pub public_read: bool,
1157}
1158
1159/// Per-camp mirror declaration from `.yah/cloud/mirrors/<id>/mirror.toml`
1160/// (folder form) or the legacy `.yah/cloud/mirrors/<id>.toml` (flat form).
1161///
1162/// The folder form is preferred for new mirrors so that per-mirror secrets
1163/// and override files can sit next to `mirror.toml` without polluting the
1164/// top-level `mirrors/` directory.
1165#[derive(Debug, Clone, Serialize, Deserialize)]
1166pub struct LegacyMirrorConfig {
1167 /// Logical camp name this mirror hosts, e.g. `"yah"` or `"noisetable"`.
1168 ///
1169 /// Serialised as `camp`; accepts the legacy `rig` spelling for files that
1170 /// predate the R137 rig→camp rename (one-time migration: `sed -i ''
1171 /// 's/^rig = /camp = /' ~/.yah/cloud/mirrors/*.toml`).
1172 #[serde(rename = "camp", alias = "rig")]
1173 pub camp: String,
1174 pub regions: Vec<String>,
1175 /// Workload names deployed as part of this mirror (references `workloads/<name>.toml`).
1176 /// Renamed from `services` in R092-F1; use `yah cloud config migrate-services-to-workloads`
1177 /// on repos that still have the old `services/` layout.
1178 #[serde(alias = "services")]
1179 pub workloads: Vec<String>,
1180 /// Base domain for Cloudflare-fronted services on this mirror's machines.
1181 /// Combined with the machine's `location` to build virtual-host names:
1182 /// e.g. `cloud_domain = "cloud.noisetable.example"` on machine in location
1183 /// `pdx` → Caddyfile site address `pdx.cloud.noisetable.example`.
1184 /// Optional: if unset the Caddyfile falls back to `:port` listeners.
1185 #[serde(default, skip_serializing_if = "Option::is_none")]
1186 pub cloud_domain: Option<String>,
1187}
1188
1189/// Error from loading or validating a single workload TOML file.
1190#[derive(Debug, Error)]
1191pub enum WorkloadConfigError {
1192 #[error("reading {path}: {source}")]
1193 Io {
1194 path: String,
1195 source: std::io::Error,
1196 },
1197 #[error("parsing {path}: {source}")]
1198 Toml {
1199 path: String,
1200 source: toml::de::Error,
1201 },
1202 #[error("invalid WorkloadSpec in {path}: {source}")]
1203 Shape {
1204 path: String,
1205 source: validate::ShapeError,
1206 },
1207}
1208
1209/// A workload declaration loaded from `.yah/cloud/workloads/<name>.toml`.
1210///
1211/// Each file is the human-authored TOML serialization of a [`WorkloadSpec`].
1212/// On load, the spec is validated against the shape layer; failures surface as
1213/// a [`CloudConfigError::Workload`] with the file path and field path.
1214#[derive(Debug, Clone, Serialize, Deserialize)]
1215pub struct WorkloadConfig {
1216 /// The validated spec.
1217 #[serde(flatten)]
1218 pub spec: WorkloadSpec,
1219}
1220
1221impl WorkloadConfig {
1222 /// Persist to `<cloud_dir>/workloads/<name>.toml`, creating the dir if needed.
1223 pub fn save(&self, cloud_dir: &Path) -> Result<()> {
1224 let dir = cloud_dir.join("workloads");
1225 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
1226 let path = dir.join(format!("{}.toml", self.spec.name));
1227 let s = toml::to_string_pretty(self)
1228 .with_context(|| format!("serializing workload {}", self.spec.name))?;
1229 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
1230 }
1231}
1232
1233/// Error surfaced by [`CloudConfig::load`] when a workload TOML fails validation.
1234#[derive(Debug, Error)]
1235pub enum CloudConfigError {
1236 #[error(transparent)]
1237 Anyhow(#[from] anyhow::Error),
1238 #[error("workload validation failed: {0}")]
1239 Workload(WorkloadConfigError),
1240}
1241
1242/// Mirror-to-machine assignment table from `.yah/cloud/topology.toml`.
1243///
1244/// Declares which logical mirror names are assigned to which machines.
1245/// This is the source-canonical placement until yubaba raft observes it
1246/// (per the migration tracker in the arch doc).
1247#[derive(Debug, Clone, Serialize, Deserialize, Default)]
1248pub struct TopologyConfig {
1249 /// Mirror→machine assignments.
1250 #[serde(default)]
1251 pub assignments: Vec<MirrorAssignment>,
1252 /// Declared buckets, logged by `yah cloud bucket create`.
1253 /// Source-canonical until yubaba raft observes actual placement.
1254 #[serde(default, skip_serializing_if = "Vec::is_empty")]
1255 pub buckets: Vec<BucketLogEntry>,
1256}
1257
1258impl TopologyConfig {
1259 /// Load from a `topology.toml` file, returning `Default` when absent.
1260 pub fn load(path: &Path) -> Result<Self> {
1261 if !path.exists() {
1262 return Ok(Self::default());
1263 }
1264 let s =
1265 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
1266 toml::from_str(&s).with_context(|| format!("parsing {}", path.display()))
1267 }
1268
1269 /// Persist to `topology.toml`, creating parent dirs if needed.
1270 pub fn save(&self, path: &Path) -> Result<()> {
1271 if let Some(parent) = path.parent() {
1272 std::fs::create_dir_all(parent)
1273 .with_context(|| format!("creating {}", parent.display()))?;
1274 }
1275 let s = toml::to_string_pretty(self).context("serializing topology")?;
1276 std::fs::write(path, s).with_context(|| format!("writing {}", path.display()))
1277 }
1278
1279 /// Find a declared bucket by name.
1280 pub fn bucket_by_name(&self, name: &str) -> Option<&BucketLogEntry> {
1281 self.buckets.iter().find(|b| b.name == name)
1282 }
1283
1284 /// Find a mutable declared bucket by name.
1285 pub fn bucket_by_name_mut(&mut self, name: &str) -> Option<&mut BucketLogEntry> {
1286 self.buckets.iter_mut().find(|b| b.name == name)
1287 }
1288
1289 /// Returns true if the bucket is declared as cross-machine (no owning machine).
1290 pub fn is_cross_machine_bucket(&self, name: &str) -> bool {
1291 self.buckets
1292 .iter()
1293 .any(|b| b.name == name && b.machine.is_none())
1294 }
1295}
1296
1297/// One mirror→machine placement entry in `topology.toml`.
1298#[derive(Debug, Clone, Serialize, Deserialize)]
1299pub struct MirrorAssignment {
1300 /// Logical mirror name, e.g. `"noisetable-pdx"`.
1301 pub mirror: String,
1302 /// Machine that hosts this mirror, e.g. `"noisetable-pdx-1"`.
1303 pub machine: String,
1304}
1305
1306/// A bucket declaration logged in `topology.toml` by `yah cloud bucket create`.
1307#[derive(Debug, Clone, Serialize, Deserialize)]
1308pub struct BucketLogEntry {
1309 pub name: String,
1310 /// Machine that owns this bucket. `None` marks it as cross-machine
1311 /// (no single-machine ownership; requires an explicit declaration in
1312 /// `topology.toml` before `yah cloud bucket create` will proceed without
1313 /// `--machine`).
1314 #[serde(default, skip_serializing_if = "Option::is_none")]
1315 pub machine: Option<String>,
1316 /// Logical location of the bucket, e.g. `"pdx"`.
1317 pub location: String,
1318 /// Current declared policy: `"private"` | `"public-read"` | `"signed-only"`.
1319 #[serde(default = "default_bucket_policy")]
1320 pub policy: String,
1321}
1322
1323fn default_bucket_policy() -> String {
1324 "private".to_string()
1325}
1326
1327/// Per-service config from `.yah/cloud/services/<name>.toml`.
1328///
1329/// **Deprecated.** The `services/` layout was replaced by `workloads/` in R092-F1.
1330/// Kept to allow in-place reads for repos that haven't migrated yet; use
1331/// `yah cloud config migrate-services-to-workloads` to upgrade.
1332#[derive(Debug, Clone, Serialize, Deserialize)]
1333pub struct LegacyServiceConfig {
1334 pub name: String,
1335 pub image: String,
1336 pub version: String,
1337 #[serde(default)]
1338 pub env: HashMap<String, String>,
1339 #[serde(default)]
1340 pub ports: Vec<PortMapping>,
1341 #[serde(default)]
1342 pub mesh_only: bool,
1343 /// Network interface this service binds to exclusively (e.g. `"tailscale0"`).
1344 ///
1345 /// When set the compose renderer emits `network_mode: "host"` and the
1346 /// service is NOT joined to the shared compose bridge network. The service
1347 /// process must bind its listen socket to the named interface's IP — for
1348 /// Postgres this means setting `POSTGRES_LISTEN_ADDRESSES` to the node's
1349 /// `tailscale ip --4` output at first boot. See [`crate::mesh_service`] for
1350 /// the standard pg_hba.conf snippet and ufw rules to pair with this field.
1351 #[serde(default, skip_serializing_if = "Option::is_none")]
1352 pub bind_interface: Option<String>,
1353
1354 /// Tenant this service belongs to (W206 isolation axis). Absent in the
1355 /// service TOML → [`TenantId::singleton`], keeping single-tenant machines
1356 /// on one shared compose network. When a machine hosts services from two
1357 /// or more distinct tenants, the compose renderer (R558-T2) splits them
1358 /// into per-tenant `<tenant>-<tier>` networks so cross-tenant stacks on the
1359 /// same host are not bridged together.
1360 #[serde(default = "TenantId::singleton")]
1361 pub tenant: TenantId,
1362}
1363
1364#[derive(Debug, Clone, Serialize, Deserialize)]
1365pub struct PortMapping {
1366 pub host: u16,
1367 pub container: u16,
1368}
1369
1370/// A loaded service plus its per-environment mirrors.
1371///
1372/// Wraps the `service.toml` body and the directory of `mirrors/<env>.toml`
1373/// files that project the service onto concrete infra.
1374#[derive(Debug, Clone, Serialize, Deserialize)]
1375pub struct ServiceWithMirrors {
1376 pub service: ServiceConfig,
1377 /// Mirrors keyed by environment name (file stem of `mirrors/<env>.toml`).
1378 pub mirrors: BTreeMap<String, MirrorConfig>,
1379 /// Transform recipe names keyed by component id. Populated from each
1380 /// static-asset component's `workload.toml` at load time — not stored
1381 /// in service.toml. Only present for components that declare
1382 /// `[asset.derive.transform] recipe = "..."`.
1383 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
1384 pub component_transform_recipes: BTreeMap<String, String>,
1385 /// Nodes each mirror's passway front door is placed on, keyed by env —
1386 /// exactly what [`MirrorConfig::passway_machines`] returns, with the envs
1387 /// that declare no passway edge left out.
1388 ///
1389 /// Derived at load time like `component_transform_recipes` above: it is
1390 /// stored in no TOML file. It exists so that a consumer of this wire type —
1391 /// the desktop `service_list` command, and through it the Services tab's
1392 /// custom-domain panel — never reconciles the two `ingress` spellings
1393 /// itself. An env present here with an **empty** list is a passway edge
1394 /// whose placement is co-located rather than declared; see
1395 /// [`MirrorConfig::passway_machines`] for why that is a different answer
1396 /// from being absent.
1397 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
1398 pub passway_machines: BTreeMap<String, Vec<String>>,
1399}
1400
1401/// All cloud config loaded from a workspace root (the parent of `.yah/`).
1402///
1403/// Reads two trees:
1404/// - `.yah/infra/` — `machines/`, `providers/`
1405/// - `.yah/services/<svc>/` — `service.toml` + `mirrors/<env>.toml`
1406///
1407/// Pre-R215 fields (`legacy_mirrors`, `legacy_services`, `workloads`,
1408/// `topology`) are still populated from `.yah/cloud/` when present so
1409/// pre-R215 callers (compose.rs, bucket commands) keep compiling — they
1410/// just see empty collections in a post-B1 workspace where the legacy
1411/// data was deleted. These fields are scheduled for removal in B3-T3.
1412#[derive(Debug)]
1413pub struct CloudConfig {
1414 /// Workspace root that was loaded — useful for path-resolving
1415 /// component references on a [`ServiceComponent`].
1416 pub workspace_root: std::path::PathBuf,
1417
1418 // ─── R215+ tree ────────────────────────────────────────────────────────
1419 /// `.yah/infra/machines/<name>.toml`
1420 pub machines: Vec<MachineConfig>,
1421 /// `.yah/infra/providers/<id>.toml`
1422 pub providers: Vec<ProviderConfig>,
1423 /// Provenance for every entry in `machines` that came from a linked
1424 /// `.yah/infra/sources.toml` source rather than this camp's own
1425 /// `.yah/infra/machines/` (R615-F2 / W274). Keyed by
1426 /// [`MachineConfig::name`]; a name absent here is camp-local. Empty from
1427 /// [`CloudConfig::load_from_config_dir`] — see its doc for why sources
1428 /// don't apply to multi-root sibling trees.
1429 pub machine_origins: BTreeMap<String, InfraOrigin>,
1430 /// Same as [`machine_origins`](Self::machine_origins), keyed by
1431 /// [`ProviderConfig::id`].
1432 pub provider_origins: BTreeMap<String, InfraOrigin>,
1433 /// `.yah/services/<svc>/` — service.toml plus mirrors/<env>.toml.
1434 pub services: BTreeMap<String, ServiceWithMirrors>,
1435 /// `.yah/domains/<name>.toml` — public-facing routing manifests
1436 /// (R347). Single file per domain; no nested per-env tree because
1437 /// domains themselves aren't projected onto infra — they describe
1438 /// how a Worker bundle ingresses requests onto services.
1439 pub domains: BTreeMap<String, DomainConfig>,
1440
1441 // ─── Pre-R215 legacy (slated for removal in B3-T3) ────────────────────
1442 /// Legacy mirrors from `.yah/cloud/mirrors/`.
1443 pub legacy_mirrors: Vec<LegacyMirrorConfig>,
1444 /// Workloads from `.yah/cloud/workloads/*.toml` (R092-F1 schema).
1445 pub workloads: Vec<WorkloadConfig>,
1446 /// Topology from `.yah/cloud/topology.toml` (mirror→machine assignments).
1447 pub topology: TopologyConfig,
1448 /// Legacy services from `.yah/cloud/services/*.toml` (pre-R092 layout).
1449 pub legacy_services: Vec<LegacyServiceConfig>,
1450}
1451
1452impl CloudConfig {
1453 /// Load all cloud config rooted at `workspace_root` (the parent of `.yah/`).
1454 ///
1455 /// Reads the R215+ tree (`.yah/infra/`, `.yah/services/<svc>/`) eagerly
1456 /// and the pre-R215 `.yah/cloud/` tree opportunistically. Returns `Err`
1457 /// immediately if any TOML fails to parse or a workload TOML fails
1458 /// shape validation; the error includes the file path and field path.
1459 ///
1460 /// Cross-ref validation runs after both trees finish loading: every
1461 /// `mirror.providers.X.use = "<id>"` must resolve to a real provider
1462 /// declared under `.yah/infra/providers/`.
1463 ///
1464 /// R844-B7 — **a missing `.yah/` is a wrong-root error, not an empty
1465 /// fleet.** Every sub-loader below tolerates a missing directory by
1466 /// returning empty, so before this check a call against the wrong
1467 /// directory produced a perfectly valid `CloudConfig` with zero machines,
1468 /// zero services and zero providers. Nothing downstream can tell that
1469 /// apart from a camp that genuinely declares nothing, so the failure
1470 /// surfaces as an operation that silently does nothing to nothing: a
1471 /// collate that renders no backends, a fanout that asks no nodes, a
1472 /// rollout that plans against an empty fleet. It was found the hard way —
1473 /// a live-fleet test in `app/yah/cli` called this with `"."`, which under
1474 /// `cargo test` is the *package* root, and passed while measuring nothing.
1475 ///
1476 /// The line is drawn at `.yah/` and only there: a workspace whose
1477 /// `.yah/infra/machines/` is absent or empty is a real, if unusual, camp
1478 /// with an empty fleet and still loads. `unknown` is not `answered with
1479 /// none`.
1480 pub fn load(workspace_root: &Path) -> Result<Self> {
1481 let yah_dir = crate::paths::yah_dir(workspace_root);
1482 if !yah_dir.is_dir() {
1483 anyhow::bail!(
1484 "not a yah workspace: no {} — expected the camp root (the parent \
1485 of `.yah/`), got {}. This is a wrong-root error, not an empty \
1486 fleet; a camp with no machines declared still has a `.yah/`.",
1487 yah_dir.display(),
1488 workspace_root.display(),
1489 );
1490 }
1491
1492 let mut providers = load_providers(&crate::paths::providers_dir(workspace_root))?;
1493 let services = load_services(&crate::paths::services_dir(workspace_root), workspace_root)?;
1494 let domains = load_domains(&crate::paths::domains_dir(workspace_root))?;
1495
1496 Self::cross_ref_validate(&providers, &services, &domains)?;
1497
1498 // Legacy `.yah/cloud/` reads — empty in post-B1 workspaces. Wrapped in
1499 // a helper so a missing tree is silent (no error, no warning).
1500 let cloud_dir = crate::paths::legacy_cloud_dir(workspace_root);
1501 let (legacy_mirrors, legacy_workloads, topology, legacy_services) = if cloud_dir.exists() {
1502 (
1503 load_mirrors(cloud_dir.join("mirrors"))?,
1504 load_workloads(cloud_dir.join("workloads"))?,
1505 load_topology(cloud_dir.join("topology.toml"))?,
1506 load_dir::<LegacyServiceConfig>(cloud_dir.join("services"))?,
1507 )
1508 } else {
1509 Default::default()
1510 };
1511
1512 // Workloads come from `.yah/infra/workloads/` (R215+). R568-T7: before
1513 // that path was read here, this field was populated *only* from the
1514 // legacy tree above — which R222-B1 emptied — so `cfg.workload(name)`
1515 // resolved nothing in every post-R215 camp and `yah cloud workload
1516 // deploy` could not find any declaration at all. The bug survived
1517 // because the only workloads ever deployed were forge/QED runs, which
1518 // build their spec in memory and never come through here. Same
1519 // dedupe-by-name shape as machines below: R215+ wins.
1520 let mut workloads = load_workloads(crate::paths::workloads_dir(workspace_root))?;
1521 let workload_names: std::collections::HashSet<String> =
1522 workloads.iter().map(|w| w.spec.name.clone()).collect();
1523 for w in legacy_workloads {
1524 if !workload_names.contains(&w.spec.name) {
1525 workloads.push(w);
1526 }
1527 }
1528
1529 // R870-B13: machines are resolved by [`resolve_fleet_inventory`] —
1530 // camp-local, the pre-R215 legacy tree, and every machine borrowed
1531 // through `.yah/infra/sources.toml`, in that precedence. This used to
1532 // be spelled out inline here, which made `CloudConfig::load` the only
1533 // reader that saw borrowed machines at all; the two *resolution*
1534 // callers in `validate`/`reconciler::domain` read a camp-local-only
1535 // loader and could not see a borrowing camp's fleet. There is now one
1536 // implementation and three callers.
1537 let fleet = resolve_fleet_inventory(workspace_root)?;
1538
1539 // Providers overlay here rather than inside `resolve_fleet_inventory`:
1540 // that function answers "which machines does this camp have", which is
1541 // the question with three readers. Providers have exactly one reader —
1542 // this load — so hoisting them would build a seam nothing crosses.
1543 let mut provider_origins = BTreeMap::new();
1544 overlay_source_providers(
1545 workspace_root,
1546 &fleet.sources,
1547 &mut providers,
1548 &mut provider_origins,
1549 );
1550
1551 Ok(Self {
1552 workspace_root: workspace_root.to_path_buf(),
1553 machines: fleet.machines,
1554 providers,
1555 machine_origins: fleet.origins,
1556 provider_origins,
1557 services,
1558 domains,
1559 legacy_mirrors,
1560 workloads,
1561 topology,
1562 legacy_services,
1563 })
1564 }
1565
1566 /// Load the R215+ tree (`infra/`, `services/`, `domains/`) rooted at an
1567 /// arbitrary config directory instead of the hardcoded `.yah/`. This is the
1568 /// building block for multi-root deployments (W206 config layout (b), sibling
1569 /// `.noisetable/` trees) — see [`crate::multi_root`]. Part of R558-F4.
1570 ///
1571 /// `config_dir` is the `.X/` directory itself (e.g. `<parent>/.noisetable`);
1572 /// `workspace_root` remains the camp dir (the config dir's parent) so a
1573 /// component's `path` reference resolves against the same tree the classic
1574 /// [`CloudConfig::load`] uses. The legacy `.yah/cloud/` reads are skipped —
1575 /// multi-root deployments are post-R215 by construction — so `legacy_*`,
1576 /// `workloads`, and `topology` come back empty. Machines are read from
1577 /// `config_dir/infra/machines` directly (sibling trees declare their own
1578 /// inventory or none).
1579 ///
1580 /// R615-F2 decision, explicit rather than silent: **sources.toml overlay
1581 /// does NOT apply here.** This function
1582 /// exists specifically because a multi-root sibling tree (W206 layout
1583 /// (b), e.g. `.noisetable/`) is a *second config root inside the same
1584 /// camp*, not a second camp — `config_dir` is already wherever the
1585 /// caller decided this tree's infra lives, and `.yah/infra/sources.toml`
1586 /// (singular, tied to `paths::infra_dir(workspace_root)`) has no
1587 /// well-defined meaning for an arbitrary `config_dir` that isn't that
1588 /// path. A sibling tree that wants borrowed infra declares its own
1589 /// `sources.toml` under whichever root actually calls
1590 /// [`CloudConfig::load`] for it; `machine_origins`/`provider_origins`
1591 /// come back empty here, not wrong — there is nothing to overlay.
1592 pub fn load_from_config_dir(config_dir: &Path, workspace_root: &Path) -> Result<Self> {
1593 let providers = load_providers(&config_dir.join("infra").join("providers"))?;
1594 let services = load_services(&config_dir.join("services"), workspace_root)?;
1595 let domains = load_domains(&config_dir.join("domains"))?;
1596
1597 Self::cross_ref_validate(&providers, &services, &domains)?;
1598
1599 let machines = load_dir::<MachineConfig>(config_dir.join("infra").join("machines"))?;
1600
1601 Ok(Self {
1602 workspace_root: workspace_root.to_path_buf(),
1603 machines,
1604 providers,
1605 machine_origins: BTreeMap::new(),
1606 provider_origins: BTreeMap::new(),
1607 services,
1608 domains,
1609 legacy_mirrors: vec![],
1610 workloads: vec![],
1611 topology: TopologyConfig::default(),
1612 legacy_services: vec![],
1613 })
1614 }
1615
1616 /// Cross-reference validation shared by [`CloudConfig::load`] and
1617 /// [`CloudConfig::load_from_config_dir`]: every mirror `providers.X.use =
1618 /// "<id>"` must resolve to a declared provider, and every domain route's
1619 /// `component = "<service>/<component-id>"` must resolve to a real component.
1620 fn cross_ref_validate(
1621 providers: &[ProviderConfig],
1622 services: &BTreeMap<String, ServiceWithMirrors>,
1623 domains: &BTreeMap<String, DomainConfig>,
1624 ) -> Result<()> {
1625 // Mirror `use = "<id>"` slots must resolve to a declared provider.
1626 let provider_ids: std::collections::HashSet<&str> =
1627 providers.iter().map(|p| p.id.as_str()).collect();
1628 for (svc_name, svc) in services {
1629 for (env, mirror) in &svc.mirrors {
1630 for (slot, body) in &mirror.providers {
1631 if let Some(id) = body.provider_id() {
1632 if !provider_ids.contains(id) {
1633 anyhow::bail!(
1634 "services/{svc_name}/mirrors/{env}.toml: \
1635 providers.{slot}.use = \"{id}\" — no such provider; \
1636 declare it at infra/providers/{id}.toml"
1637 );
1638 }
1639 }
1640 }
1641 // An `[[ingress]]` edge's own `use` is the same kind of
1642 // reference (R845) and gets the same check: a typo there is
1643 // otherwise invisible until `yah cloud apply` reaches the
1644 // Cloudflare arm and fails on a missing provider file.
1645 for (idx, edge) in mirror.ingress_edge_slice().iter().enumerate() {
1646 if let Some(id) = edge.provider_id.as_deref() {
1647 if !provider_ids.contains(id) {
1648 anyhow::bail!(
1649 "services/{svc_name}/mirrors/{env}.toml: \
1650 ingress[{idx}].use = \"{id}\" — no such provider; \
1651 declare it at infra/providers/{id}.toml"
1652 );
1653 }
1654 }
1655 }
1656 }
1657 }
1658
1659 // R870-B11. Two bundle-tier components sharing a mount would stage
1660 // into the same `app/dist/<mount>/` prefix inside the service's one
1661 // assembled bundle and silently clobber each other on disk — the
1662 // exact failure class this ticket exists to fix, one level down
1663 // (there it was two components silently overwriting the same
1664 // *workload*; here it would be two components silently overwriting
1665 // the same *path inside* the workload). A mount is owned by exactly
1666 // one component; refuse the config before the clobber happens.
1667 for (svc_name, svc) in services {
1668 let mut owner_by_mount: BTreeMap<String, &str> = BTreeMap::new();
1669 for component in &svc.service.components {
1670 if component.kind != "mesofact-static" && component.kind != "mesofact-spa" {
1671 continue;
1672 }
1673 let mount = component
1674 .mount
1675 .as_deref()
1676 .map(normalize_mount)
1677 .unwrap_or_default();
1678 if let Some(existing) = owner_by_mount.insert(mount.clone(), &component.id) {
1679 let where_ = if mount.is_empty() {
1680 "the service root (no `mount`)".to_string()
1681 } else {
1682 format!("mount = \"/{mount}\"")
1683 };
1684 anyhow::bail!(
1685 "services/{svc_name}/service.toml: components \"{existing}\" and \
1686 \"{}\" both declare {where_} — a bundle-tier component's mount is a \
1687 storage prefix inside the service's single assembled bundle \
1688 (app/dist/<mount>/), so two components at the same mount would stage \
1689 into the same path and silently overwrite each other. Give one of \
1690 them a distinct `mount`.",
1691 component.id,
1692 );
1693 }
1694 }
1695 }
1696
1697 // Every domain route's `component = "<service>/<component-id>"` must
1698 // resolve to a real component.
1699 for (dom_name, dom) in domains {
1700 for (idx, route) in dom.routes.iter().enumerate() {
1701 let Some(component_ref) = route.mode.component() else {
1702 continue; // redirects don't reference components
1703 };
1704 let Some((svc_name, comp_id)) = split_component_ref(component_ref) else {
1705 anyhow::bail!(
1706 "domains/{dom_name}.toml: routes[{idx}].component = \
1707 \"{component_ref}\" — expected \"<service>/<component-id>\""
1708 );
1709 };
1710 let Some(svc) = services.get(svc_name) else {
1711 anyhow::bail!(
1712 "domains/{dom_name}.toml: routes[{idx}].component = \
1713 \"{component_ref}\" — no such service \"{svc_name}\" \
1714 under services/"
1715 );
1716 };
1717 let Some(component) = svc.service.components.iter().find(|c| c.id == comp_id)
1718 else {
1719 anyhow::bail!(
1720 "domains/{dom_name}.toml: routes[{idx}].component = \
1721 \"{component_ref}\" — service \"{svc_name}\" has no \
1722 component with id \"{comp_id}\""
1723 );
1724 };
1725
1726 // R746: a mounted component must be routed where it publishes.
1727 // The publisher writes its bundle under the mount and the front
1728 // door looks a request up by its own path, so a route path and
1729 // a mount that disagree produce a 404 with its cause two files
1730 // away. Checked in both directions, since either one alone is
1731 // the same silent miss.
1732 //
1733 // Static routes only: `mount` is a *storage* prefix, and a
1734 // backend route proxies to an origin that owns its own paths.
1735 if !matches!(route.mode, RouteMode::Static { .. }) {
1736 continue;
1737 }
1738 let mount = component.mount.as_deref().map(normalize_mount);
1739 let route_prefix = route_path_prefix(&route.path);
1740 if let Some(mount) = mount {
1741 if mount != route_prefix {
1742 anyhow::bail!(
1743 "domains/{dom_name}.toml: routes[{idx}].path = \
1744 \"{path}\" serves \"{component_ref}\", which \
1745 declares mount = \"/{mount}\" — a mounted \
1746 component publishes under its mount, so the route \
1747 must be \"/{mount}\" or \"/{mount}/*\" (or drop \
1748 the mount to serve from the service root)",
1749 path = route.path,
1750 );
1751 }
1752 } else if !route_prefix.is_empty() {
1753 anyhow::bail!(
1754 "domains/{dom_name}.toml: routes[{idx}].path = \
1755 \"{path}\" serves \"{component_ref}\", which declares \
1756 no `mount` — its bundle publishes at the service root, \
1757 so nothing is stored under \"/{route_prefix}\". Set \
1758 mount = \"/{route_prefix}\" on the component, or route \
1759 it at \"/*\"",
1760 path = route.path,
1761 );
1762 }
1763 }
1764 }
1765 Ok(())
1766 }
1767
1768 /// Look up a domain manifest by name (file stem under `.yah/domains/`).
1769 pub fn domain(&self, name: &str) -> Option<&DomainConfig> {
1770 self.domains.get(name)
1771 }
1772
1773 pub fn machine(&self, name: &str) -> Option<&MachineConfig> {
1774 self.machines.iter().find(|m| m.name == name)
1775 }
1776
1777 /// Look up a provider by id (matches `provider.id`, not the file stem).
1778 pub fn provider(&self, id: &str) -> Option<&ProviderConfig> {
1779 self.providers.iter().find(|p| p.id == id)
1780 }
1781
1782 /// Look up a service by name (matches `service.toml`'s `name` field).
1783 pub fn service(&self, name: &str) -> Option<&ServiceWithMirrors> {
1784 self.services.get(name)
1785 }
1786
1787 /// Look up a legacy mirror by camp name (pre-R215 .yah/cloud/mirrors/).
1788 pub fn legacy_mirror(&self, camp: &str) -> Option<&LegacyMirrorConfig> {
1789 self.legacy_mirrors.iter().find(|m| m.camp == camp)
1790 }
1791
1792 pub fn workload(&self, name: &str) -> Option<&WorkloadConfig> {
1793 self.workloads.iter().find(|w| w.spec.name == name)
1794 }
1795
1796 /// Every machine declaring `sovereign_group == group`, in declaration order.
1797 ///
1798 /// W305/R742-F3. A sovereign group has no file of its own — it exists only
1799 /// as the set of machines that name the same string — so "which boxes are
1800 /// the dev cluster" has to be *derived*, and before this it was not derived
1801 /// anywhere: `yah cloud rollout plan` still takes a hand-listed
1802 /// `--voter us-west-011 --voter us-west-013 …` for a fact the machine TOMLs
1803 /// already state (W314 gap 1).
1804 ///
1805 /// **This is not placement.** Resolving a group to its members is a
1806 /// *lookup*, and it stays outside [`RequiredSpec`] on purpose — see
1807 /// [`MachineConfig::sovereign_group`]. `migrate` calls this to pick the
1808 /// candidate set it then admits a workload against; nothing here filters
1809 /// scheduling, and adding `sovereign_group` to `matches` would still be the
1810 /// category error that doc warns about.
1811 ///
1812 /// An empty result means no machine declares `group`, which is
1813 /// indistinguishable from a typo — callers should say so with
1814 /// [`Self::declared_sovereign_groups`] rather than reporting "no
1815 /// candidates".
1816 pub fn machines_in_group(&self, group: &str) -> Vec<&MachineConfig> {
1817 self.machines
1818 .iter()
1819 .filter(|m| m.sovereign_group.as_deref() == Some(group))
1820 .collect()
1821 }
1822
1823 /// Every distinct `sovereign_group` declared by any machine, sorted.
1824 ///
1825 /// Exists so a bad `--to` names the real vocabulary instead of complaining
1826 /// abstractly — the same fail-loud shape [`taint_effect`]'s legal-key list
1827 /// gives `check_inert_taints`. Standalone machines (`None`) contribute
1828 /// nothing: "in no group" is not a group you can migrate *to*.
1829 pub fn declared_sovereign_groups(&self) -> Vec<&str> {
1830 let mut groups: Vec<&str> = self
1831 .machines
1832 .iter()
1833 .filter_map(|m| m.sovereign_group.as_deref())
1834 .collect();
1835 groups.sort_unstable();
1836 groups.dedup();
1837 groups
1838 }
1839
1840 /// F16 placement v1: the first machine satisfying every hard axis of `req`
1841 /// (region/zone/provider membership + mesh_tags superset). Declaration order
1842 /// in `.yah/infra/machines/` decides ties — deterministic-greedy, no
1843 /// backtracking. A fully-unconstrained `req` matches the first machine.
1844 ///
1845 /// Fails loud with the constraint summary and the candidate machine names
1846 /// when nothing matches, so `yah cloud apply` surfaces *why* placement
1847 /// failed instead of a silent empty set.
1848 pub fn resolve_machine(&self, req: &RequiredSpec) -> Result<&MachineConfig> {
1849 resolve_machine_among(&self.machines, req)
1850 }
1851
1852 /// F16 placement at horizontal scale: the first
1853 /// [`RequiredSpec::replica_count`] machines satisfying every hard axis of
1854 /// `req`, in declaration order (R844-F8).
1855 ///
1856 /// The N-valued form of [`Self::resolve_machine`], which is the N=1 case of
1857 /// this and not a different selector — both land in [`select_matching`].
1858 /// That shared bottom is what makes the deploy resolver
1859 /// (`reconciler::mesofact_bundle::resolve_bundle_machines`, which calls
1860 /// this) and the ingress planner's
1861 /// (`reconciler::ingress::resolve_ingress_placements`, which calls
1862 /// [`resolve_machines_among`] over the same `machines` slice) agree on the
1863 /// same N machines **by construction**. They must agree set-for-set, not
1864 /// merely in count: a front door aimed at nodes the workload was never
1865 /// deployed to renders a *subset* of the backends, which is the failure that
1866 /// looks like it worked.
1867 pub fn resolve_machines(&self, req: &RequiredSpec) -> Result<Vec<&MachineConfig>> {
1868 resolve_machines_among(&self.machines, req)
1869 }
1870
1871 /// F16 placement: first machine whose `mesh_tags` is a superset of
1872 /// `required`. Declaration order in `.yah/infra/machines/` decides ties.
1873 /// Empty `required` matches the first machine; callers should treat
1874 /// empty-required as "no constraint" and skip this lookup.
1875 ///
1876 /// Back-compat thin wrapper over [`CloudConfig::resolve_machine`] for the
1877 /// mesh-tags-only call sites that predate the topology axes.
1878 pub fn resolve_machine_by_mesh_tags(&self, required: &[String]) -> Option<&MachineConfig> {
1879 let req = RequiredSpec {
1880 mesh_tags: required.to_vec(),
1881 ..Default::default()
1882 };
1883 self.resolve_machine(&req).ok()
1884 }
1885
1886 /// Admission: resolve the target machine for a remote [`WorkloadSpec`],
1887 /// honoring the R594 mesh-tag node-selector annotation
1888 /// (`velveteen_exec::remote::NODE_SELECTOR_MESH_TAGS_ANNOTATION` =
1889 /// `yah.node-selector.mesh-tags`, comma-joined).
1890 ///
1891 /// The producer side (`velveteen_exec::remote::build_workload_spec`, R594) writes
1892 /// `TaskLocation::RemoteAny.mesh_tags` — e.g. `[tag:build-worker, arch:x86]`
1893 /// from [`qed::platform::build_worker_mesh_tags`] — into the workload's
1894 /// annotations. This is the consumer: candidates are restricted to machines
1895 /// whose `mesh_tags` are a **superset** of the requested set, so an amd64
1896 /// build lands on the `arch:x86` build-worker (us-west-002) and an arm64
1897 /// build on a `arch:arm` Pi5. Declaration order in `.yah/infra/machines/`
1898 /// breaks ties.
1899 ///
1900 /// An absent or empty annotation means "no mesh-tag constraint" — pre-R594
1901 /// behavior (any node), matching [`RequiredSpec::is_unconstrained`].
1902 ///
1903 /// This is the single admission seam: R572-F5 extends it with the capacity
1904 /// floor (workload request fits node allocatable−committed) and taint
1905 /// repulsion/affinity by enriching [`RequiredSpec::matches`] /
1906 /// [`Self::resolve_machine`]. Do not fork a second selector.
1907 pub fn admit_workload(&self, ws: &WorkloadSpec) -> Result<&MachineConfig> {
1908 self.resolve_machine(&admission_spec(ws, &self.workloads))
1909 }
1910
1911 /// Every machine that admits `ws`, in declaration order — the *pool*
1912 /// [`Self::admit_workload`] returns the head of (R605-T14).
1913 ///
1914 /// # Why a pool and not just the winner
1915 ///
1916 /// `tag:build-worker` is a statement that the tagged boxes are
1917 /// **interchangeable**: a build is booked against the tag, not against
1918 /// `us-west-002`. Returning one machine forced every caller to act as if it
1919 /// were booked against a name, and admission has no liveness input — so a
1920 /// tagged box that is asleep won the file-name tie-break and its builds
1921 /// failed rather than landing on the identical box next to it. That is
1922 /// exactly what happened on 2026-09-03 when `us-west-002` regained the tag.
1923 ///
1924 /// The fix is **not** to teach this function about liveness. It stays a pure
1925 /// function of the declared inventory (see `xtask/tests/fleet_build_placement.rs`
1926 /// on why a placement pin that needs the network is a flake). It hands the
1927 /// dispatcher the whole interchangeable set instead, and the dispatcher —
1928 /// which has the network — probes and fails over within it:
1929 /// `app/yah/cli/src/yubaba_client.rs`'s `MeshYubabaClient::deploy`.
1930 ///
1931 /// Order is the declaration order `admit_workload` already used, and callers
1932 /// should preserve it as their preference order rather than load-balancing
1933 /// across it: a retried build wants the node still holding its warm
1934 /// `target/`, which is the same reason [`first_match`] is deliberately
1935 /// first-fit.
1936 ///
1937 /// `Err` — never `Ok(vec![])` — when nothing admits `ws`, carrying the same
1938 /// message [`Self::admit_workload`] would have produced. "No node admits
1939 /// this" and "the pool is empty" are the same failure and must read the same.
1940 pub fn admit_workload_candidates(&self, ws: &WorkloadSpec) -> Result<Vec<&MachineConfig>> {
1941 let req = admission_spec(ws, &self.workloads);
1942 let all: Vec<&MachineConfig> = self.machines.iter().collect();
1943 let matched = matching(&all, &req);
1944 if matched.is_empty() {
1945 // Delegate the wording so the two paths cannot drift apart.
1946 return Err(first_match(&all, &req, DECLARED_POOL, EMPTY_DECLARED_POOL)
1947 .expect_err("matching() found nothing, so first_match cannot succeed"));
1948 }
1949 Ok(matched)
1950 }
1951
1952 /// [`Self::admit_workload`] restricted to the machines of one sovereign
1953 /// group (W305/R742-F3, `yah cloud migrate --to <group>`).
1954 ///
1955 /// Same [`RequiredSpec`], same [`RequiredSpec::matches`], same
1956 /// declaration-order tie-break — only the candidate *set* differs. That is
1957 /// the whole reason this is a narrowing of the admission seam rather than a
1958 /// second selector: a workload that cannot be scheduled onto a group's
1959 /// boxes must fail here for exactly the reason it would fail anywhere else,
1960 /// and `no-appliance` on the dev Pis (W305 finding 2) is precisely the case
1961 /// that must not be silently routed around by a migration verb.
1962 ///
1963 /// `Err` when the group has no members *or* when no member admits `ws`; the
1964 /// two are different mistakes, so callers wanting to tell them apart should
1965 /// check [`Self::machines_in_group`] first.
1966 pub fn admit_workload_in_group(
1967 &self,
1968 ws: &WorkloadSpec,
1969 group: &str,
1970 ) -> Result<&MachineConfig> {
1971 let members = self.machines_in_group(group);
1972 let empty_pool = format!(
1973 "(no machine declares sovereign_group = \"{group}\" — declared groups: {})",
1974 match self.declared_sovereign_groups().as_slice() {
1975 [] => "(none)".to_string(),
1976 gs => gs.join(", "),
1977 }
1978 );
1979 first_match(
1980 &members,
1981 &admission_spec(ws, &self.workloads),
1982 &format!("machines in sovereign group '{group}'"),
1983 &empty_pool,
1984 )
1985 }
1986}
1987
1988/// **The** placement selector: the first candidate satisfying every axis of
1989/// `req`, declaration order breaking ties, deterministic-greedy with no
1990/// backtracking.
1991///
1992/// Every path that picks a machine goes through here, and the only thing any
1993/// of them varies is *which machines are candidates* — never the predicate.
1994/// [`CloudConfig::resolve_machine`] passes the whole fleet;
1995/// [`CloudConfig::admit_workload_in_group`] passes one sovereign group's
1996/// members. That split is the point: a candidate-set narrowing composes with
1997/// the [`RequiredSpec`] axes for free, whereas expressing the same narrowing
1998/// *as* an axis would put facts like blast radius into a filter they must
1999/// never be in (see [`MachineConfig::sovereign_group`]).
2000///
2001/// So a new placement scope is a new candidate set plus a `pool` label, and a
2002/// new placement *constraint* is a field on [`RequiredSpec`] — those are the
2003/// two extension points, and neither is a second selector. `pool` and
2004/// `empty_pool` exist only so the failure names the set it actually searched;
2005/// a refusal that says "no candidates" without saying *among what* is one the
2006/// operator has to reconstruct by hand.
2007/// F16 placement v1 resolution over an explicit machine list — the
2008/// `.machines`-only half of [`CloudConfig::resolve_machine`], for callers that
2009/// have loaded just the machines tree rather than the whole cross-ref-validated
2010/// config.
2011///
2012/// R772: `resolve_ingress_placements` (`reconciler::ingress`) is the reason
2013/// this is `pub(crate)` rather than staying folded into
2014/// `CloudConfig::resolve_machine` — ingress collation walks every mirror in
2015/// the workspace and has no business hard-failing over an unrelated mirror's
2016/// `providers.X.use = "<id>"` typo, which is what going through
2017/// `CloudConfig::load`'s cross-ref validation would do. "Do not fork a second
2018/// selector" (see the module doc above) still holds: this is the *same*
2019/// [`first_match`], just handed a narrower candidate set than `self.machines`.
2020pub(crate) fn resolve_machine_among<'a>(
2021 machines: &'a [MachineConfig],
2022 req: &RequiredSpec,
2023) -> Result<&'a MachineConfig> {
2024 let all: Vec<&MachineConfig> = machines.iter().collect();
2025 first_match(&all, req, DECLARED_POOL, EMPTY_DECLARED_POOL)
2026}
2027
2028/// R844-F8: [`resolve_machine_among`] widened to the constraint's own replica
2029/// count — the first [`RequiredSpec::replica_count`] matching machines, in the
2030/// same declaration order, from the same candidate slice.
2031///
2032/// **The one entry point both resolvers share.**
2033/// `reconciler::ingress::resolve_ingress_placements` calls this directly and
2034/// `reconciler::mesofact_bundle::resolve_bundle_machines` reaches it through
2035/// [`CloudConfig::resolve_machines`], both over `cfg.machines` — so the ingress
2036/// planner and the deployer cannot pick different subsets. That is a structural
2037/// guarantee, not a tested coincidence, and it has to be: discovery aimed at a
2038/// node the bundle was never placed on publishes a hostname with a dead
2039/// backend behind it, and at scale > 1 the front door still answers from the
2040/// nodes that *did* get it.
2041///
2042/// Determinism is therefore part of correctness here. `machines` arrives in
2043/// file-name order (`load_dir`, pinned by
2044/// `machines_load_in_file_name_order_not_read_dir_order`), and selection is a
2045/// stable prefix of that order — so "the first two matching" is the same two
2046/// on both sides of the same tree.
2047pub(crate) fn resolve_machines_among<'a>(
2048 machines: &'a [MachineConfig],
2049 req: &RequiredSpec,
2050) -> Result<Vec<&'a MachineConfig>> {
2051 let all: Vec<&MachineConfig> = machines.iter().collect();
2052 select_matching(
2053 &all,
2054 req,
2055 req.replica_count(),
2056 DECLARED_POOL,
2057 EMPTY_DECLARED_POOL,
2058 )
2059}
2060
2061const DECLARED_POOL: &str = "declared machines";
2062const EMPTY_DECLARED_POOL: &str = "(no machines declared under .yah/infra/machines/)";
2063
2064fn first_match<'a>(
2065 candidates: &[&'a MachineConfig],
2066 req: &RequiredSpec,
2067 pool: &str,
2068 empty_pool: &str,
2069) -> Result<&'a MachineConfig> {
2070 Ok(select_matching(candidates, req, 1, pool, empty_pool)?
2071 .into_iter()
2072 .next()
2073 .expect("select_matching errors rather than returning short"))
2074}
2075
2076/// The N-selecting core of the placement selector: the first `want` candidates
2077/// satisfying `req`, in candidate order (R844-F8).
2078///
2079/// [`first_match`] is this with `want = 1`, which is why widening a caller to a
2080/// replica count cannot introduce a second selector — the predicate, the
2081/// ordering and the failure vocabulary are all one implementation.
2082///
2083/// **A shortfall is an error.** Matching one machine when two were asked for
2084/// returns `Err` naming both numbers and the pool searched, never a one-element
2085/// vec: a half-placed workload that reports success is worse than a failed
2086/// apply, because the front door then publishes a hostname whose backend set is
2087/// quietly smaller than declared. `want = 0` is the same mistake spelled
2088/// differently and is refused for the same reason.
2089fn select_matching<'a>(
2090 candidates: &[&'a MachineConfig],
2091 req: &RequiredSpec,
2092 want: usize,
2093 pool: &str,
2094 empty_pool: &str,
2095) -> Result<Vec<&'a MachineConfig>> {
2096 let names = || {
2097 if candidates.is_empty() {
2098 empty_pool.to_string()
2099 } else {
2100 candidates
2101 .iter()
2102 .map(|m| m.name.as_str())
2103 .collect::<Vec<_>>()
2104 .join(", ")
2105 }
2106 };
2107
2108 if want == 0 {
2109 anyhow::bail!(
2110 "replicas = 0 places {} on nothing — a placement that deploys to no machine is \
2111 a typo, not a scale-down; remove the slot instead",
2112 req.describe()
2113 );
2114 }
2115
2116 let mut matched = matching(candidates, req);
2117 if matched.len() >= want {
2118 matched.truncate(want);
2119 return Ok(matched);
2120 }
2121
2122 if want == 1 {
2123 anyhow::bail!(
2124 "no candidates matching {} — {pool}: {}",
2125 req.describe(),
2126 names()
2127 );
2128 }
2129 anyhow::bail!(
2130 "only {} of {want} machines match {} — placing fewer than the declared \
2131 `replicas = {want}` would publish a smaller backend set than the mirror asks for; \
2132 {pool}: {}",
2133 matched.len(),
2134 req.describe(),
2135 names()
2136 )
2137}
2138
2139/// The predicate itself, applied to every candidate in order — the one place
2140/// `req.matches` is called on a set.
2141///
2142/// [`select_matching`] takes a prefix of this; [`CloudConfig::admit_workload_candidates`]
2143/// takes all of it. Keeping both on this function is what makes "the pool the
2144/// dispatcher failed over within" and "the machine admission picked" the same
2145/// answer by construction rather than by two filters that happen to agree.
2146fn matching<'a>(candidates: &[&'a MachineConfig], req: &RequiredSpec) -> Vec<&'a MachineConfig> {
2147 candidates
2148 .iter()
2149 .copied()
2150 .filter(|m| req.matches(m))
2151 .collect()
2152}
2153
2154/// The [`RequiredSpec`] a workload is admitted against — the single place the
2155/// axes are derived from a [`WorkloadSpec`].
2156///
2157/// Extracted from [`CloudConfig::admit_workload`] so that
2158/// [`CloudConfig::admit_workload_in_group`] narrows the candidate set without
2159/// restating the axes. Forking that derivation is how the two paths would
2160/// silently disagree about whether a workload fits a node.
2161///
2162/// # It admits a group, not a workload (R860-T4 / W338)
2163///
2164/// The axes come from [`placement_group`] — `ws` plus the transitive closure of
2165/// its `local` requirement edges — because those members are placed together or
2166/// not at all. Capacity is their **sum**, archetype repulsion their **union**,
2167/// and mesh tags their union too. `prefer-local` and `anywhere` edges bind
2168/// nothing: a spec with neither `requires` nor `depends_on` local edges has a
2169/// group of exactly itself and resolves byte-identically to the pre-R860 axes.
2170///
2171/// This is the **only** gate. Node election is CLI-side
2172/// (`MeshYubabaClient::elect_node`, which picks a live member of the pool this
2173/// produces); the yubaba node process accepts whatever it is handed and never
2174/// re-checks placement, so a wrong group here is not caught downstream.
2175///
2176/// @yah:ticket(R860-T4, "Admission: place the transitive closure of `local` edges as one group, not one workload")
2177/// @yah:status(review)
2178/// @yah:phase(P1)
2179/// @yah:at(2026-09-05T18:29:13Z)
2180/// @yah:assignee(agent:bundle-anthropic-ashguard)
2181/// @yah:parent(R860)
2182/// @yah:next("W338 §Placement consequences 1 and 2. `admission_spec()` (config.rs:1974-1998) derives its axes from ONE spec; it must derive them from the group — the transitive closure of `local` requirement edges over `effective_requirements()`. `prefer-local` and `anywhere` edges do NOT bind the group. Three consequences: memory/cpu floor becomes the SUM of the group's requests, not the requirer's alone; `repel_archetype` becomes the union over members (so a group containing an Appliance is repelled by `no-appliance` even if the requirer is a Server); and the group is non-drainable if ANY member is an Appliance, which today is a per-workload check at yubaba/src/lib.rs:3117-3128 and now has to be computed over a set.")
2183/// @yah:verify("cargo test -p cloud --lib config")
2184/// @yah:gotcha("Node election is CLI-side, not cluster-side: `MeshYubabaClient::elect_node` (app/yah/cli/src/yubaba_client.rs:235-268) calls `admit_workload_candidates` (config.rs:1753), picks one node, and POSTs the deploy there. The yubaba node process never decides placement — it accepts whatever it is handed. So group admission has to be right in `config.rs` because there is no second gate downstream to catch it.")
2185/// @arch:see(.yah/docs/working/W338-workload-dependencies-and-appliance-composition.md)
2186/// @yah:depends_on(R860-T1)
2187/// @yah:handoff("ADMISSION NOW PLACES A GROUP, NOT A WORKLOAD. `admission_spec` (oss/yubaba/crates/cloud/src/config.rs:2009) takes `(ws, declared: &[WorkloadConfig])` and derives every axis from `placement_group(ws, declared)` (:2108) — the transitive closure of `local` requirement edges over `effective_requirements()`, traversing `Requirement::provides` where present and resolving by ident against `cfg.workloads` (.yah/infra/workloads/) otherwise, mesh-identity first and workload name second. Capacity is the SUM of the members' `memory_request_mb()` / `resources.cpu_millis` (saturating). Only `local` binds: `prefer-local` and `anywhere` (which every legacy `depends_on` folds into) are skipped, so a spec without local edges has a group of exactly itself and its axes are bit-identical to the pre-R860 derivation.")
2188/// @yah:handoff("REPEL BECAME A SET. `RequiredSpec::repel_archetype: Option<LifecycleArchetype>` is now `repel_archetypes: Vec<LifecycleArchetype>` (config.rs:3878), the union over group members; `matches` (:3971) rejects a node carrying `no-<taint_key()>` for ANY of them, `describe` emits one `not-tainted(...)` part per archetype, `is_unconstrained` tests `is_empty()`. The field is `#[serde(skip)]`, so no wire or schema drift, and grep over app/ crates/ oss/ xtask/ finds no other referent of the old name and no `RequiredSpec { .. }` literal outside config.rs — the rename is contained. `admit_workload` / `admit_workload_candidates` / `admit_workload_in_group` signatures are unchanged; all three now pass `&self.workloads`.")
2189/// @yah:handoff("CYCLE GUARD, AND THE BUG IT TOOK TO GET RIGHT. The walker keeps TWO visited lists: `in_group` (member mesh identities) and `expanded` (requirement idents already resolved). The first version used one list and was silently wrong in the common case — a requirement's ident IS its provider's mesh identity, so marking the ident before resolving made every provider look already-present and `placement_group` returned a group of one. Five of the new tests caught it. If you refactor this, keep the two questions separate.")
2190/// @yah:handoff("ELECT_NODE NEEDS NO CHANGE FOR THIS TICKET — read it (app/yah/cli/src/yubaba_client.rs:235-268). It calls `admit_workload_candidates`, so it now receives a pool already filtered to nodes that can host the WHOLE group, then probes for liveness within it. That is correct for T4 because only the requirer is deployed today. It becomes load-bearing at R860-T6: `supply = \"self\"` provisioning MUST reuse the node URL `elect_node` returned for the requirer and must not re-elect per member — the probe is liveness-sensitive, so a second election can legally return a different member of the same pool and split the group across two nodes.")
2191/// @yah:handoff("DECISIONS THE BRIEF DID NOT COVER, all recorded in doc comments at the site. (1) `mesh_tags` are UNIONED over the group — the axis is already a superset/AND check, so a node that cannot host one member cannot host the group; zero regression risk since nothing in the tree declares `requires` yet. (2) `nodes` (the R833-F8 operator pin) stays REQUIRER-ONLY: it is a membership list, so intersecting two members' pins can yield an empty vec, which the axis reads as no-constraint — the exact inverse of the conflict. (3) `requires_taint` is a single Option: the requirer's wins, else the first member declaring one. Two members demanding DIFFERENT taints is not representable and would be an unplaceable group; widening that axis to a set is a follow-up if a real case appears. (4) An unresolvable `local` ident is SKIPPED, not an error — admission is a pure function of the declared inventory and must not start refusing deploys over a provider a later ticket declares; the cost is that its request does not count toward the floor, which is the exposure `depends_on` has always had.")
2192/// @yah:handoff("DRAINABILITY: placement half landed, node half deliberately NOT touched. `group_is_drainable(members)` (config.rs:2160) is the set-valued predicate W338 §Placement consequences 2 asks for — false as soon as any member is an Appliance — and the `no-appliance` repulsion that follows from it is enforced through `repel_archetypes`. The node-side loop `drain_workloads` (oss/yubaba/crates/yubaba/src/lib.rs, the R572-F4 archetype_registry skip) still decides per workload and knows nothing about requirement edges, so a Server bound to an Appliance by a `local` edge would still be drained alone. Not fixed here for two reasons: that file has three sessions live in it (the brief named them), and the fix needs group edges plumbed to the node process, which is R860-T6's rail rather than a local edit. `yubaba` already depends on `cloud`, so the predicate is directly callable from there when that plumbing exists.")
2193/// @yah:verify("BASELINE recorded before editing, tree anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2: `cargo test -p yah-cloud --lib` (from oss/yubaba) = 1081 passed, 0 failed, 4 ignored, exit 0. AFTER: 1090 passed, 0 failed, 4 ignored, exit 0 — +9, exactly the nine tests added. `cargo check -p yah-cloud --all-targets` exit 0, and `cargo check -p yubaba --all-targets` exit 0 as well (yubaba consumes `cloud`, so it is where the `repel_archetypes` rename would have surfaced). Every exit code echoed explicitly, never inferred from an empty grep.")
2194/// @yah:verify("NEW TESTS (config.rs `mod tests`, R860-T4 section at the end): a_local_edge_binds_the_provider_into_the_placement_group; prefer_local_and_anywhere_edges_do_not_bind_the_group (covers a legacy `depends_on` too); the_group_is_the_transitive_closure_and_traverses_inline_provides; an_ident_cycle_closes_the_group_instead_of_looping_forever; an_unresolvable_local_ident_is_skipped_rather_than_refused; the_capacity_floor_is_the_sum_of_the_group_not_the_requirer_alone (a 300 MiB node refuses two 256 MiB members and the error names memory_mb>=512; a 512 MiB node admits); a_server_requiring_an_appliance_locally_is_repelled_by_no_appliance (same requirer alone still lands on the tainted Pi, so the repulsion provably comes from the edge); a_group_containing_an_appliance_is_not_drainable; a_spec_with_no_local_edges_admits_exactly_as_it_did_before.")
2195/// @yah:gotcha("The camp's `yah build run` rail killed three consecutive verification runs against the shared oss/yubaba/target dir: each ended with only `Blocking waiting for file lock on build directory` in the log and no exit code, after 121s / 720s. The green result above was obtained with `CARGO_TARGET_DIR=/tmp/r860t4-target`, which sidesteps the contended lock at the cost of one cold dep build. Worth reaching for directly when the yubaba target dir is busy rather than burning three cycles discovering it.")
2196/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
2197/// @yah:next("R860-T6 (supply = \"self\"): deploy the group's non-requirer members onto the node `elect_node` already returned for the requirer — do NOT re-elect per member, or a liveness probe can split the group across two nodes. `placement_group` (config.rs:2108) hands you the member specs in traversal order, requirer first.")
2198/// @yah:next("Node-side drain is still per-workload: teach `drain_workloads` (oss/yubaba/crates/yubaba/src/lib.rs) to consult `cloud::config::group_is_drainable` over the requesting workload's placement group once R860-T6 plumbs group membership to the node. Left untouched here on purpose — three sessions were live in that file.")
2199/// @yah:handoff("LEADER RE-VERIFIED (session:69b18855, independent of the courier's self-report). `cargo test -p yah-cloud --lib` from oss/yubaba: 1090 passed / 0 failed / 4 ignored, exit 0, against the courier's recorded 1081/0/4 baseline — +9 = exactly its new tests. Confirmed by content in config.rs: `placement_group` :2120 with the `req.locality != Locality::Local` guard at :2136 (so `prefer-local` and `anywhere` correctly do NOT bind), `group_is_drainable` :2172, and `RequiredSpec::repel_archetype: Option<_>` widened to `repel_archetypes: Vec<_>` at :3890 with the union built at :2039-2050 and enforced at :4004/:4044. The repel rename is `#[serde(skip)]`, so no wire or schema drift.")
2200/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
2201/// @yah:verify("cargo test -p yah-cloud --lib (from oss/yubaba): 1090 passed / 0 failed / 4 ignored, exit 0, vs a 1081/0/4 baseline. Exit codes echoed explicitly throughout rather than inferred from an empty grep — the trap that cost R860-T1 three misses.")
2202/// @yah:gotcha("CORRECTION FROM R860-T6, and the leader propagated the error so it is worth naming: this ticket's handoff asserted \\\"`yubaba` already depends on `cloud`, so the predicate is directly callable from there\\\". THAT IS WRONG. `cloud` is a DEV-dependency of yubaba only — oss/yubaba/crates/yubaba/Cargo.toml:150-152, under the comment \\\"Integration test harness\\\" — and cloud's own Cargo.toml records that the runtime yubaba→cloud edge was DELIBERATELY avoided from R374-F3 onward. The leader repeated the claim verbatim in R860-T6's dispatch brief; T6's courier checked it against the manifest instead of trusting it, which is the only reason it did not become a runtime dependency inversion. Resolution: `group_is_drainable`'s body moved down to `workload_spec::group_is_drainable` (workload-spec/src/lib.rs:2365), the shared home both crates already depend on, and `cloud::config::group_is_drainable` (config.rs:2240) now delegates to it keeping its signature. Verified after the move: yah-cloud still 1093/0/4, yah-workload-spec 171+98/0.")
2203/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
2204/// @yah:verify("RE-VERIFIED AT HEAD 00ee20d1 (session:aa5e882d, 2026-09-05). `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` = 1141 passed / 0 failed / 4 ignored, exit 0 (was 1093 at the first leader's check, 1110 at the second; the deltas are peers' tests). Group placement confirmed by content in oss/yubaba/crates/cloud/src/config.rs: `placement_group` derivation at :2073/:2108, `repel_archetypes: Vec<LifecycleArchetype>` at :3878. NOTE FOR ANYONE RE-RUNNING THIS: `cargo test -p yah-cloud --lib` from the repo root FAILS with \"package `yah-cloud` cannot be tested because it requires dev-dependencies and is not a member of the workspace\" — yah-cloud lives in the oss/yubaba workspace, so the invocation needs `--manifest-path oss/yubaba/Cargo.toml`.")
2205fn admission_spec(ws: &WorkloadSpec, declared: &[WorkloadConfig]) -> RequiredSpec {
2206 let group = placement_group(ws, declared);
2207
2208 // Capacity is the group's demand, not the requirer's (W338 §Placement
2209 // consequences 1). Saturating rather than wrapping: an absurd declared
2210 // request must read as "nothing is big enough", never as a small number.
2211 //
2212 // `memory_request_mb()` and NOT `resources.memory_mb`: the latter is a
2213 // cgroup ceiling, and reading a ceiling as a floor made `for_forge`'s
2214 // deliberately-roomy 32 GiB limit mean "only place me on a 32 GiB node".
2215 // That excluded every build-worker in the fleet but one. The accessor falls
2216 // back to `resources.memory_mb` when no request is declared, so specs that
2217 // never set one are admitted exactly as before.
2218 let mut memory_mb: u32 = 0;
2219 let mut cpu_millis: u32 = 0;
2220 // R572-F5 taint repulsion, unioned over the group (W338 §Placement
2221 // consequences 2): a group is non-drainable — and `no-appliance`-repelled —
2222 // if *any* member is an Appliance, even when the requirer is a Server.
2223 //
2224 // R876-B7 inverted the sense. Repulsion is now unconditional in `matches`,
2225 // so what this loop collects is still the group's archetype union, but it is
2226 // converted below into the complementary TOLERATION set. Same predicate,
2227 // stated from the other side.
2228 let mut group_archetypes: Vec<LifecycleArchetype> = Vec::new();
2229 // Mesh tags are already AND-ed (a machine must be a superset), so unioning
2230 // them over the group is the same predicate applied to every member: a node
2231 // that cannot host one member cannot host the group.
2232 let mut mesh_tags = node_selector_mesh_tags(ws);
2233
2234 for member in &group {
2235 memory_mb = memory_mb.saturating_add(member.memory_request_mb());
2236 cpu_millis = cpu_millis.saturating_add(member.resources.cpu_millis);
2237 let arch = member.effective_archetype();
2238 if !group_archetypes.contains(&arch) {
2239 group_archetypes.push(arch);
2240 }
2241 for tag in node_selector_mesh_tags(member) {
2242 if !mesh_tags.contains(&tag) {
2243 mesh_tags.push(tag);
2244 }
2245 }
2246 }
2247
2248 // R860-T5 / W338 §Placement consequences 3: per-node native-exec
2249 // capability. Computed over the group for the same reason every other axis
2250 // is — a `local` edge to a native provider makes the *requirer* unplaceable
2251 // on a node without the backend, even when the requirer is an ordinary
2252 // container workload. This is the `supply = "self"` precondition W338 names:
2253 // a self-supplied native provider has to be placeable where its requirer
2254 // lands, and until now nothing upstream could see whether it was.
2255 //
2256 // Appended to `mesh_tags` rather than given its own field: the axis is
2257 // already an AND-ed superset check against `machine.mesh_tags`, `describe`
2258 // already renders it, and `RequiredSpec` needs no new shape. See
2259 // [`NATIVE_EXEC_MESH_TAG`] for why a tag and not a taint.
2260 if group.iter().any(WorkloadSpec::wants_native_exec)
2261 && !mesh_tags.iter().any(|t| t == NATIVE_EXEC_MESH_TAG)
2262 {
2263 mesh_tags.push(NATIVE_EXEC_MESH_TAG.to_string());
2264 }
2265
2266 RequiredSpec {
2267 mesh_tags,
2268 // R833-F8: imperative node pin. Derived here alongside the inferred
2269 // mesh tags rather than short-circuiting the resolver, so a pinned
2270 // workload is still checked against capacity and taints.
2271 //
2272 // Requirer-only on purpose: the pin is what the operator typed on
2273 // *this* deploy, and `nodes` is a membership list, so intersecting two
2274 // members' pins could yield an empty vec — which this axis reads as "no
2275 // constraint", i.e. the exact opposite of the conflict it represents.
2276 nodes: node_selector_node(ws).into_iter().collect(),
2277 memory_mb,
2278 cpu_millis,
2279 // R876-B7: the archetype union, restated as tolerations — every
2280 // repelling key that is NOT this group's own class. A Server group
2281 // tolerates `no-appliance` and `no-job` and is still blocked by
2282 // `no-server`, which is precisely what the pre-B7 `repel_archetypes`
2283 // axis computed. That equivalence is the migration: the `admit_workload`
2284 // path's placement answers are unchanged for every fleet machine, while
2285 // the mirror-declared path — which could never populate an archetype set
2286 // and so read no taints at all — becomes repel-by-default.
2287 //
2288 // Derived from `LifecycleArchetype::ALL` rather than a literal list, so
2289 // a fourth archetype is tolerated by unrelated groups automatically,
2290 // exactly as `taint_effect` already derives the repulsion half.
2291 tolerates: LifecycleArchetype::ALL
2292 .into_iter()
2293 .filter(|a| !group_archetypes.contains(a))
2294 .map(|a| format!("no-{}", a.taint_key()))
2295 .collect(),
2296 // R572-F5: taint affinity from the requires-taint annotation. The
2297 // requirer's wins; otherwise the first member that declares one, since
2298 // the group shares a node and this axis holds a single key. Two members
2299 // demanding *different* taints is not representable here and would be
2300 // an unplaceable group anyway — see the R860-T4 handoff.
2301 requires_taint: group
2302 .iter()
2303 .find_map(|m| m.requires_taint().map(str::to_owned)),
2304 ..Default::default()
2305 }
2306}
2307
2308/// The workloads that must be placed together with `ws`: the transitive closure
2309/// of `local` requirement edges over [`WorkloadSpec::effective_requirements`],
2310/// starting at the requirer (R860-T4 / W338 §"Each member keeps its own mesh
2311/// identity").
2312///
2313/// **Only `local` binds.** `prefer-local` explicitly "never blocks placement"
2314/// (W338's locality table) and `anywhere` is an ordinary service dependency —
2315/// treating either as a co-scheduling constraint would turn every `depends_on`
2316/// in the tree into one, since the legacy field folds in as `anywhere` + `wait`.
2317///
2318/// A group is **not** a new addressable object: every member keeps its own mesh
2319/// identity, spec and healthcheck (W338). This function returns the members'
2320/// specs so admission can take the sum / union over them, and nothing here
2321/// deploys, provisions or tears anything down — `supply = "self"` provisioning
2322/// is R860-T6 and per-node native-exec capability is R860-T5.
2323///
2324/// Two ways a member is reached, in this order:
2325/// - [`Requirement::provides`], the inline spec a `supply = "self"` requirement
2326/// carries;
2327/// - otherwise an ident lookup against `declared` (`.yah/infra/workloads/`),
2328/// matched on mesh identity first and on workload name second, because those
2329/// coincide for every spec in the tree today but the requirement is written in
2330/// the mesh-identity currency.
2331///
2332/// An ident that resolves to neither is **skipped**, not an error: admission is
2333/// a pure function of the declared inventory and must not start failing deploys
2334/// over a provider that a not-yet-written ticket will declare. The cost is that
2335/// its request does not count toward the floor, which is the same exposure
2336/// `depends_on` has always had.
2337///
2338/// **Cycle-guarded.** `validate::check_requires` bounds `provides` *nesting* to
2339/// depth 1 but nothing stops two separately-declared specs from requiring each
2340/// other, and this closure would otherwise not terminate. Each requirement ident
2341/// is resolved at most once and each member joins the group at most once, so a
2342/// cycle simply closes the group.
2343pub fn placement_group(ws: &WorkloadSpec, declared: &[WorkloadConfig]) -> Vec<WorkloadSpec> {
2344 let mut members = vec![ws.clone()];
2345 // Two separate visited sets, because the two questions differ: `in_group`
2346 // stops a workload being added twice, `expanded` stops an ident being
2347 // resolved twice. Folding them into one list makes the ident of a member
2348 // already in the group indistinguishable from the member itself — and since
2349 // a requirement's ident *is* its provider's mesh identity, that reads every
2350 // provider as already-present and silently returns a group of one.
2351 let mut in_group: Vec<String> = vec![group_key(ws)];
2352 let mut expanded: Vec<String> = Vec::new();
2353 let mut next = 0;
2354
2355 while next < members.len() {
2356 let requirements = members[next].effective_requirements();
2357 next += 1;
2358 for req in requirements {
2359 if req.locality != Locality::Local {
2360 continue;
2361 }
2362 if expanded.contains(&req.ident.0) {
2363 continue;
2364 }
2365 expanded.push(req.ident.0.clone());
2366
2367 let provider = match req.provides.as_deref() {
2368 Some(spec) => spec.clone(),
2369 None => match resolve_requirement_ident(&req.ident, declared) {
2370 Some(spec) => spec,
2371 None => continue,
2372 },
2373 };
2374 let key = group_key(&provider);
2375 if in_group.contains(&key) {
2376 continue;
2377 }
2378 in_group.push(key);
2379 members.push(provider);
2380 }
2381 }
2382
2383 members
2384}
2385
2386/// Whether a placement group may be drained off its node (W338 §Placement
2387/// consequences 2): false as soon as **any** member is an Appliance.
2388///
2389/// The set-valued form of the per-workload check the node itself makes in
2390/// `drain_workloads` (`oss/yubaba/crates/yubaba/src/lib.rs`), which skips an
2391/// Appliance by its own archetype and knows nothing about requirement edges. A
2392/// `Server` bound to an Appliance by a `local` edge has to move with it or not
2393/// at all, so draining it alone breaks the group the same way placing it alone
2394/// would.
2395///
2396/// R860-T6 moved the body to [`workload_spec::group_is_drainable`] and left this
2397/// signature untouched. The node's `drain_workloads` needs the identical
2398/// predicate, and yubaba has no runtime dependency on this crate by design
2399/// (R374-F3) — so the one implementation now lives in the crate both sides
2400/// already depend on, rather than being copied into the second caller.
2401pub fn group_is_drainable(members: &[WorkloadSpec]) -> bool {
2402 workload_spec::group_is_drainable(members)
2403}
2404
2405/// Identity a placement-group member is deduplicated by — its mesh identity,
2406/// which is the currency [`Requirement::ident`] is written in.
2407fn group_key(ws: &WorkloadSpec) -> String {
2408 ws.expose.mesh.identity.0.clone()
2409}
2410
2411/// Resolve a requirement's ident to a separately-declared spec: mesh identity
2412/// first, workload file name second.
2413fn resolve_requirement_ident(
2414 ident: &workload_spec::MeshIdent,
2415 declared: &[WorkloadConfig],
2416) -> Option<WorkloadSpec> {
2417 declared
2418 .iter()
2419 .find(|w| w.spec.expose.mesh.identity == *ident)
2420 .or_else(|| declared.iter().find(|w| w.spec.name == ident.0))
2421 .map(|w| w.spec.clone())
2422}
2423
2424/// The mesh tag a node declares to advertise that its kamaji can run **native**
2425/// (fork+exec) workloads — R860-T5 / W338 §"Placement consequences" 3.
2426///
2427/// A workload marked `yah.exec = native` ([`WorkloadSpec::wants_native_exec`])
2428/// is not containerized: kamaji fork+execs it on the node's own userland. That
2429/// backend only exists when the node's kamaji was **built** with the
2430/// `native-exec` cargo feature and **started** with `--native-exec-dir`
2431/// (`oss/kamaji/crates/kamaji-bin/src/main.rs`). Both are node-local startup
2432/// decisions, invisible to everything upstream — so before this tag, placement
2433/// happily elected a node whose kamaji then refused the deploy with
2434/// `BackendRefused: ... no native backend is available (native backend not
2435/// configured — start kamaji with --native-exec-dir)`. That is exactly how the
2436/// mesh lost its coordination server for 25 hours on 2026-09-03 (R858: raft
2437/// leadership moved headscale, a native workload, to `us-south-001`, which has
2438/// no such kamaji). [`admission_spec`] now requires this tag whenever any
2439/// placement-group member is native, which turns that dispatch-time surprise
2440/// into a placement precondition.
2441///
2442/// # Why a mesh tag and not a taint
2443///
2444/// The two vocabularies on [`MachineConfig`] mean opposite things. `mesh_tags`
2445/// are **positive capability** matched as a superset — "this node CAN" — which
2446/// is precisely the claim being made, and an extra tag on a machine can only
2447/// ever make it match *more* requirement sets, so declaring it is regression-
2448/// free. `taints` are **repulsion** — "keep this class off" — and would have to
2449/// be inverted (`no-native-exec` on every node lacking the backend, i.e. the
2450/// declaration burden falls on the majority) *and* taught to
2451/// [`taint_effect`], or [`crate::validate::check_inert_taints`] would correctly
2452/// lint the key dead.
2453///
2454/// # The `cap:` namespace
2455///
2456/// New here. The live prefixes are `tag:` (role — `tag:build-worker`,
2457/// `tag:qed`, `tag:cloud-runner`), `arch:` and `os:` (facts about the silicon
2458/// and userland), and `tier:` is reserved for the environment axis (R763, see
2459/// [`crate::validate::check_retired_arch_tags`]). A *capability the daemon was
2460/// configured with* is none of those: it is not a role an operator assigns and
2461/// not a property of the hardware, it is a fact about how kamaji was started,
2462/// and it changes when the node is rolled. Nothing validates tag prefixes, so
2463/// this costs no wiring.
2464///
2465/// # Fails closed
2466///
2467/// A node that does not declare it is not a candidate. An undeclared fleet
2468/// therefore reports "no node admits" at election time rather than dispatching
2469/// to a node that will refuse — the refusal moves earlier and names the
2470/// constraint, which is the whole point. Declared today (from readings recorded
2471/// in-repo, not inferred) on `us-west-001` and `us-west-003`; see those
2472/// machines' TOMLs for the evidence and the date.
2473pub const NATIVE_EXEC_MESH_TAG: &str = "cap:native-exec";
2474
2475/// Parse the R594 mesh-tag node-selector off a workload's annotations into the
2476/// requested tag set. Absent annotation or empty value ⇒ empty vec ("no
2477/// constraint"). Whitespace around each comma-separated tag is trimmed and
2478/// empty segments are dropped, so `"tag:build-worker, arch:x86"` and
2479/// `"tag:build-worker,arch:x86"` parse identically.
2480pub fn node_selector_mesh_tags(ws: &WorkloadSpec) -> Vec<String> {
2481 ws.annotations
2482 .get(velveteen_exec::remote::NODE_SELECTOR_MESH_TAGS_ANNOTATION)
2483 .map(|v| {
2484 v.split(',')
2485 .map(str::trim)
2486 .filter(|s| !s.is_empty())
2487 .map(String::from)
2488 .collect()
2489 })
2490 .unwrap_or_default()
2491}
2492
2493/// Parse the R833-F8 imperative node-selector off a workload's annotations —
2494/// the single machine `name` the operator pinned the run to
2495/// (`--where=node:us-west-003`). Absent or blank ⇒ `None` ("no constraint"),
2496/// which is every workload built before this axis existed.
2497///
2498/// One node, not a list: the annotation exists to express "run it *there*", and
2499/// a comma-joined set would be a worse spelling of the mesh-tag selector that
2500/// already handles "any of these".
2501pub fn node_selector_node(ws: &WorkloadSpec) -> Option<String> {
2502 ws.annotations
2503 .get(velveteen_exec::remote::NODE_SELECTOR_NODE_ANNOTATION)
2504 .map(|v| v.trim())
2505 .filter(|v| !v.is_empty())
2506 .map(String::from)
2507}
2508
2509/// Load every `.yah/infra/providers/*.toml` into a [`ProviderConfig`] list.
2510/// Missing directory → empty list.
2511fn load_providers(dir: &Path) -> Result<Vec<ProviderConfig>> {
2512 if !dir.exists() {
2513 return Ok(vec![]);
2514 }
2515 let mut items = vec![];
2516 let mut entries: Vec<_> = std::fs::read_dir(dir)
2517 .with_context(|| format!("reading {}", dir.display()))?
2518 .filter_map(|e| e.ok())
2519 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
2520 .collect();
2521 entries.sort_by_key(|e| e.file_name());
2522 for entry in entries {
2523 items.push(ProviderConfig::load(&entry.path())?);
2524 }
2525 Ok(items)
2526}
2527
2528/// Map legacy mirror file stems to their canonical tier names.
2529///
2530/// Canonical tiers: `dev` / `pond` / `cloud` / `ha`.
2531/// Legacy stems pre-R362: `local` (dev tier), `local-sim` / `sim` (pond tier), `prod` (cloud tier).
2532/// Both forms are accepted; canonical names are preferred for new files.
2533pub fn canonical_tier(stem: &str) -> &str {
2534 match stem {
2535 "local" => "dev",
2536 "local-sim" | "sim" => "pond",
2537 "prod" => "cloud",
2538 other => other,
2539 }
2540}
2541
2542/// Walk `.yah/services/<svc>/` for every service and its mirrors.
2543/// Missing directory → empty map. Mirror file stems are normalized to canonical
2544/// tier names via [`canonical_tier`] so callers always see `dev/pond/cloud/ha`.
2545fn load_services(
2546 dir: &Path,
2547 workspace_root: &Path,
2548) -> Result<BTreeMap<String, ServiceWithMirrors>> {
2549 if !dir.exists() {
2550 return Ok(BTreeMap::new());
2551 }
2552 let mut out = BTreeMap::new();
2553 let mut entries: Vec<_> = std::fs::read_dir(dir)
2554 .with_context(|| format!("reading {}", dir.display()))?
2555 .filter_map(|e| e.ok())
2556 .filter(|e| e.path().is_dir())
2557 .collect();
2558 entries.sort_by_key(|e| e.file_name());
2559
2560 for entry in entries {
2561 let svc_dir = entry.path();
2562 let service_toml = svc_dir.join("service.toml");
2563 if !service_toml.exists() {
2564 // Skip directories without a service.toml — leaves room for
2565 // future siblings (e.g. `secrets/`, `README.md`) without
2566 // triggering false-positive parse errors.
2567 continue;
2568 }
2569 let service = ServiceConfig::load(&service_toml)?;
2570 let mut mirrors = BTreeMap::new();
2571 let mirrors_dir = svc_dir.join("mirrors");
2572 if mirrors_dir.exists() {
2573 let mut menv: Vec<_> = std::fs::read_dir(&mirrors_dir)
2574 .with_context(|| format!("reading {}", mirrors_dir.display()))?
2575 .filter_map(|e| e.ok())
2576 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
2577 .collect();
2578 menv.sort_by_key(|e| e.file_name());
2579 for m in menv {
2580 let path = m.path();
2581 let stem = path
2582 .file_stem()
2583 .and_then(|s| s.to_str())
2584 .unwrap_or("")
2585 .to_string();
2586 let tier = canonical_tier(&stem).to_string();
2587 // Last-write wins if both legacy and canonical forms coexist
2588 // (e.g. local-sim.toml + pond.toml). Sort order ensures the
2589 // canonical file (pond.toml) wins because 'p' > 'l'.
2590 mirrors.insert(tier, MirrorConfig::load(&path)?);
2591 }
2592 }
2593 let mut component_transform_recipes = BTreeMap::new();
2594 for component in &service.components {
2595 if component.kind == "static-asset" {
2596 if let Some(recipe) =
2597 read_component_transform_recipe(workspace_root, &component.path)
2598 {
2599 component_transform_recipes.insert(component.id.clone(), recipe);
2600 }
2601 }
2602 }
2603 let passway_machines = mirrors
2604 .iter()
2605 .filter_map(|(env, m)| m.passway_machines().map(|ms| (env.clone(), ms)))
2606 .collect();
2607 out.insert(
2608 service.name.clone(),
2609 ServiceWithMirrors {
2610 service,
2611 mirrors,
2612 component_transform_recipes,
2613 passway_machines,
2614 },
2615 );
2616 }
2617 Ok(out)
2618}
2619
2620/// Read the first transform recipe name from a component's `workload.toml`.
2621/// Returns `None` when the file is absent or has no `[asset.derive.transform]`
2622/// section. Best-effort — parse failures are silently ignored so a malformed
2623/// workload.toml doesn't abort the entire service catalog load.
2624fn read_component_transform_recipe(workspace_root: &Path, component_path: &str) -> Option<String> {
2625 let workload_path = workspace_root.join(component_path).join("workload.toml");
2626 let text = std::fs::read_to_string(&workload_path).ok()?;
2627 let value: toml::Value = toml::from_str(&text).ok()?;
2628 let assets = value.get("asset")?.as_array()?;
2629 for asset in assets {
2630 if let Some(recipe) = asset
2631 .get("derive")
2632 .and_then(|d| d.get("transform"))
2633 .and_then(|t| t.get("recipe"))
2634 .and_then(|r| r.as_str())
2635 {
2636 return Some(recipe.to_string());
2637 }
2638 }
2639 None
2640}
2641
2642/// Load every `.yah/domains/*.toml` into a [`DomainConfig`] map keyed by
2643/// file stem. Missing directory → empty map.
2644fn load_domains(dir: &Path) -> Result<BTreeMap<String, DomainConfig>> {
2645 if !dir.exists() {
2646 return Ok(BTreeMap::new());
2647 }
2648 let mut out = BTreeMap::new();
2649 let mut entries: Vec<_> = std::fs::read_dir(dir)
2650 .with_context(|| format!("reading {}", dir.display()))?
2651 .filter_map(|e| e.ok())
2652 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
2653 .collect();
2654 entries.sort_by_key(|e| e.file_name());
2655 for entry in entries {
2656 let path = entry.path();
2657 let stem = path
2658 .file_stem()
2659 .and_then(|s| s.to_str())
2660 .unwrap_or("")
2661 .to_string();
2662 let dom = DomainConfig::load(&path)?;
2663 if dom.name != stem {
2664 anyhow::bail!(
2665 "domains/{}.toml: name = \"{}\" must match the file stem",
2666 stem,
2667 dom.name
2668 );
2669 }
2670 out.insert(dom.name.clone(), dom);
2671 }
2672 Ok(out)
2673}
2674
2675/// Load and shape-validate all `*.toml` files in `dir` as [`WorkloadConfig`].
2676fn load_workloads(dir: std::path::PathBuf) -> Result<Vec<WorkloadConfig>> {
2677 if !dir.exists() {
2678 return Ok(vec![]);
2679 }
2680 let mut items = vec![];
2681 let mut entries: Vec<_> = std::fs::read_dir(&dir)
2682 .with_context(|| format!("reading {}", dir.display()))?
2683 .filter_map(|e| e.ok())
2684 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
2685 .collect();
2686 entries.sort_by_key(|e| e.file_name());
2687
2688 for entry in entries {
2689 let path = entry.path();
2690 let path_str = path.display().to_string();
2691 let src =
2692 std::fs::read_to_string(&path).with_context(|| format!("reading {}", path_str))?;
2693 let spec: WorkloadSpec =
2694 toml::from_str(&src).with_context(|| format!("parsing {}", path_str))?;
2695
2696 // Shape-validate before accepting into the loaded config.
2697 validate::shape(&spec)
2698 .map_err(|e| anyhow::anyhow!("workload {} failed shape validation: {e}", path_str))?;
2699
2700 items.push(WorkloadConfig { spec });
2701 }
2702 Ok(items)
2703}
2704
2705/// Load all mirror configs from the `mirrors/` directory.
2706///
2707/// Handles two layouts that may coexist:
2708/// - **Folder**: `mirrors/<id>/mirror.toml` — preferred; allows secrets and
2709/// per-mirror overrides to live next to the config file.
2710/// - **Flat**: `mirrors/<id>.toml` — legacy; still supported.
2711///
2712/// Each file is parsed as [`LegacyMirrorConfig`]. A malformed file returns an error
2713/// that includes the file path and the TOML field path + line/column, so the
2714/// caller can surface it to the user directly.
2715fn load_mirrors(dir: std::path::PathBuf) -> Result<Vec<LegacyMirrorConfig>> {
2716 if !dir.exists() {
2717 return Ok(vec![]);
2718 }
2719 let mut mirrors = vec![];
2720 let mut entries: Vec<_> = std::fs::read_dir(&dir)
2721 .with_context(|| format!("reading {}", dir.display()))?
2722 .filter_map(|e| e.ok())
2723 .collect();
2724 entries.sort_by_key(|e| e.file_name());
2725
2726 for entry in entries {
2727 let path = entry.path();
2728 if path.is_dir() {
2729 // Folder layout: mirrors/<id>/mirror.toml
2730 let mirror_toml = path.join("mirror.toml");
2731 if mirror_toml.exists() {
2732 let src = std::fs::read_to_string(&mirror_toml)
2733 .with_context(|| format!("reading {}", mirror_toml.display()))?;
2734 let cfg: LegacyMirrorConfig = toml::from_str(&src)
2735 .with_context(|| format!("parsing {}", mirror_toml.display()))?;
2736 mirrors.push(cfg);
2737 }
2738 } else if path.extension().map_or(false, |e| e == "toml") {
2739 // Flat layout: mirrors/<id>.toml
2740 let src = std::fs::read_to_string(&path)
2741 .with_context(|| format!("reading {}", path.display()))?;
2742 let cfg: LegacyMirrorConfig =
2743 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
2744 mirrors.push(cfg);
2745 }
2746 }
2747 Ok(mirrors)
2748}
2749
2750/// Load `topology.toml` if it exists; return a default (empty) topology otherwise.
2751fn load_topology(path: std::path::PathBuf) -> Result<TopologyConfig> {
2752 if !path.exists() {
2753 return Ok(TopologyConfig::default());
2754 }
2755 let src =
2756 std::fs::read_to_string(&path).with_context(|| format!("reading {}", path.display()))?;
2757 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
2758}
2759
2760/// R555-S1: entries are sorted by file name before parsing, so "declaration
2761/// order in `.yah/infra/machines/` breaks ties" — the contract
2762/// [`CloudConfig::admit_workload`] documents — is actually true. `read_dir`
2763/// yields filesystem order, which is unspecified and differs between APFS and
2764/// a hashed-dir ext4; without the sort, *which* of two equally-matching nodes a
2765/// workload admits to could change when an unrelated file is added to the
2766/// directory. That was latent while each tag set had one match and became
2767/// observable the day us-west-003 joined us-west-002 on
2768/// `[tag:build-worker, arch:x86, os:linux]`. Same sort `load_providers` has
2769/// always done.
2770fn load_dir<T: for<'de> Deserialize<'de>>(dir: std::path::PathBuf) -> Result<Vec<T>> {
2771 if !dir.exists() {
2772 return Ok(vec![]);
2773 }
2774 let mut entries: Vec<_> = std::fs::read_dir(&dir)
2775 .with_context(|| format!("reading {}", dir.display()))?
2776 .collect::<std::io::Result<Vec<_>>>()
2777 .with_context(|| format!("reading {}", dir.display()))?;
2778 entries.sort_by_key(|e| e.file_name());
2779
2780 let mut items = vec![];
2781 for entry in entries {
2782 let path = entry.path();
2783 if path.extension().map_or(false, |e| e == "toml") {
2784 let src = std::fs::read_to_string(&path)
2785 .with_context(|| format!("reading {}", path.display()))?;
2786 let item: T =
2787 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
2788 items.push(item);
2789 }
2790 }
2791 Ok(items)
2792}
2793
2794// ─── New manifest shapes (R222 B2) ───────────────────────────────────────────
2795//
2796// The post-R215 layout splits substrate from service declarations:
2797//
2798// .yah/infra/providers/<id>.toml → ProviderConfig
2799// .yah/services/<svc>/service.toml → ServiceConfig
2800// .yah/services/<svc>/mirrors/<env>.toml → MirrorConfig
2801//
2802// CloudConfig::load still reads the legacy layout — B3 swaps in these types
2803// and removes the Legacy* shapes plus TopologyConfig.
2804
2805/// Tag for the infrastructure provider kind. Drives which fields are valid in
2806/// a [`ProviderConfig`] body or a [`MirrorProviderSlot::Inline`] block.
2807///
2808/// Two flavors:
2809/// - **Account/runtime providers** (`cloudflare`, `hetzner`, `local-container`)
2810/// live as files under `.yah/infra/providers/<id>.toml` and are referenced
2811/// from a mirror via `use = "<id>"`.
2812/// - **Inline-only providers** (`local-static`, `miniflare-container`,
2813/// `minio-container`) declare an operator-local stand-in directly inside a
2814/// mirror via `kind = "..."`. They carry no credentials and have no provider
2815/// file. The container-backed kinds ride on top of whichever
2816/// `local-container` runtime is declared in infra (orbstack/colima/docker);
2817/// the reconciler resolves the runtime at up-time.
2818#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
2819#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
2820#[serde(rename_all = "kebab-case")]
2821pub enum Provider {
2822 /// Cloudflare account: R2 buckets, DNS, Workers, Tunnels.
2823 Cloudflare,
2824 /// Hetzner Cloud + Object Storage account.
2825 Hetzner,
2826 /// Vultr cloud VPS — auto-provisioned via the `cloud.vps.*` Envoy
2827 /// (`VultrEnvoy`), the burst/scaling counterpart to Hetzner. Driver-backed.
2828 Vultr,
2829 /// BYO bare/static node (OVH, on-prem, anything we did NOT provision via a
2830 /// cloud API). Brought up over SSH (`stand-up-yubaba.sh` / `yah cloud
2831 /// machine bootstrap`); reach is declared in the machine's `[connect]`
2832 /// block. No create/destroy driver — placement-only.
2833 Static,
2834 /// Built-in static-file server bound to localhost. Inline-only; never
2835 /// declared as a standalone provider file because it carries no creds.
2836 LocalStatic,
2837 /// Local container runtime (orbstack/colima/docker). Configured by a
2838 /// provider file under `.yah/infra/providers/` so the discovery hints +
2839 /// runtime override sit in one place.
2840 LocalContainer,
2841 /// Dev-tier compute: the component runs as a kamaji-supervised host
2842 /// process against the operator's real workspace, no container and no
2843 /// build step per edit. Inline-only — it carries no credentials, and
2844 /// "the machine you are sitting at" is not an account to point at.
2845 /// See `reconciler::local_process`.
2846 LocalProcess,
2847 /// Containerized miniflare (workerd subprocess) fronting MinIO — the
2848 /// pond-tier stand-in for a CF Worker + R2 static surface. Inline-only;
2849 /// the reconciler spawns miniflare via the JS runtime and starts a MinIO
2850 /// container on the local-container runtime.
2851 MiniflareContainer,
2852 /// Containerized MinIO providing an S3-compatible API — the pond-tier
2853 /// stand-in for Cloudflare R2. Inline-only; the reconciler spins up the
2854 /// container on the local-container runtime and auto-creates the declared
2855 /// bucket on first up.
2856 MinioContainer,
2857 /// Dev-tier PostgreSQL — a real server speaking real pgwire on loopback,
2858 /// supervised by kamaji as the `yah-pg-dev` workload (W265, R584-F1). No
2859 /// docker daemon: the driver fetches a per-arch PostgreSQL tarball on first
2860 /// run and `initdb`s a cluster under `.yah/infra/state/dev/pg/`.
2861 ///
2862 /// Inline-only — it carries no credentials worth a provider file (the
2863 /// cluster is loopback-bound with a fixed dev password). Declared under
2864 /// [`MirrorConfig::drivers`], not `providers`:
2865 ///
2866 /// ```toml
2867 /// [drivers.pg]
2868 /// kind = "local-pg-dev"
2869 /// ```
2870 LocalPgDev,
2871}
2872
2873/// A provider account/runtime binding from `.yah/infra/providers/<id>.toml`.
2874///
2875/// The `kind` discriminator picks the schema for the remaining fields. Strict
2876/// on `kind` (unknown values are a parse error); permissive on per-kind fields
2877/// (carried as a free-form map so this loader stays stable as new fields land).
2878/// B3/B4 will tighten by introducing typed variants alongside JSON Schema.
2879#[derive(Debug, Clone, Serialize, Deserialize)]
2880#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
2881pub struct ProviderConfig {
2882 pub schema_version: u32,
2883 pub id: String,
2884 pub kind: Provider,
2885 /// Reference into the OS keystore for live credentials (e.g.
2886 /// `"keystore://cloudflare/yah"`). `None` for providers that don't need
2887 /// creds (local-static, optionally local-container).
2888 #[serde(default, skip_serializing_if = "Option::is_none")]
2889 pub credentials: Option<String>,
2890 /// Kind-specific fields. Examples:
2891 /// - cloudflare: `default_zone`
2892 /// - hetzner: `default_location`, `default_server_type`, `ssh_keys`
2893 /// - local-container: `runtime`, `discovery`
2894 #[serde(flatten)]
2895 #[cfg_attr(
2896 feature = "json-schema",
2897 schemars(with = "std::collections::BTreeMap<String, serde_json::Value>")
2898 )]
2899 pub fields: BTreeMap<String, toml::Value>,
2900}
2901
2902impl ProviderConfig {
2903 /// Parse a single `providers/<id>.toml` file.
2904 pub fn load(path: &Path) -> Result<Self> {
2905 let src =
2906 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
2907 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
2908 }
2909}
2910
2911/// An operator-facing service declaration from
2912/// `.yah/services/<svc>/service.toml`.
2913///
2914/// A service groups one or more components (a static surface, a containerized
2915/// API, an almanac…) under a single domain. Mirrors project the service onto
2916/// concrete infra; see [`MirrorConfig`].
2917#[derive(Debug, Clone, Serialize, Deserialize)]
2918#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
2919pub struct ServiceConfig {
2920 pub schema_version: u32,
2921 pub name: String,
2922 pub domain: String,
2923 #[serde(default, skip_serializing_if = "Vec::is_empty")]
2924 pub components: Vec<ServiceComponent>,
2925 /// Databases this service exposes, grouped by environment (W241). Every
2926 /// entry becomes a data-workbench / `sql_*` catalog id of the shape
2927 /// `<env>:<service>:<name>` (e.g. `pond:scrabcake:main`). Optional and
2928 /// default-empty — services without databases omit the `[db]` table
2929 /// entirely.
2930 #[serde(default, skip_serializing_if = "DbCatalog::is_empty")]
2931 pub db: DbCatalog,
2932}
2933
2934impl ServiceConfig {
2935 /// Parse a single `services/<svc>/service.toml` file.
2936 pub fn load(path: &Path) -> Result<Self> {
2937 let src =
2938 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
2939 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
2940 }
2941
2942 /// Persist to `.yah/services/<name>/service.toml`, creating the service
2943 /// directory if needed. Create-or-overwrite — the canonical replacement
2944 /// for the legacy `sites.json` write path. `workspace_root` is the camp
2945 /// dir (the parent of `.yah/`).
2946 pub fn save(&self, workspace_root: &Path) -> Result<()> {
2947 let dir = crate::paths::service_dir(workspace_root, &self.name);
2948 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
2949 let path = crate::paths::service_toml(workspace_root, &self.name);
2950 let s = toml::to_string_pretty(self)
2951 .with_context(|| format!("serializing service {}", self.name))?;
2952 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
2953 }
2954
2955 /// Remove `.yah/services/<name>/` and everything under it (service.toml
2956 /// plus its `mirrors/`). Returns `false` when the directory was already
2957 /// absent, so callers can distinguish "deleted" from "no-op".
2958 pub fn delete(workspace_root: &Path, name: &str) -> Result<bool> {
2959 let dir = crate::paths::service_dir(workspace_root, name);
2960 if !dir.exists() {
2961 return Ok(false);
2962 }
2963 std::fs::remove_dir_all(&dir).with_context(|| format!("removing {}", dir.display()))?;
2964 Ok(true)
2965 }
2966}
2967
2968/// A git source for a component (R561-F1, "BYO git").
2969///
2970/// When a [`ServiceComponent`] sets `git`, the component's code is NOT in this
2971/// workspace — it lives in an external repo that the reconciler shallow-clones
2972/// into a source cache before build (approach A: clone-at-reconcile, so config
2973/// load + validation stay offline). The component's `path` is then interpreted
2974/// relative to `<checkout>/<subdir>` instead of the workspace root.
2975#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
2976#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
2977pub struct GitSource {
2978 /// Clone URL (https or ssh) of the tenant repo.
2979 pub repo: String,
2980 /// Branch, tag, or commit SHA to check out. Defaults to `"main"`.
2981 #[serde(default = "default_git_ref")]
2982 pub r#ref: String,
2983 /// Optional sub-directory within the repo that the workspace is rooted at
2984 /// (e.g. a monorepo's `site/`). `path` is resolved relative to this.
2985 #[serde(default, skip_serializing_if = "Option::is_none")]
2986 pub subdir: Option<String>,
2987}
2988
2989fn default_git_ref() -> String {
2990 "main".to_string()
2991}
2992
2993/// How to reach an external infra root (R615-F1 / W274, "linked infra
2994/// sources"): a filesystem link to a sibling camp's live tree, or a git
2995/// checkout of an extracted infra repo.
2996#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
2997#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
2998#[serde(tag = "kind", rename_all = "kebab-case")]
2999pub enum InfraSourceKind {
3000 /// Filesystem link — reads the owner's live tree. The dev-loop shortcut,
3001 /// and the whole story until W274's "infra as its own repo" end-state.
3002 /// `path` is relative to *this* camp's root; infra is read from
3003 /// `<path>/.yah/infra/`.
3004 Path {
3005 path: String,
3006 },
3007 /// Git link — reused verbatim from [`GitSource`] (R561, "BYO git"),
3008 /// lifted here from "a component's code" to "a camp's infra registry."
3009 /// Loading stays offline (W274 §3): `yah infra sync` (R615-T3) is what
3010 /// clones/pulls this into `.yah/cache/infra/<owner>/`; `CloudConfig::load`
3011 /// only ever reads that cache, never the network.
3012 Git(GitSource),
3013}
3014
3015/// Write-gate for a linked [`InfraSource`] (R615-F1 / W274).
3016///
3017/// An enum, not a bool: the two states today are "borrower renders/plans but
3018/// cannot reconcile" and "this camp genuinely co-administers the shared
3019/// root," and a future read-write-with-approval tier is a third variant, not
3020/// a renamed boolean.
3021#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq, Eq)]
3022#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3023#[serde(rename_all = "kebab-case")]
3024pub enum SourceMode {
3025 /// Borrower can render and plan against the linked entries but cannot
3026 /// reconcile/mutate them — the owner remains the single manager. Default:
3027 /// a borrower is opt-in to write access, never opt-out of the safe state.
3028 #[default]
3029 ReadOnly,
3030 /// Escape hatch for a camp that genuinely co-administers a shared root.
3031 Manage,
3032}
3033
3034/// One `[[source]]` entry in `.yah/infra/sources.toml` (R615-F1 / W274) — an
3035/// external infra root this camp borrows machines/providers from.
3036#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3037#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3038pub struct InfraSource {
3039 /// Logical owner name, badged in the Infra tab (e.g. `"yah"`). Distinct
3040 /// from any camp/repo name the `kind` resolves through — this is what an
3041 /// operator sees on a borrowed row, not a path.
3042 pub owner: String,
3043 #[serde(flatten)]
3044 pub kind: InfraSourceKind,
3045 #[serde(default)]
3046 pub mode: SourceMode,
3047 /// Optional filter — name globs or mesh-tag selectors — to borrow a
3048 /// subset of the source root rather than everything it declares. Empty
3049 /// (the default) borrows everything.
3050 #[serde(default)]
3051 pub select: Vec<String>,
3052}
3053
3054fn default_sources_schema_version() -> u32 {
3055 1
3056}
3057
3058/// `.yah/infra/sources.toml` — the ordered list of external infra roots this
3059/// camp borrows from (R615-F1 / W274).
3060#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3061#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3062pub struct SourcesConfig {
3063 #[serde(default = "default_sources_schema_version")]
3064 pub schema_version: u32,
3065 /// `[[source]]` entries, in declaration order — overlay order matters
3066 /// when two linked sources both name the same machine (R615-F2).
3067 #[serde(default, rename = "source")]
3068 pub source: Vec<InfraSource>,
3069}
3070
3071impl Default for SourcesConfig {
3072 fn default() -> Self {
3073 Self {
3074 schema_version: default_sources_schema_version(),
3075 source: Vec::new(),
3076 }
3077 }
3078}
3079
3080impl SourcesConfig {
3081 /// Load `<infra_dir>/sources.toml`. A missing file is not an error —
3082 /// every camp without linked infra has none, which today is every camp —
3083 /// and yields an empty source list rather than `Err`.
3084 pub fn load(infra_dir: &Path) -> Result<Self> {
3085 let path = infra_dir.join("sources.toml");
3086 if !path.exists() {
3087 return Ok(Self::default());
3088 }
3089 let src =
3090 std::fs::read_to_string(&path).with_context(|| format!("reading {}", path.display()))?;
3091 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
3092 }
3093}
3094
3095impl InfraSource {
3096 /// Human-readable descriptor of *which* source this is, for
3097 /// [`InfraOrigin::source`] — distinguishes two linked sources from the
3098 /// same owner. Never includes credentials: `GitSource.repo` is a clone
3099 /// URL (https/ssh), the same thing R561 already treats as safe to log,
3100 /// with any real secret resolved separately via `keystore://` (W274's
3101 /// own precedent).
3102 fn describe(&self) -> String {
3103 match &self.kind {
3104 InfraSourceKind::Path { path } => format!("path:{path}"),
3105 InfraSourceKind::Git(g) => format!("git:{}@{}", g.repo, g.r#ref),
3106 }
3107 }
3108
3109 /// Resolve this source to an infra root directory (R615-F2 / W274 §3).
3110 /// Does no I/O and touches no network: `path` sources read the owner's
3111 /// live tree directly; `git` sources read wherever `yah infra sync`
3112 /// (R615-T3) last synced to, which may not exist yet (an unsynced git
3113 /// source overlays nothing, not an error — see [`load_dir_tolerant`]).
3114 ///
3115 /// `git.subdir` (reused verbatim from [`GitSource`]/R561) is honoured
3116 /// exactly like the component case: the checkout root when unset, or
3117 /// `<checkout>/<subdir>` when set — e.g. `subdir = "infra"` for a
3118 /// monorepo whose infra registry lives under `infra/` rather than at the
3119 /// clone's root. `yah infra sync` (R615-T3) clones into the *checkout*
3120 /// root ([`crate::paths::infra_source_cache_dir`]), never into a
3121 /// subdir-suffixed path, so this is the one place that appends `subdir`.
3122 fn infra_root(&self, workspace_root: &Path) -> std::path::PathBuf {
3123 match &self.kind {
3124 InfraSourceKind::Path { path } => workspace_root.join(path).join(".yah").join("infra"),
3125 InfraSourceKind::Git(g) => {
3126 let checkout = crate::paths::infra_source_cache_dir(workspace_root, &self.owner);
3127 match g.subdir.as_deref() {
3128 Some(subdir) => checkout.join(subdir),
3129 None => checkout,
3130 }
3131 }
3132 }
3133 }
3134}
3135
3136/// Provenance for a [`MachineConfig`] or [`ProviderConfig`] pulled in from a
3137/// linked `.yah/infra/sources.toml` entry, rather than declared in this
3138/// camp's own `.yah/infra/` (R615-F2 / W274).
3139///
3140/// Lives in [`CloudConfig::machine_origins`] / `provider_origins`, keyed by
3141/// name/id, rather than as a field on `MachineConfig`/`ProviderConfig`
3142/// themselves: those two types are constructed by struct literal in test
3143/// helpers across several crates (including ones this ticket has no reason to
3144/// touch), so widening either shape would ripple out past this crate for no
3145/// semantic gain — origin is a property of *this load*, not an inherent
3146/// property of the machine/provider. A name absent from the map is
3147/// camp-local; present means borrowed, and the Infra tab (R615-F4) / reconcile
3148/// gating (`InfraSource::mode`, copied onto `mode` below) read it from here.
3149#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3150#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3151pub struct InfraOrigin {
3152 /// The [`InfraSource::owner`] that supplied this entry, e.g. `"yah"`.
3153 pub owner: String,
3154 /// Which source, rendered — see [`InfraSource::describe`].
3155 pub source: String,
3156 /// The write-gate that applied when this entry was overlaid — copied
3157 /// from [`InfraSource::mode`] so a caller holding just the machine/
3158 /// provider doesn't need the source list in hand to know it's borrowed
3159 /// read-only.
3160 pub mode: SourceMode,
3161}
3162
3163/// Like [`load_dir`], but tolerant **per file**: a foreign infra root (an
3164/// owner's live tree, or a synced git checkout) can carry entries this
3165/// binary's `T` predates — noisetable's pre-migration machines used an older
3166/// schema than yah's, and the reverse will happen too as each side evolves
3167/// independently. One unparseable file on a source this camp doesn't own must
3168/// never sink every other entry in the same directory, let alone this camp's
3169/// own load (R615-F2 gotcha). Contrast [`load_dir`], which stays strict for
3170/// camp-local files, where a malformed TOML genuinely should be a hard error.
3171///
3172/// Returns the entries that parsed, plus `(path, error)` for every file that
3173/// didn't — the caller logs those, it doesn't drop them silently. A missing
3174/// or unreadable directory yields `(vec![], vec![])`, same "no entries" as
3175/// `load_dir`'s `!dir.exists()` case (an unsynced git source, or a source
3176/// root with no `providers/` at all, are both normal, not warnings).
3177fn load_dir_tolerant<T: for<'de> Deserialize<'de>>(
3178 dir: &Path,
3179) -> (Vec<T>, Vec<(std::path::PathBuf, anyhow::Error)>) {
3180 let Ok(read_dir) = std::fs::read_dir(dir) else {
3181 return (Vec::new(), Vec::new());
3182 };
3183 let mut entries: Vec<_> = read_dir.filter_map(|e| e.ok()).collect();
3184 entries.sort_by_key(|e| e.file_name());
3185
3186 let mut items = Vec::new();
3187 let mut skipped = Vec::new();
3188 for entry in entries {
3189 let path = entry.path();
3190 if path.extension().map_or(true, |e| e != "toml") {
3191 continue;
3192 }
3193 let parsed = std::fs::read_to_string(&path)
3194 .with_context(|| format!("reading {}", path.display()))
3195 .and_then(|src| {
3196 toml::from_str::<T>(&src).with_context(|| format!("parsing {}", path.display()))
3197 });
3198 match parsed {
3199 Ok(item) => items.push(item),
3200 Err(e) => skipped.push((path, e)),
3201 }
3202 }
3203 (items, skipped)
3204}
3205
3206/// Whether a borrowed machine passes an [`InfraSource::select`] filter
3207/// (R615-F2 / W274). Empty `select` borrows everything. A non-empty `select`
3208/// entry matches either the machine's exact `name` or literal membership in
3209/// its `mesh_tags` — the one shape W274's own example uses
3210/// (`select = ["tag:cloud-runner"]`). Not a glob engine: mesh tags are
3211/// already flat strings compared for exact equality everywhere else in this
3212/// crate (see `resolve_machine_by_mesh_tags`), so a select entry is that same
3213/// comparison, not a new pattern language.
3214fn machine_matches_select(machine: &MachineConfig, select: &[String]) -> bool {
3215 select.is_empty()
3216 || select
3217 .iter()
3218 .any(|s| *s == machine.name || machine.mesh_tags.contains(s))
3219}
3220
3221/// What one `[[source]]` in `.yah/infra/sources.toml` actually contributed to
3222/// [`FleetInventory`] on this load (R870-B13).
3223///
3224/// Recorded because a link that resolves to *nothing* is indistinguishable, at
3225/// every downstream use site, from a camp that declared no link at all — and
3226/// that is precisely the failure this ticket exists to fix. A source that
3227/// contributes zero machines is not an error here (an unsynced `kind = "git"`
3228/// source is legitimately empty, and `load()` must stay offline), so instead
3229/// the fact is *carried* to whoever fails for want of a machine. See
3230/// [`FleetInventory::describe_sources`].
3231#[derive(Debug, Clone, PartialEq, Eq)]
3232pub struct SourceContribution {
3233 /// [`InfraSource::owner`] — the name this camp knows the fleet by.
3234 pub owner: String,
3235 /// The source, rendered — see [`InfraSource::describe`].
3236 pub source: String,
3237 /// Where the link resolved to, i.e. the foreign `.yah/infra/`.
3238 pub root: std::path::PathBuf,
3239 /// Whether `root` exists on disk. `false` for a `kind = "path"` link
3240 /// aimed at a directory that is not a camp, and for a `kind = "git"`
3241 /// source that `yah infra sync` has never fetched.
3242 pub root_exists: bool,
3243 /// How many machines this source actually added to the inventory — after
3244 /// [`InfraSource::select`] filtering and after losing every name a
3245 /// camp-local entry or an earlier source already claimed.
3246 pub machines: usize,
3247}
3248
3249/// A camp's resolved machine inventory: **the** answer to "which machines does
3250/// this camp have", with exactly one implementation
3251/// ([`resolve_fleet_inventory`]) behind it (R870-B13).
3252///
3253/// A borrowing camp — one whose own `.yah/infra/machines/` is empty and which
3254/// declares `[[source]]` links to another camp's fleet in
3255/// `.yah/infra/sources.toml` — is the case this type exists for. Before it,
3256/// the overlay was applied inline inside [`CloudConfig::load`], so the two
3257/// callers that resolve a *machine name to a machine* (ingress collation and
3258/// the sovereign apex render) read a camp-local-only loader and saw an empty
3259/// fleet. There was no bug in either of them; the inventory simply had two
3260/// readers that disagreed about what the inventory was.
3261#[derive(Debug)]
3262pub struct FleetInventory {
3263 /// Camp-local machines first, then each source's contribution in
3264 /// declaration order. Camp-local wins any name collision; among sources,
3265 /// the earlier-declared one wins.
3266 pub machines: Vec<MachineConfig>,
3267 /// Provenance for the borrowed entries, keyed by [`MachineConfig::name`].
3268 /// A name absent here is camp-local. Same shape and meaning as
3269 /// [`CloudConfig::machine_origins`], which is populated from this.
3270 pub origins: BTreeMap<String, InfraOrigin>,
3271 /// The parsed `.yah/infra/sources.toml`, kept so a caller that already has
3272 /// an inventory in hand does not re-read it (`CloudConfig::load` overlays
3273 /// providers from the same list).
3274 pub sources: SourcesConfig,
3275 /// Per-source accounting — see [`SourceContribution`].
3276 pub contributions: Vec<SourceContribution>,
3277}
3278
3279impl FleetInventory {
3280 /// One line per declared `[[source]]`, for attaching to the error a caller
3281 /// raises when a machine name does not resolve (R870-B13).
3282 ///
3283 /// The failure being diagnosed is always "I was told about machine X and
3284 /// cannot find it", and the three ways a borrowing camp gets there — no
3285 /// link declared, a link pointing somewhere that is not a camp, a link
3286 /// whose `select` filtered X out — are indistinguishable from the name
3287 /// alone. Empty string when the camp declares no sources, so the caller
3288 /// can append it unconditionally without emitting a dangling header.
3289 pub fn describe_sources(&self) -> String {
3290 if self.contributions.is_empty() {
3291 return String::new();
3292 }
3293 let mut out = String::from("linked infra sources consulted:");
3294 for c in &self.contributions {
3295 out.push_str(&format!(
3296 "\n {} ({}) -> {}{} — contributed {} machine(s)",
3297 c.owner,
3298 c.source,
3299 c.root.display(),
3300 if c.root_exists {
3301 ""
3302 } else {
3303 " [ABSENT: not a camp, or an unsynced git source]"
3304 },
3305 c.machines,
3306 ));
3307 }
3308 out
3309 }
3310}
3311
3312/// Resolve a camp's machine inventory: camp-local `.yah/infra/machines/`, the
3313/// pre-R215 `.yah/cloud/machines/` tree, then every machine borrowed through
3314/// `.yah/infra/sources.toml` (R870-B13, on R615-F2's mechanism).
3315///
3316/// **How a camp names another camp's fleet**, decided here rather than
3317/// invented: through the `[[source]]` entry R615-F1 already defines — `owner`
3318/// is the logical name an operator sees, `kind = "path"` resolves against the
3319/// borrowing camp's own root and `kind = "git"` against `yah infra sync`'s
3320/// cache. There is deliberately no second naming scheme: a camp that could
3321/// name a foreign fleet two ways would be a camp whose inventory can drift
3322/// from itself, which is the thing this ticket rejected.
3323///
3324/// **There is exactly one copy.** A `kind = "path"` source reads the owner's
3325/// live tree at `<path>/.yah/infra/` on every load — the borrowing camp
3326/// persists nothing, so the two can never disagree. `kind = "git"` reads a
3327/// synced checkout, which *is* a copy, but an explicit one with a named
3328/// refresh verb (`yah infra sync`) and a pinned `ref`; that is the cache with
3329/// an invalidation story, as against a hand-maintained second inventory.
3330///
3331/// Camp-local files are strict (a malformed TOML this camp owns is a hard
3332/// error) and foreign files are tolerant per-file (R615-F2: a foreign entry
3333/// whose schema this binary predates must not sink the load). A foreign
3334/// machine skipped that way is not silently lost — it fails loudly at the
3335/// point some caller needs it, with [`FleetInventory::describe_sources`]
3336/// naming the link it should have come from.
3337///
3338/// Deliberately *without* [`CloudConfig::load`]'s R844-B7 wrong-root guard: a
3339/// missing `.yah/infra/machines/` is an empty inventory here, because the
3340/// callers that resolve against it (ingress collation, apex render) are handed
3341/// a root that a `CloudConfig::load` already accepted.
3342pub fn resolve_fleet_inventory(workspace_root: &Path) -> Result<FleetInventory> {
3343 let mut machines = load_dir::<MachineConfig>(crate::paths::machines_dir(workspace_root))?;
3344
3345 // Pre-R215 `.yah/cloud/machines/`. Shouldn't have anything since R215-B1
3346 // moved them, but if it does we dedupe by name — R215+ wins.
3347 let cloud_dir = crate::paths::legacy_cloud_dir(workspace_root);
3348 if cloud_dir.exists() {
3349 let names: std::collections::HashSet<String> =
3350 machines.iter().map(|m| m.name.clone()).collect();
3351 for m in load_dir::<MachineConfig>(cloud_dir.join("machines"))? {
3352 if !names.contains(&m.name) {
3353 machines.push(m);
3354 }
3355 }
3356 }
3357
3358 // `SourcesConfig::load` never touches the network — git sources are read
3359 // from `yah infra sync`'s cache (R615-T3) — so this keeps the whole
3360 // offline contract `CloudConfig::load` has always had.
3361 let sources = SourcesConfig::load(&crate::paths::infra_dir(workspace_root))?;
3362 let mut origins = BTreeMap::new();
3363 let contributions = overlay_source_machines(workspace_root, &sources, &mut machines, &mut origins);
3364
3365 Ok(FleetInventory {
3366 machines,
3367 origins,
3368 sources,
3369 contributions,
3370 })
3371}
3372
3373/// Overlay every linked `.yah/infra/sources.toml` source's machines into
3374/// `machines`, recording provenance into `machine_origins` (R615-F2 / W274).
3375/// Must be called AFTER camp-local entries are already in the vector:
3376/// collision resolution is "first writer wins," so seeding with camp-local
3377/// first is what makes camp-local win over every source, and an earlier source
3378/// win over a later one.
3379///
3380/// `select` filters which machines a source contributes. Returns one
3381/// [`SourceContribution`] per declared source, in declaration order.
3382fn overlay_source_machines(
3383 workspace_root: &Path,
3384 sources: &SourcesConfig,
3385 machines: &mut Vec<MachineConfig>,
3386 machine_origins: &mut BTreeMap<String, InfraOrigin>,
3387) -> Vec<SourceContribution> {
3388 let mut seen_machine_names: std::collections::HashSet<String> =
3389 machines.iter().map(|m| m.name.clone()).collect();
3390 let mut contributions = Vec::with_capacity(sources.source.len());
3391
3392 for source in &sources.source {
3393 let root = source.infra_root(workspace_root);
3394 let origin = InfraOrigin {
3395 owner: source.owner.clone(),
3396 source: source.describe(),
3397 mode: source.mode,
3398 };
3399
3400 let (foreign_machines, skipped) = load_dir_tolerant::<MachineConfig>(&root.join("machines"));
3401 for (path, e) in skipped {
3402 tracing::warn!(
3403 "infra source {:?} ({}): skipping unparseable machine {}: {e:#}",
3404 source.owner,
3405 root.display(),
3406 path.display()
3407 );
3408 }
3409 let mut added = 0usize;
3410 for m in foreign_machines {
3411 if seen_machine_names.contains(&m.name) {
3412 continue; // camp-local, or an earlier source, already claimed this name
3413 }
3414 if !machine_matches_select(&m, &source.select) {
3415 continue;
3416 }
3417 seen_machine_names.insert(m.name.clone());
3418 machine_origins.insert(m.name.clone(), origin.clone());
3419 machines.push(m);
3420 added += 1;
3421 }
3422
3423 contributions.push(SourceContribution {
3424 owner: source.owner.clone(),
3425 source: source.describe(),
3426 root_exists: root.is_dir(),
3427 root,
3428 machines: added,
3429 });
3430 }
3431
3432 contributions
3433}
3434
3435/// Overlay every linked source's providers into `providers`, recording
3436/// provenance into `provider_origins` (R615-F2 / W274). Same first-writer-wins
3437/// rule as [`overlay_source_machines`], and the same requirement that
3438/// camp-local entries already be in the vector.
3439///
3440/// [`InfraSource::select`] deliberately does not apply: nothing in W274 or
3441/// R615-F1 describes a provider-scoped filter — every provider a source
3442/// declares either overlays whole or, on an id collision, doesn't.
3443fn overlay_source_providers(
3444 workspace_root: &Path,
3445 sources: &SourcesConfig,
3446 providers: &mut Vec<ProviderConfig>,
3447 provider_origins: &mut BTreeMap<String, InfraOrigin>,
3448) {
3449 let mut seen_provider_ids: std::collections::HashSet<String> =
3450 providers.iter().map(|p| p.id.clone()).collect();
3451
3452 for source in &sources.source {
3453 let root = source.infra_root(workspace_root);
3454 let origin = InfraOrigin {
3455 owner: source.owner.clone(),
3456 source: source.describe(),
3457 mode: source.mode,
3458 };
3459
3460 let (foreign_providers, skipped) =
3461 load_dir_tolerant::<ProviderConfig>(&root.join("providers"));
3462 for (path, e) in skipped {
3463 tracing::warn!(
3464 "infra source {:?} ({}): skipping unparseable provider {}: {e:#}",
3465 source.owner,
3466 root.display(),
3467 path.display()
3468 );
3469 }
3470 for p in foreign_providers {
3471 if seen_provider_ids.contains(&p.id) {
3472 continue;
3473 }
3474 seen_provider_ids.insert(p.id.clone());
3475 provider_origins.insert(p.id.clone(), origin.clone());
3476 providers.push(p);
3477 }
3478 }
3479}
3480
3481/// One component of a [`ServiceConfig`]. The `kind` (e.g. `"mesofact-static"`,
3482/// `"almanac"`, `"container"`) selects which reconciler runs against the
3483/// pointed-at workload manifest.
3484#[derive(Debug, Clone, Serialize, Deserialize)]
3485#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3486pub struct ServiceComponent {
3487 pub id: String,
3488 pub kind: String,
3489 /// Path of the directory holding this component's `workload.toml`. Relative
3490 /// to the workspace root for in-tree components, or to the materialized
3491 /// `<checkout>/<subdir>` when [`git`](Self::git) is set.
3492 pub path: String,
3493 /// Optional external git source (R561-F1). When set, the component's code
3494 /// is materialized by shallow-clone before build; see [`GitSource`].
3495 #[serde(default, skip_serializing_if = "Option::is_none")]
3496 pub git: Option<GitSource>,
3497 /// Operator-facing role label, e.g. `"static"`, `"dynamic"`, `"compute"`.
3498 pub role: String,
3499 /// Optional artifact kind this component publishes (`"static"`,
3500 /// `"container-image"`, …). Drives mirror provider-slot routing.
3501 #[serde(default, skip_serializing_if = "Option::is_none")]
3502 pub publishes: Option<String>,
3503 /// URL sub-path a static component's build output is published under,
3504 /// relative to the service's publish prefix (R746). `None` = the service
3505 /// root, which is what every pre-R746 component means.
3506 ///
3507 /// Static publishers lay a component's `out_dir` down at
3508 /// `<bucket>/<service>/<env>/…` and the front door fetches
3509 /// `${ASSET_ORIGIN}/<request path>` — the request path *is* the key. So a
3510 /// service with two static components had them overwrite each other at
3511 /// one prefix, and there was no way to say "this bundle serves under
3512 /// /app". `mount` is that: it appends to the publish prefix, which makes
3513 /// the URL sub-path and the storage sub-path the same string by
3514 /// construction rather than by two manifests agreeing.
3515 ///
3516 /// Cross-checked against the domain route that names the component
3517 /// ([`CloudConfig::cross_ref_validate`]): a component mounted at `/app`
3518 /// must be routed at `/app` or `/app/*`, because a disagreement means
3519 /// requests land on a prefix nothing published to — a 404 whose cause is
3520 /// two files apart.
3521 #[serde(default, skip_serializing_if = "Option::is_none")]
3522 pub mount: Option<String>,
3523 /// Sync-wave index (0-based). Components in wave 0 roll out in parallel
3524 /// first; the reconciler waits for all wave-N components to become healthy
3525 /// before starting wave N+1. Defaults to 0 (all components in one wave).
3526 #[serde(default, skip_serializing_if = "is_zero_u32")]
3527 pub wave: u32,
3528}
3529
3530#[inline]
3531fn is_zero_u32(n: &u32) -> bool {
3532 *n == 0
3533}
3534
3535/// A service's declared databases, grouped by environment (W241 §Sections).
3536/// Parsed from the `[db]` table of `service.toml`; each `[[db.<env>]]` array
3537/// entry names one database. The environment tag drives backend selection at
3538/// query time (see the data-workbench's `db.query` / the `sql_*` MCP tools):
3539/// `dev` = local file, `pond` = a DB inside the running pond container stack
3540/// (reached on a declared localhost port), `cloud` = a remote libSQL/Turso or
3541/// Postgres endpoint whose auth comes from an env var (never stored in TOML).
3542#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq, Eq)]
3543#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3544pub struct DbCatalog {
3545 /// Local-file SQLite databases used in dev mode.
3546 #[serde(default, skip_serializing_if = "Vec::is_empty")]
3547 pub dev: Vec<DevDb>,
3548 /// Databases running inside the pond container stack.
3549 #[serde(default, skip_serializing_if = "Vec::is_empty")]
3550 pub pond: Vec<PondDb>,
3551 /// Remote cloud databases (Turso, Postgres).
3552 #[serde(default, skip_serializing_if = "Vec::is_empty")]
3553 pub cloud: Vec<CloudDb>,
3554}
3555
3556impl DbCatalog {
3557 /// True when no database is declared in any environment. Lets
3558 /// [`ServiceConfig`] skip serializing an empty `[db]` table.
3559 pub fn is_empty(&self) -> bool {
3560 self.dev.is_empty() && self.pond.is_empty() && self.cloud.is_empty()
3561 }
3562}
3563
3564/// A dev-mode local SQLite database (`[[db.dev]]`). `path` is resolved
3565/// relative to the workspace root and opened as a local file — read/write, no
3566/// network, no auth.
3567#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3568#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3569pub struct DevDb {
3570 /// Logical name, unique within the service's `dev` list. Forms the `name`
3571 /// segment of the catalog id `dev:<service>:<name>`.
3572 pub name: String,
3573 /// On-disk SQLite path, relative to the workspace root (or absolute).
3574 pub path: String,
3575}
3576
3577/// A database running inside the pond container stack (`[[db.pond]]`). The
3578/// pond publishes the DB on a localhost TCP port; the hub connects to
3579/// `127.0.0.1:<port>` when the pond is up and returns a clear error when it is
3580/// not. Either `port` (defaulting to a libSQL/`sqld` HTTP endpoint) or a full
3581/// `url` must be given.
3582#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3583#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3584pub struct PondDb {
3585 /// Logical name, unique within the service's `pond` list.
3586 pub name: String,
3587 /// Localhost TCP port the pond publishes the DB on. Interpreted per
3588 /// [`kind`](Self::kind). Mutually complete with `url` (provide one).
3589 #[serde(default, skip_serializing_if = "Option::is_none")]
3590 pub port: Option<u16>,
3591 /// Full connection URL, overriding `port` when set (e.g. a non-localhost
3592 /// host or an explicit scheme).
3593 #[serde(default, skip_serializing_if = "Option::is_none")]
3594 pub url: Option<String>,
3595 /// Wire protocol the pond DB speaks. Selects how a bare `port` becomes a
3596 /// URL: `turso` → `http://127.0.0.1:<port>` (libSQL/`sqld` over Hrana),
3597 /// `postgres` → `postgres://127.0.0.1:<port>`.
3598 #[serde(default)]
3599 pub kind: PondDbKind,
3600}
3601
3602/// Wire protocol of a [`PondDb`].
3603#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq, Eq)]
3604#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3605#[serde(rename_all = "kebab-case")]
3606pub enum PondDbKind {
3607 /// libSQL / `sqld` over Hrana HTTP — the default.
3608 #[default]
3609 Turso,
3610 /// PostgreSQL wire protocol.
3611 Postgres,
3612}
3613
3614/// A remote cloud database (`[[db.cloud]]`). The connection `url` is stored in
3615/// TOML but the credential never is — `auth_token_env` names an environment
3616/// variable the daemon reads at connect time, so the same declaration works
3617/// whether the token is provisioned service-locally or camp-shared (W241;
3618/// operator confirmed both scopes are needed). A camp-wide cloud DB not owned
3619/// by any single service is declared identically in `.yah/db/cloud.toml`.
3620#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3621#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3622pub struct CloudDb {
3623 /// Logical name, unique within its `cloud` list.
3624 pub name: String,
3625 /// Connection URL: `libsql://…` / `http(s)://…` (Turso, `sqld`) or
3626 /// `postgres://…`.
3627 pub url: String,
3628 /// Name of the environment variable holding the auth token. Resolved in
3629 /// the daemon at connect time (value never stored on disk). For a libSQL
3630 /// URL the token is threaded as `?auth_token=…`.
3631 #[serde(default, skip_serializing_if = "Option::is_none")]
3632 pub auth_token_env: Option<String>,
3633}
3634
3635/// A camp-shared cloud database catalog, parsed from `.yah/db/cloud.toml`.
3636/// These are cloud DBs not owned by any single service — declared once at camp
3637/// scope and addressed as `cloud:<name>` (two-segment id), distinct from a
3638/// service-local `cloud:<service>:<name>`.
3639#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq, Eq)]
3640#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3641pub struct CampCloudDbs {
3642 #[serde(default, rename = "cloud", skip_serializing_if = "Vec::is_empty")]
3643 pub cloud: Vec<CloudDb>,
3644}
3645
3646impl CampCloudDbs {
3647 /// Load `<camp_root>/.yah/db/cloud.toml`, or an empty catalog if the file
3648 /// is absent (the common case — most camps declare no shared cloud DBs).
3649 pub fn load(camp_root: &Path) -> Result<Self> {
3650 let path = camp_root.join(".yah/db/cloud.toml");
3651 if !path.exists() {
3652 return Ok(Self::default());
3653 }
3654 let src = std::fs::read_to_string(&path)
3655 .with_context(|| format!("reading {}", path.display()))?;
3656 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
3657 }
3658}
3659
3660/// Topological shape of a mirror — how its providers sit relative to each other.
3661#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
3662#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3663#[serde(rename_all = "kebab-case")]
3664pub enum MirrorShape {
3665 /// Single machine hosts compute (and any non-Cloudflare-fronted static).
3666 SingleMachine,
3667 /// Operator-local dev mirror — static via built-in file server, compute
3668 /// via the local container runtime.
3669 Local,
3670 /// Multi-machine deployment (machines listed per provider slot).
3671 MultiMachine,
3672}
3673
3674/// Which public-ingress provider fronts this mirror's compute (W267, R594-F11).
3675///
3676/// Both arms answer exactly one question — *given these local workload ports,
3677/// make them publicly reachable at these hostnames* — and they differ only in
3678/// where the ingress rules live and who supervises the front door:
3679///
3680/// | | [`CloudflareTunnel`](Self::CloudflareTunnel) | [`Passway`](Self::Passway) |
3681/// |---|---|---|
3682/// | Ingress rules live | Cloudflare's API (token-form tunnels are remotely-managed) | the pingora `Backends` set in the proxy process |
3683/// | How they get there | an API call per deployed workload | passway polls `GET /service-records?ready=true` |
3684/// | Front door lifecycle | a kamaji-supervised `cloudflared` appliance | a kamaji-supervised passway appliance |
3685///
3686/// Flipping this field is the whole tier ladder: rented edge → sovereign edge
3687/// is a one-line mirror edit, not a rewrite. The provider owns **addressing**
3688/// and never **rendering** — the W173 render cube stays in mesofact's manifest.
3689#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
3690#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3691#[serde(rename_all = "kebab-case")]
3692pub enum IngressProvider {
3693 /// No public front door for this mirror. The default: a mirror that
3694 /// publishes to R2 behind a Worker, or a mesh-only compute tier, has no
3695 /// ingress provider to reconcile.
3696 #[default]
3697 None,
3698 /// Rented edge — `cloudflared` dials *out* from the node to Cloudflare's
3699 /// edge. Zero inbound ports, no TLS to manage on the box, hostname rules
3700 /// held in Cloudflare's API.
3701 CloudflareTunnel,
3702 /// Sovereign edge — passway terminates TLS on the node and load-balances
3703 /// an upstream set discovered from yubaba's service records.
3704 Passway,
3705}
3706
3707impl IngressProvider {
3708 /// `true` when this mirror declares a front door that has to be reconciled.
3709 pub fn is_declared(self) -> bool {
3710 !matches!(self, Self::None)
3711 }
3712
3713 /// Kebab-case wire name, as it appears in `mirrors/<env>.toml`.
3714 pub fn as_str(self) -> &'static str {
3715 match self {
3716 Self::None => "none",
3717 Self::CloudflareTunnel => "cloudflare-tunnel",
3718 Self::Passway => "passway",
3719 }
3720 }
3721}
3722
3723/// One declared **edge**: a front door, the slots it fronts, and the nodes it
3724/// is placed on (W305 F2).
3725///
3726/// A mirror declares a *list* of these, which is what lets one service mix
3727/// front doors — cloudflare for the public web tier, passway for an internal or
3728/// high-throughput one. Before this, [`MirrorConfig::ingress`] was a single
3729/// [`IngressProvider`], so a mirror could **swap** front doors but never mix
3730/// them.
3731///
3732/// ```toml
3733/// [[ingress]]
3734/// provider = "passway"
3735/// machines = ["us-east-001", "us-south-001"]
3736/// slots = ["bundle"]
3737///
3738/// [[ingress]]
3739/// provider = "cloudflare-tunnel"
3740/// hostnames = ["issues.yah.dev"]
3741/// ```
3742///
3743/// **The per-node appliance is derived from this, never declared beside it.**
3744/// An edge does invoke a cloudflared or passway process on a box, but that is a
3745/// *consequence* of the service's declaration:
3746/// [`collate_front_doors`](crate::reconciler::collate_front_doors) walks every
3747/// service and derives what each node must run. Declaring it node-side too is
3748/// what produces two sources of truth for one fact.
3749#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
3750#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3751pub struct IngressEdge {
3752 /// Which front door this edge is. [`IngressProvider::None`] is rejected at
3753 /// plan time — an edge that fronts with nothing is always a typo, never an
3754 /// intent (write no edge instead).
3755 pub provider: IngressProvider,
3756 /// Nodes this front door is placed on — **independent of where the fronted
3757 /// workload runs** (R330-F37).
3758 ///
3759 /// Empty falls back to the fronted slot's own `machine` / `machines`, which
3760 /// is the co-located shape every mirror had before front-door placement was
3761 /// expressible. Listing several is what lets the ingress tier and the
3762 /// service tier scale independently: **N front doors over ONE deployment**,
3763 /// one rendered copy, so no cache coherence to settle.
3764 #[serde(default, skip_serializing_if = "Vec::is_empty")]
3765 pub machines: Vec<String>,
3766 /// Provider slot roles this edge fronts (`"bundle"`, `"compute"`, …).
3767 ///
3768 /// One of the two selectors. With a single edge both may be empty, meaning
3769 /// "every fronted slot" — the legacy shape. With **several** edges a
3770 /// selector is mandatory on each, and the partition must be total and
3771 /// disjoint: a slot claimed by no edge, or by two, is an error naming it.
3772 /// An implicit catch-all across mixed front doors would silently publish a
3773 /// service through the wrong one.
3774 #[serde(default, skip_serializing_if = "Vec::is_empty")]
3775 pub slots: Vec<String>,
3776 /// Public hostnames this edge fronts — the other selector, for partitioning
3777 /// by what the world dials rather than by which slot serves it.
3778 #[serde(default, skip_serializing_if = "Vec::is_empty")]
3779 pub hostnames: Vec<String>,
3780 /// Cloudflare Tunnel id this edge publishes through, overriding the
3781 /// fronting machine's [`MachineConfig::cloudflared`].
3782 ///
3783 /// This is W267 Gap 3's real fix, and it is the *service* side of it: a node
3784 /// can join two cohorts' orange networks, and since §Granularity argues the
3785 /// tunnel credential **is** the isolation boundary, which cohort a given
3786 /// service fronts through is a property of the service, not of the box.
3787 /// `MachineConfig.cloudflared` stays as the per-node default (one tunnel is
3788 /// the common case, and the credential does live on the node), but it is no
3789 /// longer the only way to say it — so the node never has to enumerate
3790 /// cohorts.
3791 #[serde(default, skip_serializing_if = "Option::is_none")]
3792 pub tunnel_id: Option<String>,
3793 /// Infra provider id whose credentials this edge's front door authenticates
3794 /// with — `use = "cloudflare"`, resolved through
3795 /// `.yah/infra/providers/<id>.toml` exactly as a slot's `use` is.
3796 ///
3797 /// Same split as [`tunnel_id`](Self::tunnel_id), one field over: whose
3798 /// Cloudflare account holds the tunnel is a property of the **front door**,
3799 /// not of the box that runs the compute. Without this the account was read
3800 /// off the fronted slot's own `use`, which conflates two unrelated facts —
3801 /// and is unwritable for a slot whose compute provider is `kind = "static"`
3802 /// (a borrowed bare box: placement only, no credentials). Such a mirror had
3803 /// no way to name a Cloudflare account at all, short of writing
3804 /// `use = "cloudflare"` on the compute slot and lying about what runs it
3805 /// (R845).
3806 ///
3807 /// `None` falls back to the fronted slot's `use`, which is what every
3808 /// mirror written before this field meant.
3809 #[serde(default, rename = "use", skip_serializing_if = "Option::is_none")]
3810 pub provider_id: Option<String>,
3811 /// Digest-pinned image reference for this edge's front-door appliance,
3812 /// e.g. `localhost/passway:tag@sha256:<hex>` (R870-F16).
3813 ///
3814 /// `None` is the state of every mirror on disk today: the passway arm of
3815 /// `yah cloud apply` cannot deploy an appliance the mirror doesn't name an
3816 /// image for, so it renders the manual `yah cloud ingress deploy …
3817 /// --image <passway-ref>` step instead of running it. Declaring this field
3818 /// is what makes the arm self-sufficient, matching the CloudflareTunnel
3819 /// arm's real-API-call shape rather than only printing for an operator to
3820 /// copy by hand.
3821 #[serde(default, skip_serializing_if = "Option::is_none")]
3822 pub image: Option<String>,
3823}
3824
3825impl IngressEdge {
3826 /// An edge with no selector — fronts every fronted slot, legal only when it
3827 /// is the mirror's only edge.
3828 pub fn all_slots(provider: IngressProvider, machines: Vec<String>) -> Self {
3829 Self {
3830 provider,
3831 machines,
3832 slots: Vec::new(),
3833 hostnames: Vec::new(),
3834 tunnel_id: None,
3835 provider_id: None,
3836 image: None,
3837 }
3838 }
3839
3840 /// `true` when this edge names which slots/hostnames it fronts.
3841 pub fn has_selector(&self) -> bool {
3842 !self.slots.is_empty() || !self.hostnames.is_empty()
3843 }
3844
3845 /// Does this edge claim the rule derived from `slot` publishing `hostname`?
3846 ///
3847 /// A selectorless edge claims everything; that is checked to be
3848 /// unambiguous (one edge only) before this is consulted.
3849 pub fn claims(&self, slot: &str, hostname: &str) -> bool {
3850 if !self.has_selector() {
3851 return true;
3852 }
3853 self.slots.iter().any(|s| s == slot) || self.hostnames.iter().any(|h| h == hostname)
3854 }
3855
3856 /// Human-readable identity for an error message — the provider plus
3857 /// whichever selector was written.
3858 pub fn label(&self) -> String {
3859 let sel = match (self.slots.is_empty(), self.hostnames.is_empty()) {
3860 (true, true) => "no selector".to_string(),
3861 (false, true) => format!("slots = {:?}", self.slots),
3862 (true, false) => format!("hostnames = {:?}", self.hostnames),
3863 (false, false) => format!("slots = {:?} + hostnames = {:?}", self.slots, self.hostnames),
3864 };
3865 format!("[[ingress]] provider = {:?} ({sel})", self.provider.as_str())
3866 }
3867}
3868
3869/// A mirror's `ingress` declaration, in either spelling.
3870///
3871/// The list is the general form; the bare provider is shorthand for the single
3872/// edge fronting everything, and is kept rather than migrated because it is the
3873/// honest spelling for the common case — one service, one front door. Both
3874/// normalize to the same `Vec<IngressEdge>` through
3875/// [`MirrorConfig::ingress_edges`], so nothing downstream branches on which was
3876/// written.
3877#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
3878#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3879#[serde(untagged)]
3880pub enum IngressDecl {
3881 /// `ingress = "passway"` — one edge fronting every fronted slot, placed by
3882 /// the sibling [`MirrorConfig::ingress_machines`].
3883 Provider(IngressProvider),
3884 /// `[[ingress]]` — one entry per declared edge.
3885 Edges(Vec<IngressEdge>),
3886}
3887
3888/// Hand-written because `#[serde(untagged)]` throws the real error away.
3889///
3890/// A derived untagged `Deserialize` tries each variant and, on failure, reports
3891/// only `data did not match any variant of untagged enum IngressDecl` — so a
3892/// misspelled `provider = "passwya"` says nothing about providers, nothing about
3893/// the legal values, and points at the `[[ingress]]` header rather than the
3894/// field. Dispatching on the input shape first means each arm's own error
3895/// survives: a bad string names the legal provider vocabulary, a bad edge table
3896/// names the offending field.
3897impl<'de> Deserialize<'de> for IngressDecl {
3898 fn deserialize<D: serde::Deserializer<'de>>(d: D) -> std::result::Result<Self, D::Error> {
3899 struct DeclVisitor;
3900
3901 impl<'de> serde::de::Visitor<'de> for DeclVisitor {
3902 type Value = IngressDecl;
3903
3904 fn expecting(&self, f: &mut std::fmt::Formatter) -> std::fmt::Result {
3905 f.write_str(
3906 "a provider name (`ingress = \"passway\"`) or a list of edge tables \
3907 (`[[ingress]]`)",
3908 )
3909 }
3910
3911 fn visit_str<E: serde::de::Error>(self, v: &str) -> std::result::Result<Self::Value, E> {
3912 IngressProvider::deserialize(serde::de::value::StrDeserializer::new(v))
3913 .map(IngressDecl::Provider)
3914 }
3915
3916 fn visit_seq<A: serde::de::SeqAccess<'de>>(
3917 self,
3918 seq: A,
3919 ) -> std::result::Result<Self::Value, A::Error> {
3920 Vec::<IngressEdge>::deserialize(serde::de::value::SeqAccessDeserializer::new(seq))
3921 .map(IngressDecl::Edges)
3922 }
3923 }
3924
3925 d.deserialize_any(DeclVisitor)
3926 }
3927}
3928
3929/// No front door — the shape of every mirror that publishes to R2 behind a
3930/// Worker, or runs a mesh-only compute tier.
3931impl Default for IngressDecl {
3932 fn default() -> Self {
3933 Self::Provider(IngressProvider::None)
3934 }
3935}
3936
3937impl IngressDecl {
3938 /// `true` when this mirror declares no front door at all.
3939 pub fn is_absent(&self) -> bool {
3940 match self {
3941 Self::Provider(p) => !p.is_declared(),
3942 Self::Edges(e) => e.is_empty(),
3943 }
3944 }
3945}
3946
3947impl From<IngressProvider> for IngressDecl {
3948 fn from(p: IngressProvider) -> Self {
3949 Self::Provider(p)
3950 }
3951}
3952
3953/// A service mirror — the projection of a [`ServiceConfig`] onto concrete
3954/// infra. Lives at `.yah/services/<svc>/mirrors/<env>.toml`.
3955#[derive(Debug, Clone, Serialize, Deserialize)]
3956#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3957pub struct MirrorConfig {
3958 pub schema_version: u32,
3959 pub shape: MirrorShape,
3960 /// Public-ingress edges fronting this mirror (W267, W305 F2). Defaults to
3961 /// none.
3962 ///
3963 /// Two spellings, one meaning — see [`IngressDecl`]. `ingress = "passway"`
3964 /// is one edge fronting everything; `[[ingress]]` entries declare several,
3965 /// each naming its provider plus the slots or hostnames it fronts. Read it
3966 /// through [`ingress_edges`](Self::ingress_edges), never by matching on the
3967 /// enum, so the two spellings cannot drift apart.
3968 ///
3969 /// Declared at mirror scope rather than per provider slot because a front
3970 /// door does **fan-in**: one `cloudflared` (or one passway) on a node
3971 /// multiplexes every hostname→port rule it fronts, so pinning one to a
3972 /// single slot would mint one edge connection per slot for no gain. An
3973 /// edge's `slots` selector is the general form of that — it groups slots
3974 /// behind one front door, it does not split a front door per slot.
3975 #[serde(default, skip_serializing_if = "IngressDecl::is_absent")]
3976 pub ingress: IngressDecl,
3977 /// Machines the front door is placed on — **independent of where the
3978 /// fronted workload runs** (R330-F37).
3979 ///
3980 /// The single-edge spelling of [`IngressEdge::machines`]: it applies to the
3981 /// one edge `ingress = "<provider>"` declares, and combining it with
3982 /// `[[ingress]]` entries is an error rather than a silent precedence rule.
3983 ///
3984 /// Empty (the default) keeps the pre-existing behaviour: the front door is
3985 /// co-located with the fronted slot's own `machine` / `machines`. That was
3986 /// never a design choice, it was an artifact of bundles binding
3987 /// `127.0.0.1` — nothing off-node could reach a workload, so a proxy had to
3988 /// sit on top of it. R599-F12 landed mesh binding, which removes the
3989 /// constraint: passway is a reverse proxy, and a valid front door needs a
3990 /// cert and an upstream it can *reach*, not a local copy of the service.
3991 ///
3992 /// Listing several machines is what lets the ingress tier and the service
3993 /// tier scale independently — **N front doors over ONE deployment**. There
3994 /// is still exactly one rendered copy of the site, so fanning the front door
3995 /// out introduces no cache-coherence problem; that only appears if you
3996 /// deploy the *workload* to every node instead.
3997 ///
3998 /// ```toml
3999 /// ingress = "passway"
4000 /// ingress_machines = ["us-east-001", "us-west-001"]
4001 /// ```
4002 ///
4003 /// Declaring this without [`ingress`](Self::ingress) is an error, not a
4004 /// no-op — it always means the operator expected a front door somewhere.
4005 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4006 pub ingress_machines: Vec<String>,
4007 /// Provider slots, keyed by role (`"static"`, `"compute"`, …). Each value
4008 /// either references a provider declared under `.yah/infra/providers/` or
4009 /// inlines a local-only provider (no creds, no infra file).
4010 ///
4011 /// A role is normally service-wide — one slot serves every component that
4012 /// shares it — but [`ReconcileCtx::slot`](crate::reconciler::ReconcileCtx::slot)
4013 /// looks up the component-qualified key `"<role>:<component id>"` first.
4014 /// A service with two components of the same role (e.g. two
4015 /// `mesofact-static` components under one mirror) declares
4016 /// `providers."static:<id>"` per component to give each its own port;
4017 /// omitting the qualifier keeps the pre-existing single-slot behavior.
4018 #[serde(default)]
4019 pub providers: BTreeMap<String, MirrorProviderSlot>,
4020 /// Capability→driver bindings, keyed by **capability** (`"pg"`, `"s3"`, …)
4021 /// rather than by slot role (W265 §Drivers).
4022 ///
4023 /// This is the generalization of [`Self::providers`]: `providers.static` /
4024 /// `providers.object_store` are the special case where the slot name and
4025 /// the capability happen to coincide, and keying by capability is what stops
4026 /// the slot enum growing one arm per tier-specific implementation. A service
4027 /// says "I need pg"; the mirror says which implementation of pg *this tier*
4028 /// uses; the app talks the same wire protocol either way and never forks.
4029 ///
4030 /// ```toml
4031 /// [drivers.pg]
4032 /// kind = "local-pg-dev" # dev — kamaji-supervised loopback postgres
4033 /// ```
4034 ///
4035 /// Additive in P1: `drivers` lands *alongside* `providers`, and migrating
4036 /// the existing `providers.static` / `providers.object_store` declarations
4037 /// over is a separate pass (W265 §"Open follow-ups"). A mirror that declares
4038 /// neither is unchanged.
4039 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
4040 pub drivers: BTreeMap<String, MirrorProviderSlot>,
4041 /// Per-environment alias overrides for `kind = "static-asset"` components.
4042 ///
4043 /// Keys are logical names (e.g. `"whisper-default"`); values must be
4044 /// filenames present in the component's `workload.toml` catalog.
4045 /// **Resolution only** — this table may never introduce a filename absent
4046 /// from the catalog. Validated against the workload catalog at sync time.
4047 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
4048 pub asset_aliases: BTreeMap<String, String>,
4049}
4050
4051impl MirrorConfig {
4052 /// This mirror's declared edges, with both spellings normalized (W305 F2).
4053 ///
4054 /// The single place `ingress` + `ingress_machines` are reconciled, so no
4055 /// consumer has to know which spelling was written. Returns an empty vec
4056 /// when the mirror declares no front door.
4057 ///
4058 /// Errors are the declarations that cannot mean anything:
4059 ///
4060 /// - `ingress_machines` with no `ingress` — front-door placement with no
4061 /// front door to place, always a typo (R330-F37);
4062 /// - `ingress_machines` alongside `[[ingress]]` — placement declared twice,
4063 /// in a form where one silently wins;
4064 /// - `provider = "none"` on an edge — an edge that fronts with nothing.
4065 /// The `[[ingress]]` entries exactly as written, without normalizing the
4066 /// scalar spelling or validating anything.
4067 ///
4068 /// [`ingress_edges`](Self::ingress_edges) is the one to reach for; this
4069 /// exists for the checks that must run *before* a mirror is known to be
4070 /// well-formed — cross-reference validation walks every mirror in the
4071 /// workspace, and hard-failing there on an unrelated mirror's shape error
4072 /// would report the wrong file. Empty for the scalar spelling, which has no
4073 /// edge table to carry per-edge fields.
4074 pub fn ingress_edge_slice(&self) -> &[IngressEdge] {
4075 match &self.ingress {
4076 IngressDecl::Edges(edges) => edges,
4077 IngressDecl::Provider(_) => &[],
4078 }
4079 }
4080
4081 pub fn ingress_edges(&self) -> Result<Vec<IngressEdge>> {
4082 match &self.ingress {
4083 IngressDecl::Provider(p) if !p.is_declared() => {
4084 if !self.ingress_machines.is_empty() {
4085 bail!(
4086 "mirror declares `ingress_machines = {:?}` but no `ingress` provider — \
4087 front-door placement with no front door to place. Add \
4088 `ingress = \"passway\"` (or \"cloudflare-tunnel\"), or drop \
4089 `ingress_machines`.",
4090 self.ingress_machines
4091 );
4092 }
4093 Ok(Vec::new())
4094 }
4095 IngressDecl::Provider(p) => Ok(vec![IngressEdge::all_slots(
4096 *p,
4097 self.ingress_machines.clone(),
4098 )]),
4099 IngressDecl::Edges(edges) => {
4100 if !self.ingress_machines.is_empty() {
4101 bail!(
4102 "mirror declares both `[[ingress]]` edges and the single-edge \
4103 `ingress_machines = {:?}` — front-door placement stated twice. Move \
4104 those names onto the edge they place: `machines = [...]` inside the \
4105 `[[ingress]]` entry.",
4106 self.ingress_machines
4107 );
4108 }
4109 for edge in edges {
4110 if !edge.provider.is_declared() {
4111 bail!(
4112 "{}: `provider = \"none\"` fronts nothing. An edge exists to name a \
4113 front door — delete the entry instead.",
4114 edge.label()
4115 );
4116 }
4117 }
4118 Ok(edges.clone())
4119 }
4120 }
4121 }
4122
4123 /// Nodes this mirror's **passway** front doors are placed on, in
4124 /// declaration order and de-duplicated — or `None` when the mirror declares
4125 /// no passway edge at all.
4126 ///
4127 /// `Some(vec![])` is a real and different answer from `None`: a passway edge
4128 /// is declared but names no machine, so its placement falls back to the
4129 /// fronted slot's own. That fallback is placement *resolution* — it belongs
4130 /// to [`IngressRule::machines`](crate::reconciler::IngressRule::machines)
4131 /// and the plan it is built from, not to a mirror read in isolation — so it
4132 /// is reported as "declared, placement unknown from here" rather than
4133 /// half-derived. A caller that needs a node to dial has to say so.
4134 ///
4135 /// Passway-only because the caller is tenant DNS onboarding: only a passway
4136 /// node serves yubaba's `GET /domains/{domain}/onboarding`. A
4137 /// cloudflare-tunnel edge publishes through Cloudflare's own DNS and has no
4138 /// such record to hand a tenant, so folding its machines in would point the
4139 /// UI at a node that cannot answer.
4140 ///
4141 /// Read through [`ingress_edges`](Self::ingress_edges), so both spellings
4142 /// are covered by construction. A declaration that cannot mean anything
4143 /// (`ingress_machines` with no `ingress`, or both spellings at once) reads
4144 /// as `None` rather than propagating an error: those are reported by
4145 /// cross-reference validation, which can name the offending file.
4146 pub fn passway_machines(&self) -> Option<Vec<String>> {
4147 let edges = self.ingress_edges().ok()?;
4148 let mut declared = false;
4149 let mut machines: Vec<String> = Vec::new();
4150 for edge in edges
4151 .iter()
4152 .filter(|e| matches!(e.provider, IngressProvider::Passway))
4153 {
4154 declared = true;
4155 for m in &edge.machines {
4156 if !machines.iter().any(|seen| seen == m) {
4157 machines.push(m.clone());
4158 }
4159 }
4160 }
4161 declared.then_some(machines)
4162 }
4163
4164 /// Parse a single `mirrors/<env>.toml` file.
4165 pub fn load(path: &Path) -> Result<Self> {
4166 let src =
4167 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
4168 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
4169 }
4170
4171 /// Persist to `.yah/services/<service>/mirrors/<env>.toml`, creating the
4172 /// `mirrors/` directory if needed. Create-or-overwrite. The mirror file is
4173 /// named by `env` (its stem); `service` selects the owning service dir.
4174 pub fn save(&self, workspace_root: &Path, service: &str, env: &str) -> Result<()> {
4175 let dir = crate::paths::service_mirrors_dir(workspace_root, service);
4176 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
4177 let path = crate::paths::service_mirror_toml(workspace_root, service, env);
4178 let s = toml::to_string_pretty(self)
4179 .with_context(|| format!("serializing mirror {service}/{env}"))?;
4180 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
4181 }
4182
4183 /// Remove `.yah/services/<service>/mirrors/<env>.toml`. Returns `false`
4184 /// when the file was already absent. Leaves the service and its other
4185 /// mirrors untouched.
4186 ///
4187 /// Also checks legacy stems (e.g. `local-sim` when `env = "pond"`) so
4188 /// deleting a canonical tier name removes whichever file exists on disk.
4189 pub fn delete(workspace_root: &Path, service: &str, env: &str) -> Result<bool> {
4190 let path = crate::paths::service_mirror_toml(workspace_root, service, env);
4191 if path.exists() {
4192 std::fs::remove_file(&path).with_context(|| format!("removing {}", path.display()))?;
4193 return Ok(true);
4194 }
4195 // Try legacy file stems for canonical tier names.
4196 let legacy: &[&str] = match env {
4197 "dev" => &["local"],
4198 "pond" => &["local-sim", "sim"],
4199 "cloud" => &["prod"],
4200 _ => &[],
4201 };
4202 for stem in legacy {
4203 let alt = crate::paths::service_mirror_toml(workspace_root, service, stem);
4204 if alt.exists() {
4205 std::fs::remove_file(&alt)
4206 .with_context(|| format!("removing {}", alt.display()))?;
4207 return Ok(true);
4208 }
4209 }
4210 Ok(false)
4211 }
4212}
4213
4214/// A provider slot inside a [`MirrorConfig`]. Two shapes:
4215/// - **Reference** (`use = "<provider-id>"`) — point at an infra-declared
4216/// provider; extra fields are slot-specific (bucket, zone, dns, …).
4217/// - **Inline** (`kind = "local-*"`) — for providers that need no infra
4218/// declaration because they carry no credentials.
4219#[derive(Debug, Clone, Serialize, Deserialize)]
4220#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4221#[serde(untagged)]
4222pub enum MirrorProviderSlot {
4223 Reference {
4224 #[serde(rename = "use")]
4225 provider_id: String,
4226 #[serde(flatten)]
4227 #[cfg_attr(
4228 feature = "json-schema",
4229 schemars(with = "std::collections::BTreeMap<String, serde_json::Value>")
4230 )]
4231 fields: BTreeMap<String, toml::Value>,
4232 },
4233 Inline {
4234 kind: Provider,
4235 #[serde(flatten)]
4236 #[cfg_attr(
4237 feature = "json-schema",
4238 schemars(with = "std::collections::BTreeMap<String, serde_json::Value>")
4239 )]
4240 fields: BTreeMap<String, toml::Value>,
4241 },
4242}
4243
4244impl MirrorProviderSlot {
4245 /// Provider id this slot references, or `None` for inline slots.
4246 pub fn provider_id(&self) -> Option<&str> {
4247 match self {
4248 Self::Reference { provider_id, .. } => Some(provider_id),
4249 Self::Inline { .. } => None,
4250 }
4251 }
4252
4253 /// Provider kind for inline slots, or `None` for reference slots
4254 /// (resolve via the referenced [`ProviderConfig`]).
4255 pub fn inline_kind(&self) -> Option<Provider> {
4256 match self {
4257 Self::Reference { .. } => None,
4258 Self::Inline { kind, .. } => Some(*kind),
4259 }
4260 }
4261
4262 pub fn fields(&self) -> &BTreeMap<String, toml::Value> {
4263 match self {
4264 Self::Reference { fields, .. } | Self::Inline { fields, .. } => fields,
4265 }
4266 }
4267
4268 /// F16 placement: parse the optional `required = { … }` sub-table on this
4269 /// slot. Returns `None` when absent or unparseable (callers treat as no
4270 /// constraint). See [`RequiredSpec`] for the field grammar.
4271 pub fn required(&self) -> Option<RequiredSpec> {
4272 let v = self.fields().get("required")?.clone();
4273 v.try_into().ok()
4274 }
4275}
4276
4277/// F16 placement constraints declared on a [`MirrorProviderSlot`], lives under
4278/// `[providers.<role>] required = { regions = [...], mesh_tags = [...] }` in
4279/// `mirrors/<env>.toml`.
4280///
4281/// Hard (must-satisfy) axes, all AND-ed together:
4282/// - `regions` / `zones` / `providers` — *membership*: the machine's
4283/// `region` / `zone` / `provider` must be one of the listed values.
4284/// - `mesh_tags` — *superset*: the machine's `mesh_tags` must contain every
4285/// listed tag.
4286/// - `memory_mb` / `cpu_millis` — *capacity floor* (R572-F5): the machine's
4287/// `allocatable` budget must cover the demand. `0` = no constraint.
4288/// - *taint repulsion* — **unconditional** (R876-B7): the machine must not
4289/// carry any taint that [`taint_effect`] classifies as
4290/// [`TaintEffect::Repels`], unless that exact key is listed in
4291/// [`Self::tolerates`]. This axis is not declared; it applies to every spec.
4292/// - `requires_taint` — *taint affinity* (R572-F5): the machine must carry
4293/// this taint key (in `taints` or `mesh_tags`). `None` = no affinity.
4294///
4295/// These are the **only** readers of [`MachineConfig::taints`], which is
4296/// what makes [`taint_effect`]'s closed vocabulary well-founded.
4297///
4298/// [`MachineConfig::sovereign_group`] is deliberately **not** an axis here and
4299/// must not become one (W305/R742-F1). A sovereign group is a blast radius,
4300/// not a filter: which quorum a box votes in says nothing about whether a
4301/// workload may run on it, and a dev-group node exists precisely so dev-mode
4302/// services — stateful ones included — can be scheduled onto it. Filtering on
4303/// it would re-make the mistake W305 exists to undo, where one mechanism
4304/// silently carried three unrelated properties.
4305///
4306/// An empty / zero / None on every axis means "no constraint on that axis".
4307/// A fully-unconstrained `RequiredSpec` matches every machine (see
4308/// [`RequiredSpec::is_unconstrained`]).
4309#[derive(Debug, Clone, Default, Serialize, Deserialize)]
4310#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4311pub struct RequiredSpec {
4312 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4313 pub regions: Vec<String>,
4314 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4315 pub zones: Vec<String>,
4316 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4317 pub providers: Vec<String>,
4318 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4319 pub mesh_tags: Vec<String>,
4320
4321 /// R833-F8: **imperative** placement — the machine must be one of these by
4322 /// `name`. Empty (the default) = no constraint, which is every pre-R833-F8
4323 /// caller.
4324 ///
4325 /// This is the one axis that is not a *capability* the scheduler infers.
4326 /// The operator typed `--where=node:us-west-003`, so it composes with the
4327 /// other axes exactly like the rest — a named node that fails the capacity
4328 /// floor or carries a repelling taint still does not match, and the refusal
4329 /// names why rather than silently placing the work somewhere else.
4330 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4331 pub nodes: Vec<String>,
4332
4333 /// R572-F5: minimum memory (MiB) the target node must have in its
4334 /// declared `allocatable` budget. `0` = no constraint. Filled by
4335 /// [`CloudConfig::admit_workload`] from the workload's
4336 /// `memory_request_mb()` — its placement **request**, which is not the
4337 /// same number as the `resources.memory_mb` cgroup **ceiling**.
4338 #[serde(default, skip_serializing_if = "is_zero_u32")]
4339 pub memory_mb: u32,
4340 /// R572-F5: minimum CPU (millicores) the target node must have in its
4341 /// declared `allocatable` budget. `0` = no constraint. Filled by
4342 /// [`CloudConfig::admit_workload`] from the workload's `resources.cpu_millis`.
4343 #[serde(default, skip_serializing_if = "is_zero_u32")]
4344 pub cpu_millis: u32,
4345 /// R876-B7: repelling node taints this placement **opts back in to**.
4346 /// Each entry is a machine taint key spelled exactly as it appears in
4347 /// [`MachineConfig::taints`] — `"no-appliance"`, not `"appliance"` — so the
4348 /// node side and the workload side share one vocabulary and nothing has to
4349 /// translate between them.
4350 ///
4351 /// # Why this replaced `repel_archetypes`
4352 ///
4353 /// Repulsion used to be **opt-in-to-be-repelled**: the spec named the
4354 /// archetypes it was, and only a `no-<that archetype>` taint blocked it.
4355 /// That field was `#[serde(skip)]`, so a slot declared as
4356 /// `required = { ... }` in a mirror TOML always deserialized with it empty
4357 /// and [`Self::matches`] never read [`MachineConfig::taints`] at all. Node
4358 /// taints were therefore structurally inert for every mirror-declared
4359 /// placement, and inert *silently* — `no-server` is a legal key, so
4360 /// `yah cloud validate` passed and an operator draining a node before
4361 /// maintenance got a green run and a workload that never moved (R876-S2's
4362 /// drill measured exactly this against the real tree).
4363 ///
4364 /// The sense is now inverted, which is the only shape that can survive a
4365 /// field the wire does not carry: **repulsion is unconditional and
4366 /// toleration is declared.** A spec that says nothing is repelled by every
4367 /// repelling taint — the reading an operator writing `taints = ["no-server"]`
4368 /// on a machine already assumed they were getting.
4369 ///
4370 /// Toleration is per-key and absolute; there is no wildcard. Listing a key
4371 /// no machine declares is harmless and matches nothing.
4372 ///
4373 /// [`admission_spec`] fills this from the placement group's archetypes —
4374 /// every repelling key that is *not* the group's own class — which is what
4375 /// makes the `admit_workload` path behave identically across this change
4376 /// (R860-T4 / W338 §Placement consequences 2 still hold: the group's
4377 /// archetypes are the union over `local` requirement edges, so a `Server`
4378 /// bound to an `Appliance` tolerates neither `no-server` nor
4379 /// `no-appliance`).
4380 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4381 pub tolerates: Vec<String>,
4382 /// R572-F5: taint the workload requires the target node to carry
4383 /// (annotation `yah.placement.requires-taint`). The node must have the
4384 /// key in its `taints` list or `mesh_tags`. `None` = no affinity constraint.
4385 #[serde(skip)]
4386 pub requires_taint: Option<String>,
4387
4388 /// R844-F8: **how many** machines this constraint places onto. `None` — the
4389 /// only shape on disk before this field — means one, so every mirror in the
4390 /// tree resolves byte-identically across the change.
4391 ///
4392 /// This is not a match axis: it never appears in [`Self::matches`] and never
4393 /// changes whether a given machine qualifies. It is the *cardinality* of the
4394 /// answer, which is why it lives here rather than as another filter — the
4395 /// operator declares what is required and how many of it, and the scheduler
4396 /// picks which.
4397 ///
4398 /// **Declared, never inferred.** The count is emphatically not "how many
4399 /// machines happen to match": deriving it that way would make adding a box
4400 /// to the fleet silently scale a production front door. A constraint that
4401 /// matches four machines and asks for two places on two.
4402 ///
4403 /// **Fewer matches than asked is an error** ([`select_matching`]), not a
4404 /// partial placement. Placing one of two and reporting success is the
4405 /// subset-that-looks-like-it-worked failure R844 exists to close.
4406 ///
4407 /// Deliberately absent from [`Self::is_unconstrained`], which answers "does
4408 /// every machine match" — a question about the predicate, not the count. A
4409 /// `required = { replicas = 2 }` with no axis is therefore still
4410 /// unconstrained, and the deploy side still refuses it as an
4411 /// underspecified placement.
4412 #[serde(default, skip_serializing_if = "Option::is_none")]
4413 pub replicas: Option<u32>,
4414}
4415
4416impl RequiredSpec {
4417 /// How many machines this constraint places onto — [`Self::replicas`],
4418 /// resolving the absent case to the pre-R844-F8 answer of one.
4419 ///
4420 /// The single place that default is spelled, so the ingress planner and the
4421 /// deploy resolver cannot disagree about what "no replica count" means.
4422 pub fn replica_count(&self) -> usize {
4423 self.replicas.unwrap_or(1) as usize
4424 }
4425
4426 /// True when no *declared* axis carries a constraint — every untainted
4427 /// machine matches.
4428 ///
4429 /// R876-B7: taint repulsion is deliberately absent from this conjunction,
4430 /// unlike the `repel_archetypes` it replaced. Repulsion is no longer an axis
4431 /// a spec declares — it applies to every spec — so including it would make
4432 /// the answer a property of the fleet rather than of the constraint. Nor
4433 /// does [`Self::tolerates`] belong here: a toleration *widens* the candidate
4434 /// set, and the callers of this predicate ask "did the operator narrow
4435 /// anything" in order to refuse an underspecified placement. A slot that
4436 /// declares only a toleration has still narrowed nothing.
4437 pub fn is_unconstrained(&self) -> bool {
4438 self.regions.is_empty()
4439 && self.zones.is_empty()
4440 && self.providers.is_empty()
4441 && self.mesh_tags.is_empty()
4442 && self.nodes.is_empty()
4443 && self.memory_mb == 0
4444 && self.cpu_millis == 0
4445 && self.requires_taint.is_none()
4446 }
4447
4448 /// Whether `machine` satisfies every hard axis.
4449 ///
4450 /// - Membership axes (region/zone/provider): machine must carry the field
4451 /// and it must appear in the constraint list.
4452 /// - `mesh_tags`: machine tags must be a superset of the required set.
4453 /// - **R572-F5 capacity floor**: `machine.allocatable.{memory,cpu}` must
4454 /// cover `self.{memory,cpu}`. A machine with no `allocatable` block passes
4455 /// unconditionally (capacity unknown → no constraint enforced).
4456 /// - **Taint repulsion (R876-B7)**: machine must not carry *any* taint that
4457 /// [`taint_effect`] classifies as [`TaintEffect::Repels`], unless that key
4458 /// is listed in [`Self::tolerates`]. Applied unconditionally — this is the
4459 /// axis no spec has to declare, and the one that makes a node drainable.
4460 /// - **R572-F5 taint affinity**: if `requires_taint` is set, the machine
4461 /// must carry that key in its `taints` list or `mesh_tags`.
4462 ///
4463 /// A [`TaintEffect::Attracts`] key (today just `public-ip`) does **not**
4464 /// repel: it is the affinity vocabulary, so reading it as repulsion would
4465 /// evict every workload from the three nodes that carry it. Only the
4466 /// `no-<archetype>` class repels, and [`taint_effect`] is the single
4467 /// authority on which is which — which is why
4468 /// [`crate::validate::check_inert_taints`] refuses to let an unclassifiable
4469 /// key be declared: it would read as a constraint and be none.
4470 pub fn matches(&self, machine: &MachineConfig) -> bool {
4471 let member_ok = |constraint: &[String], value: Option<&str>| -> bool {
4472 constraint.is_empty() || value.map_or(false, |v| constraint.iter().any(|c| c == v))
4473 };
4474
4475 // R833-F8: imperative node pin, checked first because it is the axis a
4476 // human asserted rather than one the scheduler derived — a refusal
4477 // should read "us-west-003 does not match" and not lead with a tag set
4478 // the operator never typed.
4479 if !member_ok(&self.nodes, Some(machine.name.as_str())) {
4480 return false;
4481 }
4482
4483 // Membership + mesh-tags (pre-existing axes).
4484 if !member_ok(&self.regions, machine.region.as_deref())
4485 || !member_ok(&self.zones, machine.zone.as_deref())
4486 || !member_ok(&self.providers, Some(machine.provider.as_str()))
4487 || !self
4488 .mesh_tags
4489 .iter()
4490 .all(|t| machine.mesh_tags.iter().any(|mt| mt == t))
4491 {
4492 return false;
4493 }
4494
4495 // R572-F5: capacity floor. Skipped when machine has no allocatable
4496 // declaration (unknown capacity → passes, consistent with pre-F5 behaviour).
4497 if self.memory_mb > 0 || self.cpu_millis > 0 {
4498 if let Some(alloc) = &machine.allocatable {
4499 if self.memory_mb > alloc.memory_mb || self.cpu_millis > alloc.cpu_millis {
4500 return false;
4501 }
4502 }
4503 }
4504
4505 // R876-B7: taint repulsion, repel-by-default. Every repelling taint on
4506 // the machine blocks placement unless this spec names it in
4507 // `tolerates`. Driven off `machine.taints` rather than off a field of
4508 // `self`, which is the whole point: a spec that arrives by deserializing
4509 // a mirror's `required = {...}` carries no repulsion declaration and
4510 // never could, so making repulsion conditional on one made node taints
4511 // structurally unreadable on that path (R876-S2).
4512 for taint in &machine.taints {
4513 if !matches!(taint_effect(taint), TaintEffect::Repels(_)) {
4514 continue;
4515 }
4516 if !self.tolerates.iter().any(|t| t == taint) {
4517 return false;
4518 }
4519 }
4520
4521 // R572-F5: taint affinity. Machine must carry the required taint key
4522 // in either its `taints` list or `mesh_tags`.
4523 if let Some(req) = &self.requires_taint {
4524 let has_it = machine.taints.iter().any(|t| t == req)
4525 || machine.mesh_tags.iter().any(|t| t == req);
4526 if !has_it {
4527 return false;
4528 }
4529 }
4530
4531 true
4532 }
4533
4534 /// Human-readable summary of the constraints, for fail-loud error messages.
4535 /// Example: `required.regions=[us-west] + required.mesh_tags=[tag:cloud-runner]`.
4536 pub fn describe(&self) -> String {
4537 let mut parts = Vec::new();
4538 let mut push = |label: &str, vals: &[String]| {
4539 if !vals.is_empty() {
4540 parts.push(format!("required.{label}=[{}]", vals.join(",")));
4541 }
4542 };
4543 push("nodes", &self.nodes);
4544 push("regions", &self.regions);
4545 push("zones", &self.zones);
4546 push("providers", &self.providers);
4547 push("mesh_tags", &self.mesh_tags);
4548 // Kept with the other list axes, and NOT moved below: `push` borrows
4549 // `parts` mutably for as long as it is live, so interleaving it with the
4550 // direct `parts.push` calls under it does not compile.
4551 push("tolerates", &self.tolerates);
4552 if self.memory_mb > 0 {
4553 parts.push(format!("memory_mb>={}", self.memory_mb));
4554 }
4555 if self.cpu_millis > 0 {
4556 parts.push(format!("cpu_millis>={}", self.cpu_millis));
4557 }
4558 if let Some(req) = &self.requires_taint {
4559 parts.push(format!("requires_taint={req}"));
4560 }
4561 if parts.is_empty() {
4562 "no constraints".to_string()
4563 } else {
4564 parts.join(" + ")
4565 }
4566 }
4567}
4568
4569/// Which front door actually serves a domain's requests (R594-F12).
4570///
4571/// Every domain manifest must say this out loud. Before it existed the
4572/// difference between "R2 serves this hostname directly" and "a Worker
4573/// serves it" was expressed *only* by whether the file happened to carry
4574/// `[[routes]]` — so binding a route-carrying domain straight to R2 was
4575/// accepted silently and served 200s on its SSG half while losing clean
4576/// URLs, SPA shell fallback, deferred-route pointers and branded error
4577/// pages. All of those live in the Worker
4578/// (`oss/mesofact/packages/mesofact-edge/src/router.ts`) or in
4579/// mesofact-serve; an R2 custom domain has none of them.
4580///
4581/// The vocabulary mirrors `scripts/cf-apex-mode.sh` (worker | grey | orange)
4582/// — this moves the choice into the config where it can be checked instead
4583/// of living in one bash script.
4584///
4585/// A front door does **fan-in** only. The render cube (SSG / SPA / SSR /
4586/// deferred / 404) is mesofact's manifest, not this one — see W173 and
4587/// `.yah/docs/working/W267-sovereign-public-ingress.md`
4588/// §"Two front doors, one render contract".
4589#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
4590#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4591#[serde(rename_all = "kebab-case")]
4592pub enum FrontDoor {
4593 /// Cloudflare R2 custom domain. Requests hit R2 objects with edge
4594 /// caching and nothing else — no clean URLs, no SPA fallback, no
4595 /// branded errors. Correct for a pure asset tier (W175's verdict for
4596 /// `cdn.yah.dev`) and wrong for anything that renders pages.
4597 /// Implies zero `[[routes]]` and no `worker_bundle_path`.
4598 BucketDirect,
4599 /// Cloudflare Worker generated from this manifest's route table.
4600 Worker,
4601 /// Sovereign L7 ingress — the `passway` proxy on yah-owned metal
4602 /// (`oss/passway`, W267). Same route table as `worker`; different
4603 /// machine terminates TLS.
4604 Passway,
4605}
4606
4607impl FrontDoor {
4608 /// Whether this front door consumes the manifest's `[[routes]]` table.
4609 /// `bucket-direct` does not; the other two are nothing without it.
4610 pub fn is_route_driven(self) -> bool {
4611 matches!(self, FrontDoor::Worker | FrontDoor::Passway)
4612 }
4613
4614 /// The manifest spelling, for error messages.
4615 pub fn as_str(self) -> &'static str {
4616 match self {
4617 FrontDoor::BucketDirect => "bucket-direct",
4618 FrontDoor::Worker => "worker",
4619 FrontDoor::Passway => "passway",
4620 }
4621 }
4622}
4623
4624/// A routing manifest for one domain, from `.yah/domains/<name>.toml`.
4625///
4626/// The domain manifest is the *only* place that knows about path routing:
4627/// services declare static/backend components by opaque ID, and this
4628/// manifest binds those components to URL paths on a public-facing
4629/// domain. Generated Worker bundles consume this. See
4630/// `.yah/docs/working/W118-yah-domain-tiers.md` (R347).
4631#[derive(Debug, Clone, Serialize, Deserialize)]
4632#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4633pub struct DomainConfig {
4634 pub schema_version: u32,
4635 /// Stable identifier for this domain (file stem of the manifest).
4636 /// Example: `"yah-dev"` for the `yah.dev` zone.
4637 pub name: String,
4638 /// The fully-qualified domain this manifest routes for. Example:
4639 /// `"yah.dev"`, `"app.yah.dev"`.
4640 pub domain: String,
4641 /// Which front door serves this domain (R594-F12). **Required** — a
4642 /// default here would silently re-create the defect the field exists to
4643 /// close. Cross-checked against `routes` / `worker_bundle_path` by
4644 /// [`DomainConfig::validate_front_door`] at load time.
4645 pub front_door: FrontDoor,
4646 /// Public CDN bucket name. Static-mode route components publish into
4647 /// this bucket. Owned by the domain, *not* by any single service.
4648 pub cdn_bucket: String,
4649 /// Optional path (relative to workspace root) where the generated
4650 /// Worker bundle lands. `None` while the bundle generator (R347-F4)
4651 /// is still being wired up.
4652 #[serde(default, skip_serializing_if = "Option::is_none")]
4653 pub worker_bundle_path: Option<String>,
4654 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4655 pub routes: Vec<DomainRoute>,
4656}
4657
4658/// One entry in a [`DomainConfig`]'s route table.
4659///
4660/// The `mode` discriminator picks the variant's body via serde's
4661/// internally-tagged enum representation. Path patterns follow the
4662/// Worker convention: a trailing `*` matches everything underneath.
4663#[derive(Debug, Clone, Serialize, Deserialize)]
4664#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4665pub struct DomainRoute {
4666 /// URL pattern this route matches. Examples: `"/"`, `"/dashboard/*"`,
4667 /// `"/camp/ws"`.
4668 pub path: String,
4669 /// Response headers the front door sets on every response served under
4670 /// this route (R746). Empty by default.
4671 ///
4672 /// This is the manifest's answer to "who decides a path's response
4673 /// headers". Before it existed the answer was *nobody*: a `_headers` file
4674 /// is a Cloudflare Pages / Netlify convention, and neither of this
4675 /// repo's front doors reads one — a Worker returns what it fetched from
4676 /// R2, and R2 serves only the object's own httpMetadata. So a site could
4677 /// carry a `_headers` file declaring COOP/COEP and ship without them,
4678 /// which is exactly how it was found: `SharedArrayBuffer` is simply
4679 /// absent in a document served cross-origin-isolation-free, with no
4680 /// error anywhere to say why.
4681 ///
4682 /// Deliberately a free-form `name -> value` map rather than named fields
4683 /// for the isolation headers: the domain manifest has no business
4684 /// knowing which headers a route's payload happens to need. Ordering
4685 /// follows the route table's own rule — first matching route wins, no
4686 /// merging across routes (see the Worker's `applyRouteHeaders`).
4687 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
4688 pub headers: BTreeMap<String, String>,
4689 #[serde(flatten)]
4690 pub mode: RouteMode,
4691}
4692
4693/// Body of a [`DomainRoute`]. Three modes:
4694/// - **Static** — Worker reads from the domain's CDN bucket. Component
4695/// ref points at a `kind = "mesofact-static"` (or similar) service
4696/// component.
4697/// - **Backend** — Worker proxies to an HTTP origin owned by a backend
4698/// component (yubaba workload, gateway, etc.).
4699/// - **Redirect** — Worker emits a 30x to the target URL. Used to keep
4700/// old paths alive during domain refactors.
4701#[derive(Debug, Clone, Serialize, Deserialize)]
4702#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4703#[serde(tag = "mode", rename_all = "kebab-case")]
4704pub enum RouteMode {
4705 Static {
4706 /// Component reference `"<service>/<component-id>"`. Validated
4707 /// at [`CloudConfig::load`] time.
4708 component: String,
4709 },
4710 Backend {
4711 /// Component reference `"<service>/<component-id>"`. Validated
4712 /// at [`CloudConfig::load`] time.
4713 component: String,
4714 /// Origin URL the Worker `fetch()`es. Schema-permissive — could
4715 /// be `https://...`, `wss://...`, or a yah-internal mesh URL
4716 /// resolved by yubaba.
4717 origin: String,
4718 },
4719 Redirect {
4720 /// Absolute URL or path the Worker emits a 30x to.
4721 target: String,
4722 /// HTTP status code. Defaults to 308 (permanent + method-preserving)
4723 /// so deprecations don't silently turn POSTs into GETs.
4724 #[serde(default = "default_redirect_status")]
4725 status: u16,
4726 },
4727}
4728
4729fn default_redirect_status() -> u16 {
4730 308
4731}
4732
4733/// Normalize a component `mount` to a storage/URL key prefix: strip the
4734/// surrounding slashes. `"/app"`, `"app/"`, `"/app/"` → `"app"`; `"/"`, `""`
4735/// → `""` (the service root).
4736///
4737/// One producer on purpose — the publisher's key prefix, the route-path
4738/// cross-check and the front door's key lookup must all agree on what `/app`
4739/// means down to the byte, and three copies of `trim_matches('/')` is how they
4740/// stop agreeing.
4741pub fn normalize_mount(raw: &str) -> String {
4742 raw.trim_matches('/').to_string()
4743}
4744
4745/// The key prefix a domain route pattern serves under: `"/*"` → `""`,
4746/// `"/app/*"` and `"/app"` → `"app"`. The twin of [`normalize_mount`] on the
4747/// routing side.
4748pub fn route_path_prefix(path: &str) -> String {
4749 normalize_mount(path.strip_suffix('*').unwrap_or(path))
4750}
4751
4752/// The route-driven domain whose route table binds a component of `service`,
4753/// if any. Used by static publishers to pick up the per-route response
4754/// headers a service's paths were declared with.
4755///
4756/// Deterministic by `BTreeMap` key order when more than one domain routes the
4757/// same service (a legitimate shape: an apex and a staging host serving one
4758/// bundle). Returning the first is a real limitation, not a considered
4759/// choice — the day two such domains want *different* headers for one
4760/// component, this needs the domain identity threaded in rather than inferred.
4761pub fn domain_serving_service<'a>(
4762 domains: &'a BTreeMap<String, DomainConfig>,
4763 service: &str,
4764) -> Option<&'a DomainConfig> {
4765 domains
4766 .values()
4767 .find(|d| d.front_door.is_route_driven() && d.serves_service(service))
4768}
4769
4770/// The `ROUTE_HEADERS` Worker-binding value for `service`, read from the
4771/// workspace's domain manifests. `"[]"` when no route-driven domain routes the
4772/// service, or when the one that does declares no headers.
4773///
4774/// Reads `.yah/domains/` directly rather than taking a loaded [`CloudConfig`]:
4775/// the static reconcilers are handed a per-component [`ReconcileCtx`], not the
4776/// whole workspace config, and threading a config reference through all 22 of
4777/// its construction sites to reach one string would be a wide change for a
4778/// narrow read. Manifest parse errors propagate — a domain file that no longer
4779/// loads is a deploy-stopping fact, not a reason to ship a Worker with the
4780/// headers quietly missing.
4781pub fn route_headers_for_service(workspace_root: &Path, service: &str) -> Result<String> {
4782 let domains = load_domains(&crate::paths::domains_dir(workspace_root))?;
4783 Ok(domain_serving_service(&domains, service)
4784 .map(DomainConfig::route_headers_json)
4785 .unwrap_or_else(|| "[]".to_string()))
4786}
4787
4788impl DomainConfig {
4789 /// Parse a single `.yah/domains/<name>.toml`, rejecting a manifest whose
4790 /// declared front door contradicts its route table
4791 /// ([`Self::validate_front_door`]).
4792 pub fn load(path: &Path) -> Result<Self> {
4793 let src =
4794 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
4795 let dom: Self =
4796 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
4797 dom.validate_front_door()
4798 .with_context(|| format!("validating {}", path.display()))?;
4799 dom.validate_route_headers()
4800 .with_context(|| format!("validating {}", path.display()))?;
4801 Ok(dom)
4802 }
4803
4804 /// R594-F12 — the front door must agree with the rest of the manifest.
4805 ///
4806 /// - `bucket-direct` is an R2 custom domain: a Worker route table would
4807 /// never be consulted, so declaring one means the author expected
4808 /// Worker behaviour (clean URLs, SPA fallback, branded errors) from a
4809 /// surface that cannot provide it. Rejected rather than silently
4810 /// ignored. Same for `worker_bundle_path` — nothing would deploy it.
4811 /// - `worker` / `passway` with an empty route table is a silent 404
4812 /// machine: the front door exists, has nothing to serve, and every
4813 /// request falls through to the catch-all.
4814 ///
4815 /// Called from [`Self::load`], so both [`CloudConfig::load`] and
4816 /// [`CloudConfig::load_from_config_dir`] enforce it.
4817 pub fn validate_front_door(&self) -> Result<()> {
4818 match self.front_door {
4819 FrontDoor::BucketDirect => {
4820 if let Some(route) = self.routes.first() {
4821 anyhow::bail!(
4822 "front_door = \"bucket-direct\" but routes[0].path = \"{}\" — \
4823 an R2 custom domain never consults a route table, so this \
4824 route would silently do nothing (no clean URLs, no SPA \
4825 fallback, no branded errors). Set front_door = \"worker\" \
4826 (or \"passway\") to keep the routes, or drop the [[routes]] \
4827 to keep the bucket-direct binding.",
4828 route.path
4829 );
4830 }
4831 if let Some(path) = &self.worker_bundle_path {
4832 anyhow::bail!(
4833 "front_door = \"bucket-direct\" but worker_bundle_path = \
4834 \"{path}\" — nothing deploys a Worker bundle for a domain \
4835 bound straight to R2"
4836 );
4837 }
4838 }
4839 FrontDoor::Worker | FrontDoor::Passway => {
4840 if self.routes.is_empty() {
4841 anyhow::bail!(
4842 "front_door = \"{}\" but [[routes]] is empty — a front door \
4843 with no route table is a silent 404 machine. Declare at \
4844 least one route, or set front_door = \"bucket-direct\" if \
4845 this domain really is served straight from R2.",
4846 self.front_door.as_str()
4847 );
4848 }
4849 }
4850 }
4851 Ok(())
4852 }
4853
4854 /// The `ROUTE_HEADERS` Worker binding for this domain (R746) — the route
4855 /// table's `path` + `headers` pairs, in manifest order, with routes that
4856 /// declare no headers dropped. `"[]"` when nothing declares any.
4857 ///
4858 /// Order is load-bearing and must survive serialization: the front door
4859 /// applies the FIRST matching rule, so `/app/*` above `/*` is what gives
4860 /// the app its isolation headers and leaves the marketing site alone.
4861 /// That is why this is a `Vec` of pairs and not a map keyed by path.
4862 ///
4863 /// Infallible by design — [`Self::validate_route_headers`] has already run
4864 /// at [`Self::load`], so by the time a reconciler calls this the table is
4865 /// known to be one both front doors can apply.
4866 pub fn route_headers_json(&self) -> String {
4867 #[derive(Serialize)]
4868 struct Rule<'a> {
4869 path: &'a str,
4870 headers: &'a BTreeMap<String, String>,
4871 }
4872 let rules: Vec<Rule<'_>> = self
4873 .routes
4874 .iter()
4875 .filter(|r| !r.headers.is_empty())
4876 .map(|r| Rule {
4877 path: &r.path,
4878 headers: &r.headers,
4879 })
4880 .collect();
4881 serde_json::to_string(&rules).unwrap_or_else(|_| "[]".to_string())
4882 }
4883
4884 /// R749-T5 — everything [`Self::route_headers_json`] emits must be
4885 /// *applicable*, checked here where the table is PRODUCED.
4886 ///
4887 /// That method serializes a typed struct, so the table's JSON *shape* is
4888 /// sound by construction. Its contents are not: a route's `headers` map is
4889 /// a free-form `name -> value` read verbatim out of hand-written TOML, so
4890 /// `"Cross Origin Opener Policy"` (spaces instead of hyphens) or a value
4891 /// carrying a newline ships a structurally-valid table that neither front
4892 /// door can apply — and they fail *differently*, neither naming the
4893 /// manifest line responsible:
4894 ///
4895 /// - **passway** — `mesofact::route_headers::RouteHeaderTable::parse`
4896 /// refuses the start, so the origin is simply down.
4897 /// - **worker** — `validateRouteHeaderTable` accepts it (it checks shape,
4898 /// not header validity) and `applyRouteHeaders` then throws inside the
4899 /// exported `fetch`, which is a 500 on every request, not the
4900 /// serve-without-the-headers degradation that code intends.
4901 ///
4902 /// So the strictness lives at the producer: a table that cannot be applied
4903 /// fails `yah cloud apply` at manifest load, naming domain, route and
4904 /// header. This is deliberately *not* a second parser — the check is
4905 /// `HeaderName`/`HeaderValue`'s own, the very constructors the passway door
4906 /// runs on the far side, and route *matching* semantics stay defined once,
4907 /// at the doors. Only routes that contribute to the table are checked, so
4908 /// the invariant is exactly "`route_headers_json`'s output parses".
4909 ///
4910 /// Called from [`Self::load`], alongside [`Self::validate_front_door`].
4911 pub fn validate_route_headers(&self) -> Result<()> {
4912 use axum::http::{HeaderName, HeaderValue};
4913
4914 for route in self.routes.iter().filter(|r| !r.headers.is_empty()) {
4915 if route.path.is_empty() {
4916 anyhow::bail!(
4917 "domain \"{}\" declares response headers on a route whose `path` is \
4918 empty — a rule that matches nothing (or everything, depending on \
4919 which front door reads it) is not a policy",
4920 self.name
4921 );
4922 }
4923 for (name, value) in &route.headers {
4924 HeaderName::try_from(name.as_str()).with_context(|| {
4925 format!(
4926 "domain \"{}\" route \"{}\" declares {name:?}, which is not a valid \
4927 HTTP header name — names are token characters only, so it is \
4928 `Cross-Origin-Opener-Policy`, never `Cross Origin Opener Policy`",
4929 self.name, route.path
4930 )
4931 })?;
4932 HeaderValue::try_from(value.as_str()).with_context(|| {
4933 format!(
4934 "domain \"{}\" route \"{}\" declares {name} = {value:?}, which is not \
4935 a valid HTTP header value — no newlines and no control characters",
4936 self.name, route.path
4937 )
4938 })?;
4939 }
4940 }
4941 Ok(())
4942 }
4943
4944 /// Whether this domain's route table binds any component of `service`.
4945 pub fn serves_service(&self, service: &str) -> bool {
4946 self.routes.iter().any(|r| {
4947 r.mode
4948 .component()
4949 .and_then(split_component_ref)
4950 .is_some_and(|(svc, _)| svc == service)
4951 })
4952 }
4953
4954 /// Persist to `.yah/domains/<name>.toml`, creating the domains
4955 /// directory if needed. Create-or-overwrite.
4956 pub fn save(&self, workspace_root: &Path) -> Result<()> {
4957 let dir = crate::paths::domains_dir(workspace_root);
4958 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
4959 let path = crate::paths::domain_toml(workspace_root, &self.name);
4960 let s = toml::to_string_pretty(self)
4961 .with_context(|| format!("serializing domain {}", self.name))?;
4962 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
4963 }
4964
4965 /// Remove `.yah/domains/<name>.toml`. Returns `false` when the file
4966 /// was already absent.
4967 pub fn delete(workspace_root: &Path, name: &str) -> Result<bool> {
4968 let path = crate::paths::domain_toml(workspace_root, name);
4969 if !path.exists() {
4970 return Ok(false);
4971 }
4972 std::fs::remove_file(&path).with_context(|| format!("removing {}", path.display()))?;
4973 Ok(true)
4974 }
4975}
4976
4977impl RouteMode {
4978 /// Component reference for static/backend modes; `None` for redirects.
4979 pub fn component(&self) -> Option<&str> {
4980 match self {
4981 Self::Static { component } | Self::Backend { component, .. } => Some(component),
4982 Self::Redirect { .. } => None,
4983 }
4984 }
4985}
4986
4987// ─── Service-group vault (R706 / W294) ───────────────────────────────────────
4988
4989/// A camp's declaration of one cluster secret, from
4990/// `.yah/infra/secrets/<slug>.toml`.
4991///
4992/// This is the *authoring* side of the fleet's cluster-secret store: it names
4993/// where the value lives in the camp (a `fob` vault slot), what the fleet should
4994/// call it, and — the point of R706 — which workloads are allowed to mount it.
4995///
4996/// The declaration is not itself the enforcement point. `yah cloud secret put`
4997/// reads this file, seals the vault value under the cluster KEK, and ships the
4998/// ciphertext **with its access rule** into raft; yubaba's `ClusterResolver`
4999/// evaluates the rule on the node at mount time. Deleting this file does not
5000/// revoke anything — the record in raft is the live authority. That asymmetry is
5001/// deliberate: a rule that lived only in a git-tracked camp file would be
5002/// trivially bypassed by anyone who could reach the fleet without the camp.
5003///
5004/// ```toml
5005/// #:schema ../../schema/secret.toml.schema.json
5006/// schema_version = 1
5007/// name = "cheers/cloud-admin/verify-key"
5008/// vault_slot = "cheers-cloud-admin-verify-key"
5009/// description = "Ed25519 public key yah-cloud-admin verifies operator PASETOs with"
5010///
5011/// [access]
5012/// workloads = [{ workload = "yah-cloud-admin" }]
5013///
5014/// [target]
5015/// kind = "file"
5016/// path = "/run/secrets/cheers-verify.key"
5017/// mode = 0o400
5018/// ```
5019#[derive(Debug, Clone, Serialize, Deserialize)]
5020#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5021pub struct SecretConfig {
5022 pub schema_version: u32,
5023
5024 /// Logical cluster-secret key, as `SecretRef::Cluster { name }` spells it —
5025 /// e.g. `"tls/yah.dev/cert"`, `"cheers/cloud-admin/verify-key"`. May contain
5026 /// `/`; the file stem is a filesystem-safe slug and carries no meaning.
5027 pub name: String,
5028
5029 /// The `fob` vault slot in this camp holding the plaintext value. Read by
5030 /// `yah cloud secret put` at ship time and never recorded anywhere else — in
5031 /// particular the value is not in this file, so the declaration is safe to
5032 /// commit.
5033 pub vault_slot: String,
5034
5035 /// Human note for `yah cloud secret ls`. What this secret is and who minted
5036 /// it — the thing nobody remembers 6 months later.
5037 #[serde(default, skip_serializing_if = "Option::is_none")]
5038 pub description: Option<String>,
5039
5040 /// How the vault slot's text decodes into the bytes the consumer expects.
5041 ///
5042 /// `fob` slots hold strings, but plenty of real secrets are **binary** — an
5043 /// Ed25519 key is exactly 32 raw bytes, and `yah-cloud-admin` rejects a key
5044 /// file of any other length. Without this field the only way to ship such a
5045 /// key would be to hope its bytes happened to be valid UTF-8, which for a
5046 /// random key they are not.
5047 ///
5048 /// Defaults to [`SecretEncoding::Utf8`] — the right answer for tokens,
5049 /// passwords, and PEM, which is most secrets.
5050 #[serde(default)]
5051 pub encoding: SecretEncoding,
5052
5053 /// Who may mount it. Stamped onto the raft record verbatim.
5054 ///
5055 /// Defaults to [`SecretAccess::default`] — the deny-all empty allow-list. A
5056 /// declaration that forgets this field produces a secret nobody can mount,
5057 /// which is the correct direction to fail in.
5058 ///
5059 /// Three forms:
5060 ///
5061 /// ```toml
5062 /// access = "allow_any" # explicit escape hatch
5063 ///
5064 /// [access] # named workloads
5065 /// workloads = [{ workload = "yah-cloud-admin" }]
5066 ///
5067 /// [access] # signed recipes (R555-F5)
5068 /// recipes = [{ recipe = "rusty-v8-musl", key = "3d40…" }]
5069 /// ```
5070 ///
5071 /// Use the `recipes` form for a credential a **dispatched build** needs (the
5072 /// R2 write key, the cosign signing key). A remote QED run's workload name
5073 /// is a fresh `forge-<uuid>` every time, so `workloads` cannot name it and
5074 /// `allow_any` over-answers — see W235 §Seam (c) secret scoping. `key` is
5075 /// the hex Ed25519 public key from the recipe's `[admission]` block.
5076 #[serde(default)]
5077 pub access: SecretAccess,
5078
5079 /// Advisory: the mount shape a consuming workload should declare. Not
5080 /// enforced — yubaba honours whatever the `WorkloadSpec` asks for — but it
5081 /// lets `yah cloud secret put` print the exact `SecretMount` to paste, so
5082 /// the consumer and the declaration can't drift on path or mode.
5083 #[serde(default, skip_serializing_if = "Option::is_none")]
5084 pub target: Option<SecretTargetDecl>,
5085}
5086
5087/// How a [`SecretConfig`]'s vault text becomes the bytes delivered to the
5088/// container (R706 / W294).
5089#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
5090#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5091#[serde(rename_all = "kebab-case")]
5092pub enum SecretEncoding {
5093 /// Ship the vault string's UTF-8 bytes verbatim. Tokens, passwords, PEM.
5094 #[default]
5095 Utf8,
5096 /// The vault string is hex; ship the decoded bytes. Use for binary key
5097 /// material — e.g. a raw Ed25519 key, which must land as exactly 32 bytes.
5098 Hex,
5099}
5100
5101/// Advisory mount shape on a [`SecretConfig`]. Mirrors
5102/// `workload_spec::SecretTarget` in a TOML-friendly, externally-tagged-free
5103/// shape (a `kind` discriminator reads better in a hand-written manifest than
5104/// serde's default enum encoding).
5105#[derive(Debug, Clone, Serialize, Deserialize)]
5106#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5107#[serde(tag = "kind", rename_all = "kebab-case")]
5108pub enum SecretTargetDecl {
5109 /// Mounted as a tmpfs-backed file inside the container.
5110 File {
5111 /// Absolute path inside the container.
5112 path: String,
5113 /// Unix permission bits. Defaults to `0o400` (owner-read-only).
5114 #[serde(default = "default_secret_mode")]
5115 mode: u32,
5116 },
5117 /// Injected as an environment variable. Prefer `file` — env vars leak
5118 /// through subprocess environments and log dumps.
5119 EnvVar { name: String },
5120}
5121
5122fn default_secret_mode() -> u32 {
5123 0o400
5124}
5125
5126impl SecretTargetDecl {
5127 /// The `workload_spec` target this declaration describes.
5128 pub fn to_target(&self) -> workload_spec::SecretTarget {
5129 match self {
5130 Self::File { path, mode } => workload_spec::SecretTarget::File {
5131 path: path.into(),
5132 mode: *mode,
5133 },
5134 Self::EnvVar { name } => workload_spec::SecretTarget::EnvVar { name: name.clone() },
5135 }
5136 }
5137}
5138
5139impl SecretConfig {
5140 /// Parse a single `.yah/infra/secrets/<slug>.toml`.
5141 pub fn load(path: &Path) -> Result<Self> {
5142 let src =
5143 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
5144 let cfg: Self =
5145 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
5146 cfg.validate()
5147 .with_context(|| format!("validating {}", path.display()))?;
5148 Ok(cfg)
5149 }
5150
5151 /// Load every declaration in `dir`, keyed by logical secret name. A missing
5152 /// directory is an empty map (a camp with no cluster secrets is normal).
5153 ///
5154 /// Two files declaring the same `name` is a hard error, not a last-writer-
5155 /// wins merge: they would race to define the access rule for one record, and
5156 /// whichever lost would look correct in git while being inert on the fleet.
5157 pub fn load_dir(dir: &Path) -> Result<BTreeMap<String, Self>> {
5158 let mut out: BTreeMap<String, Self> = BTreeMap::new();
5159 if !dir.exists() {
5160 return Ok(out);
5161 }
5162 for entry in std::fs::read_dir(dir).with_context(|| format!("reading {}", dir.display()))? {
5163 let path = entry?.path();
5164 if path.extension().is_none_or(|e| e != "toml") {
5165 continue;
5166 }
5167 let cfg = Self::load(&path)?;
5168 if let Some(prev) = out.insert(cfg.name.clone(), cfg) {
5169 anyhow::bail!(
5170 "two secret declarations both claim name {:?} (one of them is {}); \
5171 a cluster secret must have exactly one declaration so its access \
5172 rule has one author",
5173 prev.name,
5174 path.display()
5175 );
5176 }
5177 }
5178 Ok(out)
5179 }
5180
5181 /// Reject declarations that would produce an unusable or dangerous record.
5182 pub fn validate(&self) -> Result<()> {
5183 if self.name.trim().is_empty() {
5184 anyhow::bail!("`name` must not be empty");
5185 }
5186 if self.vault_slot.trim().is_empty() {
5187 anyhow::bail!(
5188 "`vault_slot` must not be empty — it names the fob slot holding the value"
5189 );
5190 }
5191 // A deny-all rule is a *valid* record (it is the fail-closed default the
5192 // resolver relies on) but it is never a useful thing to deliberately
5193 // ship, so catching it here saves an operator the round-trip of
5194 // deploying a workload that mysteriously can't see its own secret.
5195 if let SecretAccess::Workloads(entries) = &self.access {
5196 if entries.is_empty() {
5197 anyhow::bail!(
5198 "`[access]` admits nobody: list the workloads allowed to mount {:?} \
5199 (e.g. `workloads = [{{ workload = \"my-service\" }}]`), or set \
5200 `access = \"allow_any\"` to store it unrestricted",
5201 self.name
5202 );
5203 }
5204 if let Some(bad) = entries.iter().find(|e| e.workload.trim().is_empty()) {
5205 anyhow::bail!("`[access]` entry has an empty `workload` name: {bad:?}");
5206 }
5207 }
5208 Ok(())
5209 }
5210}
5211
5212#[cfg(test)]
5213mod secret_config_tests {
5214 use super::*;
5215
5216 fn parse(body: &str) -> Result<SecretConfig> {
5217 let cfg: SecretConfig = toml::from_str(body)?;
5218 cfg.validate()?;
5219 Ok(cfg)
5220 }
5221
5222 #[test]
5223 fn minimal_declaration_parses_with_narrow_defaults() {
5224 let cfg = parse(
5225 r#"
5226schema_version = 1
5227name = "svc/token"
5228vault_slot = "svc-token"
5229[access]
5230workloads = [{ workload = "svc" }]
5231"#,
5232 )
5233 .unwrap();
5234
5235 assert_eq!(cfg.encoding, SecretEncoding::Utf8, "text is the default");
5236 assert!(cfg.target.is_none());
5237 // The omitted tenant/namespace must narrow to the singletons, not widen
5238 // to a wildcard.
5239 assert!(cfg
5240 .access
5241 .admits(&workload_spec::secrets::SecretConsumer::workload("svc")));
5242 assert!(!cfg
5243 .access
5244 .admits(&workload_spec::secrets::SecretConsumer::workload("other")));
5245 }
5246
5247 #[test]
5248 fn allow_any_is_spelled_as_a_bare_string() {
5249 // The operator-facing spelling, pinned: `access = "allow_any"`.
5250 let cfg = parse(
5251 r#"
5252schema_version = 1
5253name = "public/thing"
5254vault_slot = "slot"
5255access = "allow_any"
5256"#,
5257 )
5258 .unwrap();
5259 assert_eq!(cfg.access, SecretAccess::AllowAny);
5260 }
5261
5262 #[test]
5263 fn a_declaration_with_no_access_block_is_rejected() {
5264 // Omitting `[access]` defaults to deny-all, which is the correct
5265 // *runtime* default but never a correct authoring intent — so it must
5266 // not silently produce a secret nobody can mount.
5267 let err = parse(
5268 r#"
5269schema_version = 1
5270name = "svc/token"
5271vault_slot = "svc-token"
5272"#,
5273 )
5274 .unwrap_err()
5275 .to_string();
5276 assert!(err.contains("admits nobody"), "got {err}");
5277 }
5278
5279 #[test]
5280 fn empty_name_or_slot_is_rejected() {
5281 assert!(parse(
5282 r#"
5283schema_version = 1
5284name = ""
5285vault_slot = "slot"
5286access = "allow_any"
5287"#
5288 )
5289 .is_err());
5290 assert!(parse(
5291 r#"
5292schema_version = 1
5293name = "x"
5294vault_slot = " "
5295access = "allow_any"
5296"#
5297 )
5298 .is_err());
5299 }
5300
5301 #[test]
5302 fn target_declaration_maps_onto_the_workload_spec_type() {
5303 let cfg = parse(
5304 r#"
5305schema_version = 1
5306name = "svc/token"
5307vault_slot = "slot"
5308access = "allow_any"
5309[target]
5310kind = "file"
5311path = "/run/secrets/t"
5312"#,
5313 )
5314 .unwrap();
5315 match cfg.target.unwrap().to_target() {
5316 workload_spec::SecretTarget::File { path, mode } => {
5317 assert_eq!(path, std::path::PathBuf::from("/run/secrets/t"));
5318 assert_eq!(mode, 0o400, "owner-read-only by default");
5319 }
5320 other => panic!("expected File, got {other:?}"),
5321 }
5322 }
5323
5324 #[test]
5325 fn load_dir_is_empty_for_a_camp_with_no_secrets() {
5326 let tmp = tempfile::TempDir::new().unwrap();
5327 assert!(SecretConfig::load_dir(&tmp.path().join("nope"))
5328 .unwrap()
5329 .is_empty());
5330 }
5331}
5332
5333/// Split a `"<service>/<component-id>"` ref. Returns `None` if the ref
5334/// isn't shaped like `service/component`.
5335fn split_component_ref(s: &str) -> Option<(&str, &str)> {
5336 let (svc, comp) = s.split_once('/')?;
5337 if svc.is_empty() || comp.is_empty() || comp.contains('/') {
5338 return None;
5339 }
5340 Some((svc, comp))
5341}
5342
5343#[cfg(test)]
5344mod tests {
5345 use super::*;
5346 use std::path::PathBuf;
5347
5348 fn make_machine(name: &str, mesh_tags: Vec<&str>) -> MachineConfig {
5349 MachineConfig {
5350 name: name.into(),
5351 provider: "hetzner".into(),
5352 location: Some("hil".into()),
5353 server_type: Some("ccx13".into()),
5354 hosts_mirrors: vec![],
5355 mesh_tags: mesh_tags.into_iter().map(String::from).collect(),
5356 region: None,
5357 zone: None,
5358 arch: None,
5359 bucket: None,
5360 vendor: None,
5361 nickname: None,
5362 legacy_hostkey_fingerprint: None,
5363 registration: Default::default(),
5364 ssh_keys: vec![],
5365 cloudflared: None,
5366 hosts_operator_bridge: false,
5367 connect: None,
5368 allocatable: None,
5369 taints: vec![],
5370 sovereign_group: None,
5371 sovereign_role: None,
5372 ingress_floating_ip: None,
5373 }
5374 }
5375
5376 /// Like [`make_machine`] but with explicit topology axes for F16 tests.
5377 fn make_machine_topo(
5378 name: &str,
5379 provider: &str,
5380 region: &str,
5381 mesh_tags: Vec<&str>,
5382 ) -> MachineConfig {
5383 MachineConfig {
5384 provider: provider.into(),
5385 region: Some(region.into()),
5386 zone: Some(region.into()),
5387 ..make_machine(name, mesh_tags)
5388 }
5389 }
5390
5391 fn make_empty_cfg(machines: Vec<MachineConfig>) -> CloudConfig {
5392 CloudConfig {
5393 workspace_root: PathBuf::new(),
5394 machines,
5395 providers: vec![],
5396 machine_origins: BTreeMap::new(),
5397 provider_origins: BTreeMap::new(),
5398 services: BTreeMap::new(),
5399 domains: BTreeMap::new(),
5400 legacy_mirrors: vec![],
5401 workloads: vec![],
5402 topology: TopologyConfig::default(),
5403 legacy_services: vec![],
5404 }
5405 }
5406
5407 #[test]
5408 fn required_spec_parses_from_provider_fields() {
5409 let toml_src = r#"
5410use = "hetzner-primary"
5411[required]
5412mesh_tags = ["tag:cloud-runner"]
5413"#;
5414 let slot: MirrorProviderSlot = toml::from_str(toml_src).unwrap();
5415 let req = slot.required().expect("required block present");
5416 assert_eq!(req.mesh_tags, vec!["tag:cloud-runner"]);
5417 }
5418
5419 #[test]
5420 fn required_spec_absent_when_field_missing() {
5421 let slot: MirrorProviderSlot = toml::from_str(r#"use = "hetzner-primary""#).unwrap();
5422 assert!(slot.required().is_none());
5423 }
5424
5425 #[test]
5426 fn db_catalog_parses_all_env_blocks() {
5427 // W241 / R571-F8: a service.toml [db] table with dev/pond/cloud.
5428 let toml_src = r#"
5429schema_version = 1
5430name = "scrabcake"
5431domain = "scrabcake.net.yah.dev"
5432
5433[[db.dev]]
5434name = "main"
5435path = "data/dev.sqlite"
5436
5437[[db.pond]]
5438name = "main"
5439port = 5433
5440
5441[[db.pond]]
5442name = "pg"
5443port = 5432
5444kind = "postgres"
5445
5446[[db.cloud]]
5447name = "main"
5448url = "libsql://scrabcake.turso.io"
5449auth_token_env = "SCRABCAKE_TURSO_TOKEN"
5450"#;
5451 let svc: ServiceConfig = toml::from_str(toml_src).unwrap();
5452 assert_eq!(svc.db.dev.len(), 1);
5453 assert_eq!(svc.db.dev[0].path, "data/dev.sqlite");
5454 assert_eq!(svc.db.pond.len(), 2);
5455 assert_eq!(svc.db.pond[0].port, Some(5433));
5456 assert_eq!(svc.db.pond[0].kind, PondDbKind::Turso); // default
5457 assert_eq!(svc.db.pond[1].kind, PondDbKind::Postgres);
5458 assert_eq!(
5459 svc.db.cloud[0].auth_token_env.as_deref(),
5460 Some("SCRABCAKE_TURSO_TOKEN")
5461 );
5462 }
5463
5464 #[test]
5465 fn service_without_db_table_has_empty_catalog() {
5466 let svc: ServiceConfig =
5467 toml::from_str("schema_version = 1\nname = \"s\"\ndomain = \"s.dev\"\n").unwrap();
5468 assert!(svc.db.is_empty());
5469 // And an empty [db] must not appear when re-serialized.
5470 let out = toml::to_string(&svc).unwrap();
5471 assert!(
5472 !out.contains("[db"),
5473 "empty db table should be skipped: {out}"
5474 );
5475 }
5476
5477 #[test]
5478 fn camp_shared_cloud_toml_parses() {
5479 let src = r#"
5480[[cloud]]
5481name = "analytics"
5482url = "postgres://shared/analytics"
5483"#;
5484 let shared: CampCloudDbs = toml::from_str(src).unwrap();
5485 assert_eq!(shared.cloud.len(), 1);
5486 assert_eq!(shared.cloud[0].name, "analytics");
5487 }
5488
5489 #[test]
5490 fn resolve_machine_by_mesh_tags_superset_match() {
5491 let cfg = make_empty_cfg(vec![
5492 make_machine("yah-bnt-1", vec!["tag:primary-yah", "tag:tier-scratch"]),
5493 make_machine("us-west-001", vec!["tag:primary-yah", "tag:cloud-runner"]),
5494 ]);
5495 let picked = cfg
5496 .resolve_machine_by_mesh_tags(&["tag:cloud-runner".into()])
5497 .map(|m| m.name.as_str());
5498 assert_eq!(picked, Some("us-west-001"));
5499 }
5500
5501 #[test]
5502 fn resolve_machine_by_mesh_tags_returns_none_when_no_match() {
5503 let cfg = make_empty_cfg(vec![make_machine("yah-bnt-1", vec!["tag:primary-yah"])]);
5504 assert!(cfg
5505 .resolve_machine_by_mesh_tags(&["tag:cloud-runner".into()])
5506 .is_none());
5507 }
5508
5509 // ─── R590-F1 mesh-tag node-selector admission ───────────────────────────
5510
5511 /// Build a forge WorkloadSpec carrying the R594 node-selector annotation.
5512 /// `selector` is the comma-joined mesh-tag set; `None` omits the annotation
5513 /// entirely (pre-R594 "no constraint").
5514 fn ws_with_selector(selector: Option<&str>) -> WorkloadSpec {
5515 use workload_spec::{ImageRef, TierTag};
5516 let mut ws = WorkloadSpec::for_forge(
5517 "R590-F1-test",
5518 ImageRef {
5519 registry: "docker.io".into(),
5520 repository: "library/busybox".into(),
5521 tag: "latest".into(),
5522 digest: workload_spec::testing::test_digest(),
5523 },
5524 TierTag("infra".into()),
5525 vec![],
5526 );
5527 if let Some(sel) = selector {
5528 ws.annotations.insert(
5529 velveteen_exec::remote::NODE_SELECTOR_MESH_TAGS_ANNOTATION.into(),
5530 sel.into(),
5531 );
5532 }
5533 ws
5534 }
5535
5536 /// The build-worker fleet shape: one x86 node (us-west-002) and one arm
5537 /// node (a Pi5), both carrying `tag:build-worker`.
5538 fn build_worker_fleet() -> CloudConfig {
5539 make_empty_cfg(vec![
5540 make_machine("us-west-002", vec!["tag:build-worker", "arch:x86"]),
5541 make_machine("pi5-001", vec!["tag:build-worker", "arch:arm"]),
5542 ])
5543 }
5544
5545 #[test]
5546 fn admit_workload_routes_amd64_to_x86_worker() {
5547 let cfg = build_worker_fleet();
5548 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
5549 let picked = cfg.admit_workload(&ws).unwrap();
5550 assert_eq!(picked.name, "us-west-002");
5551 }
5552
5553 #[test]
5554 fn admit_workload_routes_arm64_to_pi5_worker() {
5555 let cfg = build_worker_fleet();
5556 let ws = ws_with_selector(Some("tag:build-worker,arch:arm"));
5557 let picked = cfg.admit_workload(&ws).unwrap();
5558 assert_eq!(picked.name, "pi5-001");
5559 }
5560
5561 /// A forge run must be admissible on a build-worker smaller than its own
5562 /// cgroup ceiling.
5563 ///
5564 /// The fleet's arm build-workers are 8 GiB Pi-5s and `for_forge` sets a
5565 /// 32 GiB ceiling, so while admission read `resources.memory_mb` as the
5566 /// capacity floor this returned "no candidates" and *every* offloaded qed
5567 /// step to those nodes failed at dispatch — measured on desktop-release run
5568 /// b04cef47, where the aarch64-linux row died in 1.6s. The other
5569 /// build-workers (16 GiB us-west-003, and the arm Pi-5s) were excluded the
5570 /// same way, leaving one 47 GiB node as the fleet's only legal target for
5571 /// remote CI.
5572 #[test]
5573 fn admit_workload_places_a_forge_run_on_a_worker_smaller_than_its_ceiling() {
5574 let mut pi = make_machine("pi5-001", vec!["tag:build-worker", "arch:arm"]);
5575 pi.allocatable = Some(NodeAllocatable {
5576 memory_mb: 8192,
5577 cpu_millis: 4000,
5578 });
5579 let cfg = make_empty_cfg(vec![pi]);
5580
5581 let ws = ws_with_selector(Some("tag:build-worker,arch:arm"));
5582 assert!(
5583 ws.resources.memory_mb > 8192,
5584 "precondition: the ceiling must exceed the node, or this proves nothing"
5585 );
5586
5587 let picked = cfg
5588 .admit_workload(&ws)
5589 .expect("an 8 GiB build-worker must admit a forge run");
5590 assert_eq!(picked.name, "pi5-001");
5591 }
5592
5593 /// The floor is still enforced — the fix separates two numbers, it does not
5594 /// disable the R572-F5 capacity check.
5595 #[test]
5596 fn admit_workload_still_rejects_a_node_below_the_declared_request() {
5597 let mut tiny = make_machine("tiny-001", vec!["tag:build-worker", "arch:arm"]);
5598 tiny.allocatable = Some(NodeAllocatable {
5599 memory_mb: 512,
5600 cpu_millis: 4000,
5601 });
5602 let cfg = make_empty_cfg(vec![tiny]);
5603
5604 let ws = ws_with_selector(Some("tag:build-worker,arch:arm"));
5605 assert!(
5606 cfg.admit_workload(&ws).is_err(),
5607 "a 512 MiB node cannot satisfy a 2 GiB forge request"
5608 );
5609 }
5610
5611 // ─── R833-F8 imperative node-selector admission ─────────────────────────
5612
5613 /// Build a forge WorkloadSpec carrying the R833-F8 imperative node
5614 /// selector — the operator's `--where=node:<machine>`.
5615 fn ws_pinned_to(node: &str) -> WorkloadSpec {
5616 let mut ws = ws_with_selector(None);
5617 ws.annotations.insert(
5618 velveteen_exec::remote::NODE_SELECTOR_NODE_ANNOTATION.into(),
5619 node.into(),
5620 );
5621 ws
5622 }
5623
5624 /// The ticket's acceptance shape: a named node wins over the
5625 /// declaration-order tie-break that would otherwise decide placement.
5626 /// `us-west-002` is declared first and carries every tag, so an inferred
5627 /// placement lands there; the pin must reach `pi5-001` regardless.
5628 #[test]
5629 fn admit_workload_honours_an_explicitly_named_node() {
5630 let cfg = build_worker_fleet();
5631 assert_eq!(
5632 cfg.admit_workload(&ws_with_selector(Some("tag:build-worker")))
5633 .unwrap()
5634 .name,
5635 "us-west-002",
5636 "precondition: inference elects the first-declared node",
5637 );
5638 assert_eq!(
5639 cfg.admit_workload(&ws_pinned_to("pi5-001")).unwrap().name,
5640 "pi5-001",
5641 );
5642 }
5643
5644 /// A pin at a machine that is not declared fails loud, naming the
5645 /// constraint and the pool — the operator mistyped a node, and silently
5646 /// running the build somewhere else is the one outcome that must not
5647 /// happen.
5648 #[test]
5649 fn admit_workload_refuses_a_node_that_is_not_declared() {
5650 let cfg = build_worker_fleet();
5651 let err = cfg
5652 .admit_workload(&ws_pinned_to("us-west-404"))
5653 .unwrap_err()
5654 .to_string();
5655 assert!(err.contains("required.nodes=[us-west-404]"), "{err}");
5656 assert!(err.contains("us-west-002"), "the pool must be named: {err}");
5657 }
5658
5659 /// The pin narrows the candidate set; it does not suspend the other axes.
5660 /// A named node that cannot fit the workload still refuses, rather than
5661 /// being handed work it has no room for.
5662 #[test]
5663 fn a_pinned_node_is_still_checked_against_capacity() {
5664 let mut tiny = make_machine("tiny-001", vec!["tag:build-worker", "arch:arm"]);
5665 tiny.allocatable = Some(NodeAllocatable {
5666 memory_mb: 512,
5667 cpu_millis: 4000,
5668 });
5669 let cfg = make_empty_cfg(vec![tiny]);
5670 assert!(cfg.admit_workload(&ws_pinned_to("tiny-001")).is_err());
5671 }
5672
5673 /// Inference is untouched: with no node annotation the `nodes` axis is
5674 /// empty, which is "no constraint" — every pre-R833-F8 workload is admitted
5675 /// exactly as before.
5676 #[test]
5677 fn an_unpinned_workload_carries_no_node_constraint() {
5678 assert!(node_selector_node(&ws_with_selector(Some("arch:x86"))).is_none());
5679 assert_eq!(
5680 node_selector_node(&ws_pinned_to("us-west-003")).as_deref(),
5681 Some("us-west-003")
5682 );
5683 assert!(RequiredSpec::default().is_unconstrained());
5684 assert!(!RequiredSpec {
5685 nodes: vec!["us-west-003".into()],
5686 ..Default::default()
5687 }
5688 .is_unconstrained());
5689 }
5690
5691 #[test]
5692 fn admit_workload_rejects_node_missing_required_tag() {
5693 // Only an arm worker exists; an x86 build must NOT land on it.
5694 let cfg = make_empty_cfg(vec![make_machine(
5695 "pi5-001",
5696 vec!["tag:build-worker", "arch:arm"],
5697 )]);
5698 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
5699 assert!(cfg.admit_workload(&ws).is_err());
5700 }
5701
5702 /// R555-S1 regression: with TWO nodes carrying the same tag set, which one
5703 /// admits must be decided by *declaration order* (file name), which is the
5704 /// contract `admit_workload` documents — not by `read_dir` order, which is
5705 /// filesystem-dependent and can change when an unrelated file appears in
5706 /// the directory. Written creation-order-reversed so a filesystem that
5707 /// yields creation order (rather than sorted order) trips it without the
5708 /// sort in `load_dir`.
5709 ///
5710 /// Live consequence this guards: `.yah/infra/machines/` carries both
5711 /// us-west-002 and us-west-003 on `[tag:build-worker, arch:x86, os:linux]`,
5712 /// so an x86 QED offload has two equal candidates. Unstable selection means
5713 /// a retried build cannot be relied on to land back on the node whose
5714 /// working state it left behind.
5715 #[test]
5716 fn equally_matching_machines_admit_in_file_name_order() {
5717 let tmp = tempfile::TempDir::new().unwrap();
5718 let machines = tmp.path().join(".yah").join("infra").join("machines");
5719 std::fs::create_dir_all(&machines).unwrap();
5720 let toml_for = |name: &str| {
5721 format!(
5722 r#"name = "{name}"
5723provider = "static"
5724mesh_tags = ["tag:build-worker", "arch:x86"]
5725"#
5726 )
5727 };
5728 // Reverse-of-sorted creation order on purpose.
5729 std::fs::write(machines.join("b-second.toml"), toml_for("b-second")).unwrap();
5730 std::fs::write(machines.join("a-first.toml"), toml_for("a-first")).unwrap();
5731
5732 let cfg = CloudConfig::load(tmp.path()).unwrap();
5733 assert_eq!(
5734 cfg.machines.iter().map(|m| m.name.as_str()).collect::<Vec<_>>(),
5735 vec!["a-first", "b-second"],
5736 "machines must load in file-name order, not read_dir order"
5737 );
5738
5739 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
5740 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "a-first");
5741
5742 // R605-T14: the same two nodes, seen as the pool they are. The head is
5743 // what `admit_workload` returns, and the tail is what the dispatcher
5744 // fails over to when the head does not answer — so these two views must
5745 // come from one predicate, not two.
5746 assert_eq!(
5747 cfg.admit_workload_candidates(&ws)
5748 .unwrap()
5749 .iter()
5750 .map(|m| m.name.as_str())
5751 .collect::<Vec<_>>(),
5752 vec!["a-first", "b-second"],
5753 "the pool must be every admissible node, in the same declaration order"
5754 );
5755 }
5756
5757 /// A pool of one is still a pool, and a pool of none is an `Err` that reads
5758 /// exactly like `admit_workload`'s — "nothing admits this" is one failure
5759 /// with one wording, not two.
5760 #[test]
5761 fn admit_workload_candidates_matches_admit_workload_on_the_edges() {
5762 let cfg = make_empty_cfg(vec![
5763 make_machine("x86-box", vec!["tag:build-worker", "arch:x86"]),
5764 make_machine("arm-box", vec!["tag:build-worker", "arch:arm"]),
5765 ]);
5766
5767 let one = ws_with_selector(Some("arch:arm"));
5768 assert_eq!(
5769 cfg.admit_workload_candidates(&one)
5770 .unwrap()
5771 .iter()
5772 .map(|m| m.name.as_str())
5773 .collect::<Vec<_>>(),
5774 vec!["arm-box"],
5775 "only one node carries arch:arm, so the pool is that one node"
5776 );
5777
5778 let none = ws_with_selector(Some("arch:riscv"));
5779 let pool_err = cfg.admit_workload_candidates(&none).unwrap_err().to_string();
5780 let single_err = cfg.admit_workload(&none).unwrap_err().to_string();
5781 assert_eq!(
5782 pool_err, single_err,
5783 "an empty pool must be refused in the same words as an unadmitted workload"
5784 );
5785 }
5786
5787 /// R844-B7 — the wrong-root half of the distinction. A directory with no
5788 /// `.yah/` at all used to load as a valid config with zero machines, so a
5789 /// caller pointed at the wrong directory got a green result that measured
5790 /// nothing. Asserting `load` merely *succeeds* is what let that through;
5791 /// the shape that catches it is a non-zero machine count, or — here — an
5792 /// `Err` naming the path that was looked for.
5793 #[test]
5794 fn loading_a_directory_that_is_not_a_yah_workspace_is_an_error() {
5795 let tmp = tempfile::TempDir::new().unwrap();
5796 // A plausible-looking package root: real files, real subdirectories,
5797 // no `.yah/`. This is exactly what `load_cloud(".")` reads when a test
5798 // runs under `cargo test` from a member crate.
5799 std::fs::create_dir_all(tmp.path().join("src")).unwrap();
5800 std::fs::write(tmp.path().join("Cargo.toml"), "[package]\nname = \"x\"\n").unwrap();
5801
5802 let err = CloudConfig::load(tmp.path()).expect_err(
5803 "a directory with no .yah/ is the WRONG DIRECTORY, not a fleet with no machines",
5804 );
5805 let msg = format!("{err:#}");
5806 assert!(
5807 msg.contains("not a yah workspace"),
5808 "error must say the root is not a workspace, got: {msg}"
5809 );
5810 assert!(
5811 msg.contains(&tmp.path().join(".yah").display().to_string()),
5812 "error must name the path it looked for so an operator sees the \
5813 wrong-root immediately, got: {msg}"
5814 );
5815 }
5816
5817 /// R844-B7 — the other half, and the reason the check is drawn at `.yah/`
5818 /// rather than at the machine list: a camp that declares no machines is a
5819 /// real workspace and must keep loading. Blanket-erroring on an empty
5820 /// fleet would conflate `unknown` with `answered with none`, which is the
5821 /// exact confusion the check exists to remove.
5822 #[test]
5823 fn a_workspace_with_no_machines_declared_still_loads() {
5824 let tmp = tempfile::TempDir::new().unwrap();
5825 std::fs::create_dir_all(tmp.path().join(".yah")).unwrap();
5826
5827 let cfg = CloudConfig::load(tmp.path())
5828 .expect("a `.yah/` with no infra/machines/ is an empty fleet, not a wrong root");
5829 assert!(cfg.machines.is_empty(), "nothing was declared");
5830 assert!(cfg.services.is_empty());
5831 assert!(cfg.providers.is_empty());
5832
5833 // And an existing-but-empty machines dir is the same answer, not a
5834 // second special case.
5835 std::fs::create_dir_all(crate::paths::machines_dir(tmp.path())).unwrap();
5836 let cfg = CloudConfig::load(tmp.path()).expect("an empty machines/ dir still loads");
5837 assert!(cfg.machines.is_empty());
5838 }
5839
5840 #[test]
5841 fn admit_workload_empty_selector_is_unconstrained() {
5842 // Absent annotation ⇒ no mesh-tag constraint ⇒ first declared machine
5843 // (pre-R594 behavior preserved).
5844 let cfg = build_worker_fleet();
5845 let ws = ws_with_selector(None);
5846 let picked = cfg.admit_workload(&ws).unwrap();
5847 assert_eq!(picked.name, "us-west-002");
5848 }
5849
5850 #[test]
5851 fn node_selector_mesh_tags_trims_and_drops_empties() {
5852 let ws = ws_with_selector(Some(" tag:build-worker , arch:x86 ,"));
5853 assert_eq!(
5854 node_selector_mesh_tags(&ws),
5855 vec!["tag:build-worker".to_string(), "arch:x86".to_string()]
5856 );
5857 assert!(node_selector_mesh_tags(&ws_with_selector(None)).is_empty());
5858 }
5859
5860 // ─── F16 topology-aware resolver ────────────────────────────────────────
5861
5862 fn two_region_fleet() -> CloudConfig {
5863 make_empty_cfg(vec![
5864 make_machine_topo(
5865 "us-west-001",
5866 "hetzner",
5867 "us-west",
5868 vec!["tag:cloud-runner"],
5869 ),
5870 make_machine_topo(
5871 "eu-west-001",
5872 "hetzner",
5873 "eu-west",
5874 vec!["tag:cloud-runner"],
5875 ),
5876 ])
5877 }
5878
5879 #[test]
5880 fn resolve_machine_matches_on_region_plus_mesh_tags() {
5881 let cfg = two_region_fleet();
5882 let req = RequiredSpec {
5883 regions: vec!["us-west".into()],
5884 mesh_tags: vec!["tag:cloud-runner".into()],
5885 ..Default::default()
5886 };
5887 let picked = cfg.resolve_machine(&req).unwrap();
5888 assert_eq!(picked.name, "us-west-001");
5889 }
5890
5891 #[test]
5892 fn resolve_machine_region_disambiguates_same_tag() {
5893 // Both boxes carry tag:cloud-runner; the region axis selects eu-west.
5894 let cfg = two_region_fleet();
5895 let req = RequiredSpec {
5896 regions: vec!["eu-west".into()],
5897 mesh_tags: vec!["tag:cloud-runner".into()],
5898 ..Default::default()
5899 };
5900 assert_eq!(cfg.resolve_machine(&req).unwrap().name, "eu-west-001");
5901 }
5902
5903 #[test]
5904 fn resolve_machine_fails_loud_with_constraint_summary() {
5905 let cfg = two_region_fleet();
5906 let req = RequiredSpec {
5907 regions: vec!["us-central".into()],
5908 mesh_tags: vec!["tag:cloud-runner".into()],
5909 ..Default::default()
5910 };
5911 let err = cfg.resolve_machine(&req).unwrap_err().to_string();
5912 assert!(err.contains("required.regions=[us-central]"), "got: {err}");
5913 assert!(
5914 err.contains("required.mesh_tags=[tag:cloud-runner]"),
5915 "got: {err}"
5916 );
5917 // Names the candidates it rejected.
5918 assert!(err.contains("us-west-001"), "got: {err}");
5919 }
5920
5921 #[test]
5922 fn resolve_machine_provider_axis_filters() {
5923 let cfg = make_empty_cfg(vec![
5924 make_machine_topo("aws-west-1", "aws", "us-west", vec!["tag:cloud-runner"]),
5925 make_machine_topo("hz-west-1", "hetzner", "us-west", vec!["tag:cloud-runner"]),
5926 ]);
5927 let req = RequiredSpec {
5928 regions: vec!["us-west".into()],
5929 providers: vec!["hetzner".into()],
5930 ..Default::default()
5931 };
5932 assert_eq!(cfg.resolve_machine(&req).unwrap().name, "hz-west-1");
5933 }
5934
5935 #[test]
5936 fn unconstrained_required_spec_matches_first_machine() {
5937 let cfg = two_region_fleet();
5938 assert!(RequiredSpec::default().is_unconstrained());
5939 assert_eq!(
5940 cfg.resolve_machine(&RequiredSpec::default()).unwrap().name,
5941 "us-west-001"
5942 );
5943 }
5944
5945 #[test]
5946 fn required_spec_parses_topology_axes_from_toml() {
5947 let toml_src = r#"
5948use = "hetzner-primary"
5949[required]
5950regions = ["us-west"]
5951mesh_tags = ["tag:cloud-runner"]
5952"#;
5953 let slot: MirrorProviderSlot = toml::from_str(toml_src).unwrap();
5954 let req = slot.required().expect("required block present");
5955 assert_eq!(req.regions, vec!["us-west"]);
5956 assert_eq!(req.mesh_tags, vec!["tag:cloud-runner"]);
5957 assert!(req.zones.is_empty());
5958 }
5959
5960 // ─── R844-F8 replica count ──────────────────────────────────────────────
5961
5962 fn three_runner_fleet() -> CloudConfig {
5963 make_empty_cfg(vec![
5964 make_machine_topo("us-east-001", "hetzner", "us-east", vec!["tag:cloud-runner"]),
5965 make_machine_topo(
5966 "us-south-001",
5967 "hetzner",
5968 "us-south",
5969 vec!["tag:cloud-runner"],
5970 ),
5971 make_machine_topo(
5972 "us-west-001",
5973 "hetzner",
5974 "us-west",
5975 vec!["tag:cloud-runner"],
5976 ),
5977 ])
5978 }
5979
5980 #[test]
5981 fn an_absent_replica_count_still_places_exactly_one_machine() {
5982 // The migration is additive: every mirror on disk omits `replicas`, and
5983 // must resolve byte-identically to the pre-R844-F8 answer.
5984 let cfg = three_runner_fleet();
5985 let req = RequiredSpec {
5986 mesh_tags: vec!["tag:cloud-runner".into()],
5987 ..Default::default()
5988 };
5989 assert_eq!(req.replica_count(), 1);
5990 let names: Vec<&str> = cfg
5991 .resolve_machines(&req)
5992 .unwrap()
5993 .iter()
5994 .map(|m| m.name.as_str())
5995 .collect();
5996 assert_eq!(names, vec!["us-east-001"]);
5997 assert_eq!(cfg.resolve_machine(&req).unwrap().name, "us-east-001");
5998 }
5999
6000 #[test]
6001 fn a_replica_count_places_that_many_machines_not_every_match() {
6002 // Three machines match; two are asked for; two are placed. Inferring the
6003 // count from the match count would make adding a box to the fleet
6004 // silently scale a production front door.
6005 let cfg = three_runner_fleet();
6006 let req = RequiredSpec {
6007 mesh_tags: vec!["tag:cloud-runner".into()],
6008 replicas: Some(2),
6009 ..Default::default()
6010 };
6011 let names: Vec<&str> = cfg
6012 .resolve_machines(&req)
6013 .unwrap()
6014 .iter()
6015 .map(|m| m.name.as_str())
6016 .collect();
6017 assert_eq!(names, vec!["us-east-001", "us-south-001"]);
6018 }
6019
6020 #[test]
6021 fn fewer_matches_than_replicas_is_an_error_naming_both_numbers() {
6022 // Never a partial placement: one of two reported as success is the
6023 // subset-that-looks-like-it-worked failure in its purest form.
6024 let cfg = three_runner_fleet();
6025 let req = RequiredSpec {
6026 regions: vec!["us-east".into()],
6027 replicas: Some(2),
6028 ..Default::default()
6029 };
6030 let err = cfg.resolve_machines(&req).unwrap_err().to_string();
6031 assert!(err.contains("only 1 of 2"), "got: {err}");
6032 assert!(err.contains("required.regions=[us-east]"), "got: {err}");
6033 // …and names the pool it searched, like every other placement refusal.
6034 assert!(err.contains("declared machines"), "got: {err}");
6035 assert!(err.contains("us-south-001"), "got: {err}");
6036 }
6037
6038 #[test]
6039 fn zero_replicas_is_refused_rather_than_placing_nothing() {
6040 let cfg = three_runner_fleet();
6041 let req = RequiredSpec {
6042 mesh_tags: vec!["tag:cloud-runner".into()],
6043 replicas: Some(0),
6044 ..Default::default()
6045 };
6046 let err = cfg.resolve_machines(&req).unwrap_err().to_string();
6047 assert!(err.contains("replicas = 0"), "got: {err}");
6048 }
6049
6050 #[test]
6051 fn replicas_parses_from_the_inline_required_form() {
6052 // The INLINE form specifically: `[providers.bundle.required]` as a table
6053 // HEADER ends the slot's table and reparents every key below it.
6054 let slot: MirrorProviderSlot = toml::from_str(
6055 r#"
6056use = "hetzner-primary"
6057port = 8080
6058required = { regions = ["us-east"], mesh_tags = ["tag:cloud-runner"], replicas = 2 }
6059"#,
6060 )
6061 .unwrap();
6062 assert_eq!(
6063 slot.fields().get("port").and_then(|v| v.as_integer()),
6064 Some(8080),
6065 "the inline form leaves the slot's other keys where they were"
6066 );
6067 let req = slot.required().expect("required block present");
6068 assert_eq!(req.replicas, Some(2));
6069 assert_eq!(req.replica_count(), 2);
6070 // A count is not a match axis — it says how many, not which.
6071 assert!(!req.is_unconstrained());
6072 assert!(RequiredSpec {
6073 replicas: Some(2),
6074 ..Default::default()
6075 }
6076 .is_unconstrained());
6077 }
6078
6079 #[test]
6080 fn round_trip_machine() {
6081 let cfg = MachineConfig {
6082 name: "test-pdx-1".into(),
6083 provider: "hetzner".into(),
6084 location: Some("pdx".into()),
6085 server_type: Some("cpx22".into()),
6086 hosts_mirrors: vec!["noisetable".into()],
6087 mesh_tags: vec!["region:pdx".into()],
6088 region: Some("us-west".into()),
6089 zone: Some("pdx".into()),
6090 arch: None,
6091 bucket: Some(BucketSpec {
6092 name: "test-assets-pdx-1".into(),
6093 public_read: false,
6094 }),
6095 vendor: None,
6096 nickname: None,
6097 legacy_hostkey_fingerprint: None,
6098 registration: Default::default(),
6099 ssh_keys: vec![],
6100 cloudflared: None,
6101 hosts_operator_bridge: false,
6102 connect: None,
6103 allocatable: None,
6104 taints: vec![],
6105 sovereign_group: None,
6106 sovereign_role: None,
6107 ingress_floating_ip: None,
6108 };
6109 let s = toml::to_string(&cfg).unwrap();
6110 let back: MachineConfig = toml::from_str(&s).unwrap();
6111 assert_eq!(back.name, cfg.name);
6112 assert_eq!(back.location, cfg.location);
6113 assert_eq!(back.region.as_deref(), Some("us-west"));
6114 assert_eq!(back.zone.as_deref(), Some("pdx"));
6115 }
6116
6117 #[test]
6118 fn round_trip_mirror() {
6119 let cfg = LegacyMirrorConfig {
6120 camp: "noisetable".into(),
6121 regions: vec!["pdx".into(), "iad".into()],
6122 workloads: vec!["asset-registry".into()],
6123 cloud_domain: None,
6124 };
6125 let s = toml::to_string(&cfg).unwrap();
6126 let back: LegacyMirrorConfig = toml::from_str(&s).unwrap();
6127 assert_eq!(back.camp, cfg.camp);
6128 assert_eq!(back.regions, cfg.regions);
6129 assert_eq!(back.workloads, cfg.workloads);
6130 }
6131
6132 #[test]
6133 fn mirror_serialises_as_camp_key() {
6134 // Serialised form should use `camp`, not `rig`.
6135 let cfg = LegacyMirrorConfig {
6136 camp: "noisetable".into(),
6137 regions: vec!["pdx".into()],
6138 workloads: vec![],
6139 cloud_domain: None,
6140 };
6141 let s = toml::to_string(&cfg).unwrap();
6142 assert!(
6143 s.contains("camp = "),
6144 "serialised key should be 'camp': {s}"
6145 );
6146 assert!(!s.contains("rig = "), "old key should not appear: {s}");
6147 }
6148
6149 #[test]
6150 fn mirror_rig_alias_still_loads() {
6151 // Old mirrors/*.toml files use `rig = "..."` before the R137 rename;
6152 // the alias keeps them loading until the one-time `sed` migration runs.
6153 let toml_str =
6154 "rig = \"noisetable\"\nregions = [\"pdx\"]\nworkloads = [\"asset-registry\"]\n";
6155 let cfg: LegacyMirrorConfig = toml::from_str(toml_str).unwrap();
6156 assert_eq!(cfg.camp, "noisetable");
6157 }
6158
6159 #[test]
6160 fn mirror_services_alias_still_loads() {
6161 // Old mirrors/*.toml files use `services = [...]`; the alias keeps them
6162 // loading without a migration step.
6163 let toml_str =
6164 "camp = \"noisetable\"\nregions = [\"pdx\"]\nservices = [\"asset-registry\"]\n";
6165 let cfg: LegacyMirrorConfig = toml::from_str(toml_str).unwrap();
6166 assert_eq!(cfg.workloads, vec!["asset-registry"]);
6167 }
6168
6169 #[test]
6170 fn round_trip_service_legacy() {
6171 let cfg = LegacyServiceConfig {
6172 name: "asset-registry".into(),
6173 image: "ghcr.io/noisetable/asset-registry".into(),
6174 version: "v1.0.0".into(),
6175 env: HashMap::new(),
6176 ports: vec![PortMapping {
6177 host: 8080,
6178 container: 8080,
6179 }],
6180 mesh_only: false,
6181 bind_interface: None,
6182 tenant: TenantId::singleton(),
6183 };
6184 let s = toml::to_string(&cfg).unwrap();
6185 let back: LegacyServiceConfig = toml::from_str(&s).unwrap();
6186 assert_eq!(back.name, cfg.name);
6187 assert_eq!(back.image, cfg.image);
6188 }
6189
6190 #[test]
6191 fn service_bind_interface_round_trips() {
6192 let cfg = LegacyServiceConfig {
6193 name: "postgres".into(),
6194 image: "postgres".into(),
6195 version: "16".into(),
6196 env: HashMap::new(),
6197 ports: vec![PortMapping {
6198 host: 5432,
6199 container: 5432,
6200 }],
6201 mesh_only: true,
6202 bind_interface: Some("tailscale0".into()),
6203 tenant: TenantId::singleton(),
6204 };
6205 let s = toml::to_string(&cfg).unwrap();
6206 let back: LegacyServiceConfig = toml::from_str(&s).unwrap();
6207 assert_eq!(back.bind_interface.as_deref(), Some("tailscale0"));
6208 }
6209
6210 #[test]
6211 fn service_bind_interface_absent_is_none() {
6212 let toml_str = "name = \"app\"\nimage = \"app\"\nversion = \"v1\"\n";
6213 let cfg: LegacyServiceConfig = toml::from_str(toml_str).unwrap();
6214 assert!(
6215 cfg.bind_interface.is_none(),
6216 "bind_interface should default to None"
6217 );
6218 }
6219
6220 #[test]
6221 fn service_bind_interface_skipped_when_none() {
6222 let cfg = LegacyServiceConfig {
6223 name: "app".into(),
6224 image: "app".into(),
6225 version: "v1".into(),
6226 env: HashMap::new(),
6227 ports: vec![],
6228 mesh_only: false,
6229 bind_interface: None,
6230 tenant: TenantId::singleton(),
6231 };
6232 let s = toml::to_string(&cfg).unwrap();
6233 assert!(!s.contains("bind_interface"), "None should be skipped: {s}");
6234 }
6235
6236 #[test]
6237 fn load_dir_missing_is_empty() {
6238 let dir = std::path::PathBuf::from("/nonexistent/path");
6239 let result: Vec<MachineConfig> = load_dir(dir).unwrap();
6240 assert!(result.is_empty());
6241 }
6242
6243 #[test]
6244 fn topology_round_trip() {
6245 let topo = TopologyConfig {
6246 assignments: vec![
6247 MirrorAssignment {
6248 mirror: "noisetable-pdx".into(),
6249 machine: "noisetable-pdx-1".into(),
6250 },
6251 MirrorAssignment {
6252 mirror: "noisetable-iad".into(),
6253 machine: "noisetable-iad-1".into(),
6254 },
6255 ],
6256 buckets: vec![],
6257 };
6258 let s = toml::to_string(&topo).unwrap();
6259 let back: TopologyConfig = toml::from_str(&s).unwrap();
6260 assert_eq!(back.assignments.len(), 2);
6261 assert_eq!(back.assignments[0].mirror, "noisetable-pdx");
6262 assert_eq!(back.assignments[1].machine, "noisetable-iad-1");
6263 }
6264
6265 #[test]
6266 fn topology_absent_returns_default() {
6267 let tmp = tempfile::TempDir::new().unwrap();
6268 let path = tmp.path().join("topology.toml");
6269 // file doesn't exist
6270 let topo = load_topology(path).unwrap();
6271 assert!(topo.assignments.is_empty());
6272 }
6273
6274 /// Helper: lay out a `<workspace_root>/.yah/cloud/` legacy tree for the
6275 /// pre-R215 cargo tests below; returns the legacy cloud_dir for writes.
6276 fn make_legacy_cloud_dir(root: &std::path::Path) -> std::path::PathBuf {
6277 let cloud_dir = root.join(".yah").join("cloud");
6278 std::fs::create_dir_all(&cloud_dir).unwrap();
6279 cloud_dir
6280 }
6281
6282 #[test]
6283 fn cloud_config_load_and_lookup() {
6284 let tmp = tempfile::TempDir::new().unwrap();
6285 let root = tmp.path();
6286 let cloud_dir = make_legacy_cloud_dir(root);
6287
6288 let machine = MachineConfig {
6289 name: "noisetable-pdx-1".into(),
6290 provider: "hetzner".into(),
6291 location: Some("pdx".into()),
6292 server_type: Some("cpx22".into()),
6293 hosts_mirrors: vec!["noisetable".into(), "yah".into()],
6294 mesh_tags: vec!["region:pdx".into(), "tier:t2".into()],
6295 region: None,
6296 zone: None,
6297 arch: None,
6298 bucket: Some(BucketSpec {
6299 name: "noisetable-assets-pdx-1".into(),
6300 public_read: false,
6301 }),
6302 vendor: None,
6303 nickname: None,
6304 legacy_hostkey_fingerprint: None,
6305 registration: Default::default(),
6306 ssh_keys: vec![],
6307 cloudflared: None,
6308 hosts_operator_bridge: false,
6309 connect: None,
6310 allocatable: None,
6311 taints: vec![],
6312 sovereign_group: None,
6313 sovereign_role: None,
6314 ingress_floating_ip: None,
6315 };
6316 // Land in the legacy tree so the legacy machine loader picks it up.
6317 machine.save(&cloud_dir).unwrap();
6318
6319 let mirror_toml = "camp = \"noisetable\"\nregions = [\"pdx\", \"iad\", \"fsn\"]\nworkloads = [\"asset-registry\"]\n";
6320 std::fs::create_dir_all(cloud_dir.join("mirrors")).unwrap();
6321 std::fs::write(cloud_dir.join("mirrors/noisetable.toml"), mirror_toml).unwrap();
6322
6323 // Legacy services/ dir (backward compat)
6324 let svc_toml = "name = \"asset-registry\"\nimage = \"ghcr.io/noisetable/asset-registry\"\nversion = \"v1.0.0\"\nmesh_only = false\n";
6325 std::fs::create_dir_all(cloud_dir.join("services")).unwrap();
6326 std::fs::write(cloud_dir.join("services/asset-registry.toml"), svc_toml).unwrap();
6327
6328 let cfg = CloudConfig::load(root).unwrap();
6329
6330 assert_eq!(cfg.machines.len(), 1);
6331 assert_eq!(cfg.legacy_mirrors.len(), 1);
6332 assert_eq!(cfg.legacy_services.len(), 1);
6333 assert_eq!(cfg.workloads.len(), 0); // no workloads/ dir yet
6334 assert!(cfg.services.is_empty(), "no R215+ services/ tree");
6335 assert!(cfg.providers.is_empty(), "no R215+ providers/ tree");
6336
6337 let m = cfg.machine("noisetable-pdx-1").unwrap();
6338 assert_eq!(m.location(), "pdx");
6339 assert_eq!(m.bucket.as_ref().unwrap().name, "noisetable-assets-pdx-1");
6340
6341 let mir = cfg.legacy_mirror("noisetable").unwrap();
6342 assert_eq!(mir.regions, vec!["pdx", "iad", "fsn"]);
6343 assert_eq!(mir.workloads, vec!["asset-registry"]);
6344 }
6345
6346 #[test]
6347 fn mirror_folder_layout_loads() {
6348 // Folder layout: mirrors/<id>/mirror.toml — new preferred form.
6349 let tmp = tempfile::TempDir::new().unwrap();
6350 let root = tmp.path();
6351 let cloud_dir = make_legacy_cloud_dir(root);
6352 let mirror_dir = cloud_dir.join("mirrors").join("yah-com");
6353 std::fs::create_dir_all(&mirror_dir).unwrap();
6354 std::fs::write(
6355 mirror_dir.join("mirror.toml"),
6356 "camp = \"yah\"\nregions = [\"pdx\"]\nworkloads = [\"yah-web\"]\n",
6357 )
6358 .unwrap();
6359
6360 let cfg = CloudConfig::load(root).unwrap();
6361 assert_eq!(cfg.legacy_mirrors.len(), 1);
6362 let mir = cfg.legacy_mirror("yah").unwrap();
6363 assert_eq!(mir.camp, "yah");
6364 assert_eq!(mir.workloads, vec!["yah-web"]);
6365 }
6366
6367 #[test]
6368 fn mirror_folder_and_flat_coexist() {
6369 // Both layouts may coexist in the same mirrors/ directory.
6370 let tmp = tempfile::TempDir::new().unwrap();
6371 let root = tmp.path();
6372 let cloud_dir = make_legacy_cloud_dir(root);
6373 let mirrors_root = cloud_dir.join("mirrors");
6374 std::fs::create_dir_all(&mirrors_root).unwrap();
6375
6376 // Flat legacy mirror
6377 std::fs::write(
6378 mirrors_root.join("noisetable.toml"),
6379 "camp = \"noisetable\"\nregions = [\"pdx\"]\nworkloads = []\n",
6380 )
6381 .unwrap();
6382
6383 // Folder-form mirror
6384 let yah_com_dir = mirrors_root.join("yah-com");
6385 std::fs::create_dir_all(&yah_com_dir).unwrap();
6386 std::fs::write(
6387 yah_com_dir.join("mirror.toml"),
6388 "camp = \"yah\"\nregions = [\"pdx\"]\nworkloads = []\n",
6389 )
6390 .unwrap();
6391
6392 let cfg = CloudConfig::load(root).unwrap();
6393 assert_eq!(cfg.legacy_mirrors.len(), 2);
6394 assert!(cfg.legacy_mirror("noisetable").is_some());
6395 assert!(cfg.legacy_mirror("yah").is_some());
6396 }
6397
6398 #[test]
6399 fn mirror_malformed_fails_with_field_path() {
6400 // A malformed mirror.toml should fail at load with a clear error
6401 // that includes the file path.
6402 let tmp = tempfile::TempDir::new().unwrap();
6403 let root = tmp.path();
6404 let cloud_dir = make_legacy_cloud_dir(root);
6405 let mirror_dir = cloud_dir.join("mirrors").join("bad");
6406 std::fs::create_dir_all(&mirror_dir).unwrap();
6407 // Missing required `camp` field
6408 std::fs::write(
6409 mirror_dir.join("mirror.toml"),
6410 "regions = [\"pdx\"]\nworkloads = []\n",
6411 )
6412 .unwrap();
6413
6414 let err = CloudConfig::load(root).unwrap_err();
6415 let msg = err.to_string();
6416 assert!(
6417 msg.contains("mirror.toml"),
6418 "error should reference the file path, got: {msg}"
6419 );
6420 }
6421
6422 #[test]
6423 fn workload_config_load_and_validate() {
6424 use workload_spec::{
6425 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
6426 RestartPolicy, SchemaVersion, StopPolicy, TenantId, TierTag, WorkloadSpec,
6427 };
6428
6429 let tmp = tempfile::TempDir::new().unwrap();
6430 let root = tmp.path();
6431 let cloud_dir = make_legacy_cloud_dir(root);
6432 std::fs::create_dir_all(cloud_dir.join("workloads")).unwrap();
6433
6434 let spec = WorkloadSpec {
6435 schema_version: SchemaVersion::V1,
6436 name: "asset-registry".into(),
6437 image: ImageRef {
6438 registry: "ghcr.io".into(),
6439 repository: "noisetable/asset-registry".into(),
6440 tag: "v1.0.0".into(),
6441 digest: workload_spec::testing::test_digest(),
6442 },
6443 tier: TierTag("tenant".into()),
6444 replicas: 1,
6445 command: None,
6446 entrypoint: None,
6447 workdir: None,
6448 user: None,
6449 env: vec![],
6450 secrets: vec![],
6451 volumes: vec![],
6452 resources: ResourceLimits {
6453 memory_mb: 256,
6454 cpu_millis: 512,
6455 ephemeral_storage_mb: 512,
6456 },
6457 depends_on: vec![],
6458 requires: vec![],
6459 healthcheck: None,
6460 restart_policy: RestartPolicy::Always,
6461 archetype: None,
6462 stop_policy: StopPolicy {
6463 signal: 15,
6464 grace_period: workload_spec::Millis::from_secs(10),
6465 },
6466 expose: ExposeSpec {
6467 mesh: MeshExpose {
6468 identity: MeshIdent("asset-registry.pdx".into()),
6469 ports: MeshExpose::anonymous_ports([8080]),
6470 allow_from: vec![],
6471 },
6472 public: None,
6473 operator: None,
6474 },
6475 tenant: TenantId::singleton(),
6476 namespace: NamespaceId::singleton(),
6477 labels: Default::default(),
6478 annotations: Default::default(),
6479 };
6480
6481 let toml_str = toml::to_string_pretty(&spec).unwrap();
6482 std::fs::write(cloud_dir.join("workloads/asset-registry.toml"), &toml_str).unwrap();
6483
6484 let cfg = CloudConfig::load(root).unwrap();
6485 assert_eq!(cfg.workloads.len(), 1);
6486 assert_eq!(cfg.workloads[0].spec.name, "asset-registry");
6487 assert_eq!(cfg.workload("asset-registry").unwrap().spec.replicas, 1);
6488 }
6489
6490 /// Minimal valid spec for the R215+ loader tests below. Kept as a helper so
6491 /// the two tests differ only in *where* the file lands, which is the whole
6492 /// thing under test.
6493 #[cfg(test)]
6494 fn minimal_spec(name: &str, replicas: u32) -> workload_spec::WorkloadSpec {
6495 use workload_spec::{
6496 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
6497 RestartPolicy, SchemaVersion, StopPolicy, TenantId, TierTag, WorkloadSpec,
6498 };
6499 WorkloadSpec {
6500 schema_version: SchemaVersion::V1,
6501 name: name.into(),
6502 image: ImageRef {
6503 registry: "cr.yah.dev".into(),
6504 repository: name.into(),
6505 tag: "v1".into(),
6506 digest: workload_spec::testing::test_digest(),
6507 },
6508 tier: TierTag("infra".into()),
6509 replicas,
6510 command: None,
6511 entrypoint: None,
6512 workdir: None,
6513 user: None,
6514 env: vec![],
6515 secrets: vec![],
6516 volumes: vec![],
6517 resources: ResourceLimits {
6518 memory_mb: 256,
6519 cpu_millis: 250,
6520 ephemeral_storage_mb: 128,
6521 },
6522 depends_on: vec![],
6523 requires: vec![],
6524 healthcheck: None,
6525 restart_policy: RestartPolicy::Always,
6526 archetype: None,
6527 stop_policy: StopPolicy {
6528 signal: 15,
6529 grace_period: workload_spec::Millis::from_secs(10),
6530 },
6531 expose: ExposeSpec {
6532 mesh: MeshExpose {
6533 identity: MeshIdent(name.into()),
6534 ports: MeshExpose::anonymous_ports([4325]),
6535 allow_from: vec![],
6536 },
6537 public: None,
6538 operator: None,
6539 },
6540 tenant: TenantId::singleton(),
6541 namespace: NamespaceId::singleton(),
6542 labels: Default::default(),
6543 annotations: Default::default(),
6544 }
6545 }
6546
6547 /// R568-T7. Workloads must load from the R215+ tree.
6548 ///
6549 /// Before the fix this function tested, `CloudConfig::load` read workloads
6550 /// ONLY from the pre-R215 `.yah/cloud/workloads/` — which R222-B1 emptied —
6551 /// so in any modern camp `cfg.workload(name)` returned `None` for every
6552 /// name and the entire `yah cloud workload …` surface was unreachable. The
6553 /// CLI's own error text has said `.yah/infra/workloads/` throughout, so the
6554 /// bug read as "you must have typoed the filename".
6555 ///
6556 /// Note the fixture writes NO legacy `.yah/cloud/` dir at all: that is the
6557 /// shape of a real post-R215 camp, and it is exactly the shape the old code
6558 /// could not serve.
6559 #[test]
6560 fn workloads_load_from_the_infra_tree() {
6561 let tmp = tempfile::TempDir::new().unwrap();
6562 let root = tmp.path();
6563 let dir = crate::paths::workloads_dir(root);
6564 std::fs::create_dir_all(&dir).unwrap();
6565 std::fs::write(
6566 dir.join("yah-cloud-admin.toml"),
6567 toml::to_string_pretty(&minimal_spec("yah-cloud-admin", 1)).unwrap(),
6568 )
6569 .unwrap();
6570
6571 let cfg = CloudConfig::load(root).unwrap();
6572 assert_eq!(cfg.workloads.len(), 1);
6573 assert_eq!(
6574 cfg.workload("yah-cloud-admin").unwrap().spec.replicas,
6575 1,
6576 "a workload declared under .yah/infra/workloads/ must be resolvable by name"
6577 );
6578 }
6579
6580 /// A camp mid-migration can have both trees. R215+ wins on a name
6581 /// collision — same precedence the machine loader applies — so moving a
6582 /// declaration into `.yah/infra/workloads/` takes effect immediately
6583 /// instead of being silently shadowed by the copy left behind.
6584 #[test]
6585 fn infra_workload_shadows_the_legacy_copy_of_the_same_name() {
6586 let tmp = tempfile::TempDir::new().unwrap();
6587 let root = tmp.path();
6588
6589 let legacy = make_legacy_cloud_dir(root);
6590 std::fs::create_dir_all(legacy.join("workloads")).unwrap();
6591 std::fs::write(
6592 legacy.join("workloads/shared.toml"),
6593 toml::to_string_pretty(&minimal_spec("shared", 9)).unwrap(),
6594 )
6595 .unwrap();
6596 // Legacy-only name, to prove the old tree is still read rather than
6597 // replaced wholesale.
6598 std::fs::write(
6599 legacy.join("workloads/legacy-only.toml"),
6600 toml::to_string_pretty(&minimal_spec("legacy-only", 3)).unwrap(),
6601 )
6602 .unwrap();
6603
6604 let infra = crate::paths::workloads_dir(root);
6605 std::fs::create_dir_all(&infra).unwrap();
6606 std::fs::write(
6607 infra.join("shared.toml"),
6608 toml::to_string_pretty(&minimal_spec("shared", 1)).unwrap(),
6609 )
6610 .unwrap();
6611
6612 let cfg = CloudConfig::load(root).unwrap();
6613 assert_eq!(cfg.workloads.len(), 2, "one `shared`, plus `legacy-only`");
6614 assert_eq!(
6615 cfg.workload("shared").unwrap().spec.replicas,
6616 1,
6617 "the .yah/infra/ copy must win over the legacy one"
6618 );
6619 assert_eq!(cfg.workload("legacy-only").unwrap().spec.replicas, 3);
6620 }
6621
6622 #[test]
6623 fn workload_loader_rejects_bad_spec() {
6624 use workload_spec::{
6625 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
6626 RestartPolicy, SchemaVersion, StopPolicy, TenantId, TierTag, WorkloadSpec,
6627 };
6628
6629 let tmp = tempfile::TempDir::new().unwrap();
6630 let root = tmp.path();
6631 let cloud_dir = make_legacy_cloud_dir(root);
6632 std::fs::create_dir_all(cloud_dir.join("workloads")).unwrap();
6633
6634 // Construct a spec that round-trips through TOML but fails shape
6635 // validation: replicas = 200 is above the max of 100.
6636 let mut spec = WorkloadSpec {
6637 schema_version: SchemaVersion::V1,
6638 name: "asset-registry".into(),
6639 image: ImageRef {
6640 registry: "ghcr.io".into(),
6641 repository: "test/app".into(),
6642 tag: "v1".into(),
6643 digest: workload_spec::testing::test_digest(),
6644 },
6645 tier: TierTag("tenant".into()),
6646 replicas: 200, // ← invalid: exceeds max 100
6647 command: None,
6648 entrypoint: None,
6649 workdir: None,
6650 user: None,
6651 env: vec![],
6652 secrets: vec![],
6653 volumes: vec![],
6654 resources: ResourceLimits {
6655 memory_mb: 256,
6656 cpu_millis: 512,
6657 ephemeral_storage_mb: 512,
6658 },
6659 depends_on: vec![],
6660 requires: vec![],
6661 healthcheck: None,
6662 restart_policy: RestartPolicy::Always,
6663 archetype: None,
6664 stop_policy: StopPolicy {
6665 signal: 15,
6666 grace_period: workload_spec::Millis::from_secs(10),
6667 },
6668 expose: ExposeSpec {
6669 mesh: MeshExpose {
6670 identity: MeshIdent("asset-registry.pdx".into()),
6671 ports: MeshExpose::anonymous_ports([8080]),
6672 allow_from: vec![],
6673 },
6674 public: None,
6675 operator: None,
6676 },
6677 tenant: TenantId::singleton(),
6678 namespace: NamespaceId::singleton(),
6679 labels: Default::default(),
6680 annotations: Default::default(),
6681 };
6682
6683 let toml_str = toml::to_string_pretty(&spec).unwrap();
6684 std::fs::write(cloud_dir.join("workloads/bad.toml"), &toml_str).unwrap();
6685
6686 let result = CloudConfig::load(root);
6687 assert!(
6688 result.is_err(),
6689 "loading a WorkloadSpec with replicas=200 should return Err"
6690 );
6691 let msg = result.unwrap_err().to_string();
6692 assert!(
6693 msg.contains("shape validation")
6694 || msg.contains("Replicas")
6695 || msg.contains("replicas"),
6696 "error should mention shape validation or replicas field, got: {msg}"
6697 );
6698
6699 // The `spec` binding is only used for the write — suppress warning.
6700 let _ = &mut spec;
6701 }
6702
6703 #[test]
6704 fn workload_config_save_round_trip() {
6705 use workload_spec::{
6706 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
6707 RestartPolicy, SchemaVersion, StopPolicy, TenantId, TierTag, WorkloadSpec,
6708 };
6709
6710 let tmp = tempfile::TempDir::new().unwrap();
6711 let root = tmp.path();
6712
6713 let spec = WorkloadSpec {
6714 schema_version: SchemaVersion::V1,
6715 name: "signing-service".into(),
6716 image: ImageRef {
6717 registry: "ghcr.io".into(),
6718 repository: "noisetable/signing".into(),
6719 tag: "v2.0.0".into(),
6720 digest: workload_spec::testing::test_digest(),
6721 },
6722 tier: TierTag("private".into()),
6723 replicas: 2,
6724 command: None,
6725 entrypoint: None,
6726 workdir: None,
6727 user: None,
6728 env: vec![],
6729 secrets: vec![],
6730 volumes: vec![],
6731 resources: ResourceLimits {
6732 memory_mb: 128,
6733 cpu_millis: 256,
6734 ephemeral_storage_mb: 256,
6735 },
6736 depends_on: vec![],
6737 requires: vec![],
6738 healthcheck: None,
6739 restart_policy: RestartPolicy::Always,
6740 archetype: None,
6741 stop_policy: StopPolicy {
6742 signal: 15,
6743 grace_period: workload_spec::Millis::from_secs(5),
6744 },
6745 expose: ExposeSpec {
6746 mesh: MeshExpose {
6747 identity: MeshIdent("signing.pdx".into()),
6748 ports: MeshExpose::anonymous_ports([9090]),
6749 allow_from: vec![],
6750 },
6751 public: None,
6752 operator: None,
6753 },
6754 tenant: TenantId::singleton(),
6755 namespace: NamespaceId::singleton(),
6756 labels: Default::default(),
6757 annotations: Default::default(),
6758 };
6759
6760 let wc = WorkloadConfig { spec };
6761 let cloud_dir = make_legacy_cloud_dir(root);
6762 wc.save(&cloud_dir).unwrap();
6763
6764 let loaded = CloudConfig::load(root).unwrap();
6765 assert_eq!(loaded.workloads.len(), 1);
6766 assert_eq!(loaded.workloads[0].spec.name, "signing-service");
6767 assert_eq!(loaded.workloads[0].spec.replicas, 2);
6768 }
6769
6770 #[test]
6771 fn machine_save_write_back_fingerprint() {
6772 let tmp = tempfile::TempDir::new().unwrap();
6773 let root = tmp.path();
6774
6775 let mut machine = MachineConfig {
6776 name: "test-pdx-1".into(),
6777 provider: "hetzner".into(),
6778 location: Some("pdx".into()),
6779 server_type: Some("cpx22".into()),
6780 hosts_mirrors: vec![],
6781 mesh_tags: vec![],
6782 region: None,
6783 zone: None,
6784 arch: None,
6785 bucket: None,
6786 vendor: None,
6787 nickname: None,
6788 legacy_hostkey_fingerprint: None,
6789 registration: Default::default(),
6790 ssh_keys: vec![],
6791 cloudflared: None,
6792 hosts_operator_bridge: false,
6793 connect: None,
6794 allocatable: None,
6795 taints: vec![],
6796 sovereign_group: None,
6797 sovereign_role: None,
6798 ingress_floating_ip: None,
6799 };
6800 machine.save(root).unwrap();
6801
6802 // Simulate A4: write back the hostkey fingerprint after provision.
6803 // R707-T1: registration is the write target; the accessor is the read.
6804 machine.registration.hostkey_fingerprint = Some("SHA256:abc123".into());
6805 machine.save(root).unwrap();
6806
6807 let reloaded: Vec<MachineConfig> = load_dir(root.join("machines")).unwrap();
6808 assert_eq!(reloaded.len(), 1);
6809 assert_eq!(reloaded[0].hostkey_fingerprint(), Some("SHA256:abc123"));
6810 }
6811
6812 // ─── New-shape (R222 B2) parse tests ────────────────────────────────────
6813 //
6814 // These mirror the Phase-A manifests committed under `.yah/services/` and
6815 // `.yah/infra/providers/`. Keeping the test strings inline (rather than
6816 // reading the on-disk files) so the loader stays runnable in any workdir
6817 // and so accidental edits to the on-disk files don't silently change
6818 // schema expectations.
6819
6820 #[test]
6821 fn provider_cloudflare_round_trips() {
6822 let src = r#"
6823schema_version = 1
6824id = "cloudflare"
6825kind = "cloudflare"
6826credentials = "keystore://cloudflare/yah"
6827default_zone = "yah.dev"
6828"#;
6829 let cfg: ProviderConfig = toml::from_str(src).unwrap();
6830 assert_eq!(cfg.id, "cloudflare");
6831 assert_eq!(cfg.kind, Provider::Cloudflare);
6832 assert_eq!(
6833 cfg.credentials.as_deref(),
6834 Some("keystore://cloudflare/yah")
6835 );
6836 assert_eq!(
6837 cfg.fields.get("default_zone").and_then(|v| v.as_str()),
6838 Some("yah.dev"),
6839 );
6840 let back = toml::to_string(&cfg).unwrap();
6841 let again: ProviderConfig = toml::from_str(&back).unwrap();
6842 assert_eq!(again.id, cfg.id);
6843 assert_eq!(again.kind, cfg.kind);
6844 }
6845
6846 #[test]
6847 fn provider_hetzner_round_trips() {
6848 let src = r#"
6849schema_version = 1
6850id = "hetzner"
6851kind = "hetzner"
6852credentials = "keystore://hetzner/yah"
6853default_location = "pdx"
6854default_server_type = "cpx11"
6855ssh_keys = []
6856"#;
6857 let cfg: ProviderConfig = toml::from_str(src).unwrap();
6858 assert_eq!(cfg.kind, Provider::Hetzner);
6859 assert_eq!(
6860 cfg.fields.get("default_location").and_then(|v| v.as_str()),
6861 Some("pdx"),
6862 );
6863 assert!(
6864 cfg.fields
6865 .get("ssh_keys")
6866 .map(|v| v.as_array().unwrap().is_empty())
6867 .unwrap_or(false),
6868 "ssh_keys must round-trip as empty array, got {:?}",
6869 cfg.fields.get("ssh_keys"),
6870 );
6871 }
6872
6873 #[test]
6874 fn provider_orbstack_local_container_round_trips() {
6875 let src = r#"
6876schema_version = 1
6877id = "orbstack"
6878kind = "local-container"
6879runtime = "auto"
6880
6881[discovery]
6882orbstack = "~/.orbstack/run/docker.sock"
6883colima = "~/.colima/default/docker.sock"
6884docker = "/var/run/docker.sock"
6885"#;
6886 let cfg: ProviderConfig = toml::from_str(src).unwrap();
6887 assert_eq!(cfg.kind, Provider::LocalContainer);
6888 assert_eq!(
6889 cfg.fields.get("runtime").and_then(|v| v.as_str()),
6890 Some("auto"),
6891 );
6892 let discovery = cfg
6893 .fields
6894 .get("discovery")
6895 .and_then(|v| v.as_table())
6896 .expect("discovery table");
6897 assert!(discovery.contains_key("orbstack"));
6898 assert!(discovery.contains_key("colima"));
6899 assert!(discovery.contains_key("docker"));
6900 }
6901
6902 #[test]
6903 fn provider_unknown_kind_fails() {
6904 let src = r#"
6905schema_version = 1
6906id = "made-up"
6907kind = "fly-io"
6908"#;
6909 let err = toml::from_str::<ProviderConfig>(src).unwrap_err();
6910 let msg = err.to_string();
6911 assert!(
6912 msg.contains("kind") || msg.contains("variant"),
6913 "unknown provider kind should surface as a serde error, got: {msg}"
6914 );
6915 }
6916
6917 #[test]
6918 fn service_dev_yah_round_trips() {
6919 let src = r#"
6920schema_version = 1
6921name = "dev-yah"
6922domain = "yah.dev"
6923
6924[[components]]
6925id = "site"
6926kind = "mesofact-static"
6927path = "app/yah/web"
6928role = "static"
6929"#;
6930 let cfg: ServiceConfig = toml::from_str(src).unwrap();
6931 assert_eq!(cfg.name, "dev-yah");
6932 assert_eq!(cfg.domain, "yah.dev");
6933 assert_eq!(cfg.components.len(), 1);
6934 let c = &cfg.components[0];
6935 assert_eq!(c.id, "site");
6936 assert_eq!(c.kind, "mesofact-static");
6937 assert_eq!(c.path, "app/yah/web");
6938 assert_eq!(c.role, "static");
6939 assert!(c.publishes.is_none());
6940
6941 let back = toml::to_string(&cfg).unwrap();
6942 let again: ServiceConfig = toml::from_str(&back).unwrap();
6943 assert_eq!(again.name, cfg.name);
6944 assert_eq!(again.components[0].kind, c.kind);
6945 }
6946
6947 #[test]
6948 fn mirror_prod_cloudflare_reference_parses() {
6949 let src = r#"
6950schema_version = 1
6951shape = "single-machine"
6952
6953[providers.static]
6954use = "cloudflare"
6955bucket = "yah-dev"
6956zone = "yah.dev"
6957dns = { record = "@", type = "CNAME" }
6958"#;
6959 let cfg: MirrorConfig = toml::from_str(src).unwrap();
6960 assert_eq!(cfg.shape, MirrorShape::SingleMachine);
6961 let slot = cfg.providers.get("static").expect("static slot");
6962 assert_eq!(slot.provider_id(), Some("cloudflare"));
6963 assert!(slot.inline_kind().is_none());
6964 if let MirrorProviderSlot::Reference { fields, .. } = slot {
6965 assert_eq!(
6966 fields.get("bucket").and_then(|v| v.as_str()),
6967 Some("yah-dev")
6968 );
6969 assert_eq!(fields.get("zone").and_then(|v| v.as_str()), Some("yah.dev"));
6970 let dns = fields
6971 .get("dns")
6972 .and_then(|v| v.as_table())
6973 .expect("dns table");
6974 assert_eq!(dns.get("record").and_then(|v| v.as_str()), Some("@"));
6975 assert_eq!(dns.get("type").and_then(|v| v.as_str()), Some("CNAME"));
6976 } else {
6977 panic!("expected Reference slot");
6978 }
6979 }
6980
6981 #[test]
6982 fn mirror_local_inline_static_and_orbstack_compute_parse() {
6983 let src = r#"
6984schema_version = 1
6985shape = "local"
6986
6987[providers.static]
6988kind = "local-static"
6989port = 4321
6990artifact_dir = ".yah/infra/state/local/static"
6991
6992[providers.compute]
6993use = "orbstack"
6994"#;
6995 let cfg: MirrorConfig = toml::from_str(src).unwrap();
6996 assert_eq!(cfg.shape, MirrorShape::Local);
6997
6998 let static_slot = cfg.providers.get("static").expect("static slot");
6999 assert_eq!(static_slot.inline_kind(), Some(Provider::LocalStatic));
7000 assert!(static_slot.provider_id().is_none());
7001 if let MirrorProviderSlot::Inline { fields, .. } = static_slot {
7002 assert_eq!(fields.get("port").and_then(|v| v.as_integer()), Some(4321));
7003 assert_eq!(
7004 fields.get("artifact_dir").and_then(|v| v.as_str()),
7005 Some(".yah/infra/state/local/static"),
7006 );
7007 } else {
7008 panic!("expected Inline slot for static");
7009 }
7010
7011 let compute_slot = cfg.providers.get("compute").expect("compute slot");
7012 assert_eq!(compute_slot.provider_id(), Some("orbstack"));
7013 }
7014
7015 #[test]
7016 fn mirror_pond_miniflare_minio_parse() {
7017 // pond-tier mirror: miniflare-container + minio, both inline.
7018 // T1 just needs these inline kinds to parse — the reconciler dispatch
7019 // arrives in R256-T3.
7020 let src = r#"
7021schema_version = 1
7022shape = "local"
7023
7024[providers.static]
7025kind = "miniflare-container"
7026port = 4322
7027bucket = "yah-dev"
7028
7029[providers.object_store]
7030kind = "minio-container"
7031api_port = 9000
7032console_port = 9001
7033bucket = "yah-dev"
7034"#;
7035 let cfg: MirrorConfig = toml::from_str(src).unwrap();
7036 assert_eq!(cfg.shape, MirrorShape::Local);
7037
7038 let static_slot = cfg.providers.get("static").expect("static slot");
7039 assert_eq!(
7040 static_slot.inline_kind(),
7041 Some(Provider::MiniflareContainer)
7042 );
7043 if let MirrorProviderSlot::Inline { fields, .. } = static_slot {
7044 assert_eq!(fields.get("port").and_then(|v| v.as_integer()), Some(4322));
7045 assert_eq!(
7046 fields.get("bucket").and_then(|v| v.as_str()),
7047 Some("yah-dev")
7048 );
7049 } else {
7050 panic!("expected Inline slot for miniflare-container static");
7051 }
7052
7053 let object_store_slot = cfg
7054 .providers
7055 .get("object_store")
7056 .expect("object_store slot");
7057 assert_eq!(
7058 object_store_slot.inline_kind(),
7059 Some(Provider::MinioContainer)
7060 );
7061 if let MirrorProviderSlot::Inline { fields, .. } = object_store_slot {
7062 assert_eq!(
7063 fields.get("api_port").and_then(|v| v.as_integer()),
7064 Some(9000)
7065 );
7066 assert_eq!(
7067 fields.get("console_port").and_then(|v| v.as_integer()),
7068 Some(9001)
7069 );
7070 assert_eq!(
7071 fields.get("bucket").and_then(|v| v.as_str()),
7072 Some("yah-dev")
7073 );
7074 } else {
7075 panic!("expected Inline slot for minio-container object_store");
7076 }
7077 }
7078
7079 #[test]
7080 fn provider_miniflare_container_kind_round_trips() {
7081 // Inline-only kind; never declared as a standalone provider file but
7082 // the enum round-trip is still exercised through ProviderConfig because
7083 // schemars/serde share the variant table.
7084 let cfg = MirrorProviderSlot::Inline {
7085 kind: Provider::MiniflareContainer,
7086 fields: BTreeMap::new(),
7087 };
7088 let s = toml::to_string(&cfg).unwrap();
7089 assert!(
7090 s.contains("kind = \"miniflare-container\""),
7091 "kebab-case wire form expected, got: {s}"
7092 );
7093 let back: MirrorProviderSlot = toml::from_str(&s).unwrap();
7094 assert_eq!(back.inline_kind(), Some(Provider::MiniflareContainer));
7095 }
7096
7097 #[test]
7098 fn provider_minio_container_kind_round_trips() {
7099 let cfg = MirrorProviderSlot::Inline {
7100 kind: Provider::MinioContainer,
7101 fields: BTreeMap::new(),
7102 };
7103 let s = toml::to_string(&cfg).unwrap();
7104 assert!(
7105 s.contains("kind = \"minio-container\""),
7106 "kebab-case wire form expected, got: {s}"
7107 );
7108 let back: MirrorProviderSlot = toml::from_str(&s).unwrap();
7109 assert_eq!(back.inline_kind(), Some(Provider::MinioContainer));
7110 }
7111
7112 #[test]
7113 fn mirror_compute_slot_with_machine_reference_parses() {
7114 // The on-disk prod.toml has a commented-out compute slot; this test
7115 // covers the form Phase B will need once yubaba is provisioned.
7116 let src = r#"
7117schema_version = 1
7118shape = "single-machine"
7119
7120[providers.compute]
7121use = "hetzner"
7122machine = "yah-cloud-1"
7123"#;
7124 let cfg: MirrorConfig = toml::from_str(src).unwrap();
7125 let slot = cfg.providers.get("compute").expect("compute slot");
7126 assert_eq!(slot.provider_id(), Some("hetzner"));
7127 if let MirrorProviderSlot::Reference { fields, .. } = slot {
7128 assert_eq!(
7129 fields.get("machine").and_then(|v| v.as_str()),
7130 Some("yah-cloud-1"),
7131 );
7132 }
7133 }
7134
7135 #[test]
7136 fn machine_yah_cloud_1_round_trips_with_existing_shape() {
7137 // The current machine TOML predates B2 — MachineConfig hasn't been
7138 // reshaped yet. This locks the expected shape so we notice if B3
7139 // accidentally regresses it.
7140 let src = r#"
7141name = "yah-cloud-1"
7142provider = "hetzner"
7143location = "pdx"
7144server_type = "cpx11"
7145hosts_mirrors = []
7146mesh_tags = ["tag:tier-scratch", "tag:primary-yah"]
7147ssh_keys = [111513970, 111525493]
7148"#;
7149 let cfg: MachineConfig = toml::from_str(src).unwrap();
7150 assert_eq!(cfg.name, "yah-cloud-1");
7151 assert_eq!(cfg.provider, "hetzner");
7152 assert_eq!(cfg.ssh_keys.len(), 2);
7153 }
7154
7155 #[test]
7156 fn static_node_omits_location_server_type_and_carries_connect() {
7157 // BYO Phase-0: a `static` node we brought up over SSH has no provider
7158 // DC code or SKU; it declares reach in `[connect]` instead. Must load.
7159 let src = r#"
7160name = "us-south-001"
7161provider = "static"
7162region = "us-south"
7163mesh_tags = ["tag:cloud-runner", "tag:voter-candidate"]
7164
7165[connect]
7166address = "45.32.194.254"
7167ssh = "root@45.32.194.254"
7168identity_file = "~/.ssh/yah"
7169yubaba = "http://127.0.0.1:7443"
7170arch = "x86_64"
7171"#;
7172 let cfg: MachineConfig = toml::from_str(src).unwrap();
7173 assert_eq!(cfg.provider, "static");
7174 assert!(cfg.location.is_none());
7175 assert!(cfg.server_type.is_none());
7176 assert_eq!(cfg.location(), ""); // accessor defaults empty
7177 let c = cfg.connect.as_ref().expect("connect block");
7178 assert_eq!(c.ssh, "root@45.32.194.254");
7179 // Loopback is a *declared* reach placeholder, so it stays in [connect]
7180 // verbatim and composes straight through (R707-T1).
7181 assert_eq!(c.yubaba.as_deref(), Some("http://127.0.0.1:7443"));
7182 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://127.0.0.1:7443"));
7183 assert_eq!(cfg.mesh_ipv4(), None);
7184 // Static providers have no driver, so validate() is a no-op pass.
7185 assert!(!provider_has_machine_driver(&cfg.provider));
7186 cfg.validate().unwrap();
7187 }
7188
7189 // ─── R707-T1: declaration / registration split ──────────────────────────
7190
7191 /// The pre-split shape — top-level `hostkey_fingerprint`, mesh IP baked
7192 /// into `[connect].yubaba` — must keep parsing, and must read back through
7193 /// the accessors identically. Every machine TOML in the fleet was written
7194 /// this way, and other camps' inventories still are.
7195 #[test]
7196 fn legacy_shape_still_parses_and_reads_through_accessors() {
7197 let src = r#"
7198name = "us-west-001"
7199provider = "static"
7200region = "us-west"
7201arch = "x86_64"
7202mesh_tags = ["tag:cloud-runner"]
7203hostkey_fingerprint = "SHA256:dmpq"
7204
7205[connect]
7206address = "15.204.89.240"
7207ssh = "debian@15.204.89.240"
7208identity_file = "~/.ssh/yah"
7209yubaba = "http://100.64.0.1:7443"
7210"#;
7211 let cfg: MachineConfig = toml::from_str(src).unwrap();
7212 assert_eq!(cfg.hostkey_fingerprint(), Some("SHA256:dmpq"));
7213 assert_eq!(cfg.mesh_ipv4(), Some("100.64.0.1"));
7214 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://100.64.0.1:7443"));
7215 }
7216
7217 /// The post-split shape reads identically to the legacy one above — same
7218 /// three accessor answers from a file that separates the two halves. This
7219 /// is the "unchanged in meaning" guarantee the fleet migration rests on.
7220 #[test]
7221 fn split_shape_is_equivalent_to_legacy_shape() {
7222 let legacy = r#"
7223name = "m"
7224provider = "static"
7225mesh_tags = []
7226hostkey_fingerprint = "SHA256:dmpq"
7227
7228[connect]
7229address = "15.204.89.240"
7230ssh = "debian@15.204.89.240"
7231identity_file = "~/.ssh/yah"
7232yubaba = "http://100.64.0.1:7443"
7233"#;
7234 let split = r#"
7235name = "m"
7236provider = "static"
7237mesh_tags = []
7238
7239[connect]
7240address = "15.204.89.240"
7241ssh = "debian@15.204.89.240"
7242identity_file = "~/.ssh/yah"
7243
7244[registration]
7245hostkey_fingerprint = "SHA256:dmpq"
7246mesh_ipv4 = "100.64.0.1"
7247"#;
7248 let old: MachineConfig = toml::from_str(legacy).unwrap();
7249 let new: MachineConfig = toml::from_str(split).unwrap();
7250 assert_eq!(old.hostkey_fingerprint(), new.hostkey_fingerprint());
7251 assert_eq!(old.mesh_ipv4(), new.mesh_ipv4());
7252 assert_eq!(old.yubaba_url(), new.yubaba_url());
7253 }
7254
7255 /// A non-default `[connect].yubaba_port` is declared reach and composes
7256 /// with the observed mesh address rather than being pinned into a URL.
7257 #[test]
7258 fn declared_port_composes_with_observed_mesh_address() {
7259 let src = r#"
7260name = "m"
7261provider = "static"
7262mesh_tags = []
7263
7264[connect]
7265address = "10.0.0.1"
7266ssh = "yah@10.0.0.1"
7267identity_file = "~/.ssh/yah"
7268yubaba_port = 9443
7269
7270[registration]
7271mesh_ipv4 = "100.64.0.9"
7272"#;
7273 let cfg: MachineConfig = toml::from_str(src).unwrap();
7274 assert_eq!(cfg.connect.as_ref().unwrap().yubaba_port(), 9443);
7275 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://100.64.0.9:9443"));
7276 }
7277
7278 /// R605-T10 inverts R707-T6 for the private-literal case, and this is the
7279 /// node it was inverted for: us-west-014's shape, mesh-joined AND declaring
7280 /// a LAN `[connect].yubaba`. R707-T6 made the literal win outright so
7281 /// `rollout::yubaba::membership_to_nodes` could match the dev group's
7282 /// LAN-addressed raft membership — which fused identity into reach and made
7283 /// every automated dial go to an address only bldg-2506 can route.
7284 /// `lan_endpoint()` now serves that match, so the mesh address wins the
7285 /// dial and the literal is inert.
7286 #[test]
7287 fn a_private_literal_loses_to_the_registered_mesh_address() {
7288 let src = r#"
7289name = "us-west-014"
7290provider = "static"
7291mesh_tags = []
7292
7293[connect]
7294address = "192.168.10.14"
7295ssh = "yah@192.168.10.14"
7296identity_file = "~/.ssh/yah"
7297yubaba = "http://192.168.10.14:7443"
7298
7299[registration]
7300mesh_ipv4 = "100.64.0.6"
7301"#;
7302 let cfg: MachineConfig = toml::from_str(src).unwrap();
7303 assert_eq!(cfg.mesh_ipv4(), Some("100.64.0.6"), "still mesh-joined");
7304 assert_eq!(
7305 cfg.yubaba_url().as_deref(),
7306 Some("http://100.64.0.6:7443"),
7307 "automation dials the mesh, never the LAN literal"
7308 );
7309 assert_eq!(
7310 cfg.lan_endpoint().as_deref(),
7311 Some("192.168.10.14:7443"),
7312 "the LAN address is still recorded — as identity, not as reach"
7313 );
7314 }
7315
7316 /// The refusal R605-T10 asks for: a node whose ONLY declared reach is a LAN
7317 /// literal is unresolvable, and says so by name rather than returning a URL
7318 /// that will time out. us-west-011's shape before this ticket.
7319 #[test]
7320 fn a_lan_only_node_refuses_with_a_named_reason() {
7321 let src = r#"
7322name = "us-west-011"
7323provider = "static"
7324mesh_tags = []
7325
7326[connect]
7327address = "192.168.10.11"
7328ssh = "yah@192.168.10.11"
7329identity_file = "~/.ssh/yah"
7330yubaba = "http://192.168.10.11:7443"
7331"#;
7332 let cfg: MachineConfig = toml::from_str(src).unwrap();
7333 assert_eq!(cfg.yubaba_url(), None);
7334 let err = cfg.reach().unwrap_err();
7335 assert!(err.contains("us-west-011"), "{err}");
7336 assert!(err.contains("192.168.10.11"), "{err}");
7337 assert!(err.contains("mesh_ipv4"), "{err}");
7338 }
7339
7340 /// The loopback placeholder is a genuine declaration ("reach me through the
7341 /// SSH tunnel"), not a LAN literal — 127/8 is not RFC1918. It must keep
7342 /// resolving verbatim; `hub::coordinator::is_loopback_url` is what judges it
7343 /// downstream.
7344 #[test]
7345 fn a_loopback_placeholder_still_resolves_verbatim() {
7346 let src = r#"
7347name = "m"
7348provider = "static"
7349mesh_tags = []
7350
7351[connect]
7352address = "192.168.10.99"
7353ssh = "yah@192.168.10.99"
7354identity_file = "~/.ssh/yah"
7355yubaba = "http://127.0.0.1:7443"
7356"#;
7357 let cfg: MachineConfig = toml::from_str(src).unwrap();
7358 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://127.0.0.1:7443"));
7359 }
7360
7361 #[test]
7362 fn private_ranges_are_exactly_rfc1918() {
7363 for lan in [
7364 "http://192.168.10.11:7443",
7365 "http://10.0.0.5:7443",
7366 "http://172.16.4.1:7443",
7367 ] {
7368 assert!(private_ipv4_from_url(lan).is_some(), "{lan}");
7369 }
7370 for not_lan in [
7371 "http://100.64.0.6:7443", // mesh
7372 "http://127.0.0.1:7443", // loopback
7373 "http://172.32.0.1:7443", // just past 172.16/12
7374 "http://45.32.194.254:80", // public
7375 "http://us-west-001:7443", // name, not a literal
7376 ] {
7377 assert!(private_ipv4_from_url(not_lan).is_none(), "{not_lan}");
7378 }
7379 }
7380
7381 /// `normalize` migrates in place: the legacy fingerprint moves into
7382 /// `[registration]`, the mesh IP is lifted out of the URL, and the derived
7383 /// `[connect].yubaba` is cleared so the two halves cannot drift.
7384 #[test]
7385 fn normalize_migrates_legacy_fields_and_is_idempotent() {
7386 let src = r#"
7387name = "m"
7388provider = "static"
7389mesh_tags = []
7390hostkey_fingerprint = "SHA256:dmpq"
7391
7392[connect]
7393address = "15.204.89.240"
7394ssh = "debian@15.204.89.240"
7395identity_file = "~/.ssh/yah"
7396yubaba = "http://100.64.0.1:7443"
7397"#;
7398 let mut cfg: MachineConfig = toml::from_str(src).unwrap();
7399 cfg.normalize();
7400 assert!(cfg.legacy_hostkey_fingerprint.is_none());
7401 assert_eq!(
7402 cfg.registration.hostkey_fingerprint.as_deref(),
7403 Some("SHA256:dmpq")
7404 );
7405 assert_eq!(cfg.registration.mesh_ipv4.as_deref(), Some("100.64.0.1"));
7406 assert!(cfg.connect.as_ref().unwrap().yubaba.is_none());
7407 // Accessors still answer the same, and re-running changes nothing.
7408 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://100.64.0.1:7443"));
7409 let once = format!("{cfg:?}");
7410 cfg.normalize();
7411 assert_eq!(once, format!("{cfg:?}"));
7412 }
7413
7414 /// A loopback `[connect].yubaba` is a declaration ("no mesh address yet —
7415 /// reach me through the SSH tunnel"), not a stale observation, so
7416 /// `normalize` must leave it alone. us-west-003/011/013 depend on this.
7417 #[test]
7418 fn normalize_leaves_pre_mesh_loopback_declaration_intact() {
7419 let src = r#"
7420name = "m"
7421provider = "static"
7422mesh_tags = []
7423
7424[connect]
7425address = "192.168.10.11"
7426ssh = "yah@192.168.10.11"
7427identity_file = "~/.ssh/yah"
7428yubaba = "http://127.0.0.1:7443"
7429"#;
7430 let mut cfg: MachineConfig = toml::from_str(src).unwrap();
7431 cfg.normalize();
7432 assert_eq!(
7433 cfg.connect.as_ref().unwrap().yubaba.as_deref(),
7434 Some("http://127.0.0.1:7443")
7435 );
7436 assert!(cfg.registration.is_empty());
7437 assert_eq!(cfg.mesh_ipv4(), None);
7438 }
7439
7440 /// `save` normalizes, so a legacy file that round-trips through the writer
7441 /// comes back on the split shape with nothing lost — the property that
7442 /// keeps `yah cloud machine attach` from re-emitting the old layout.
7443 #[test]
7444 fn save_writes_the_split_shape_from_a_legacy_config() {
7445 let tmp = tempfile::TempDir::new().unwrap();
7446 let root = tmp.path();
7447 let src = r#"
7448name = "m"
7449provider = "static"
7450mesh_tags = []
7451hostkey_fingerprint = "SHA256:dmpq"
7452
7453[connect]
7454address = "15.204.89.240"
7455ssh = "debian@15.204.89.240"
7456identity_file = "~/.ssh/yah"
7457yubaba = "http://100.64.0.1:7443"
7458"#;
7459 let cfg: MachineConfig = toml::from_str(src).unwrap();
7460 cfg.save(root).unwrap();
7461
7462 let written = std::fs::read_to_string(root.join("machines/m.toml")).unwrap();
7463 let reg_at = written
7464 .find("[registration]")
7465 .unwrap_or_else(|| panic!("no [registration] table: {written}"));
7466 let fp_at = written
7467 .find("hostkey_fingerprint")
7468 .unwrap_or_else(|| panic!("fingerprint dropped: {written}"));
7469 assert!(
7470 fp_at > reg_at,
7471 "legacy top-level field must not be re-emitted: {written}"
7472 );
7473 assert!(
7474 !written.contains("yubaba ="),
7475 "derived URL must not be re-emitted alongside mesh_ipv4: {written}"
7476 );
7477
7478 let reloaded: MachineConfig = toml::from_str(&written).unwrap();
7479 assert_eq!(reloaded.hostkey_fingerprint(), Some("SHA256:dmpq"));
7480 assert_eq!(
7481 reloaded.yubaba_url().as_deref(),
7482 Some("http://100.64.0.1:7443")
7483 );
7484 }
7485
7486 /// `[registration]` is omitted entirely for a machine nothing has been
7487 /// observed about — a scaffolded declaration stays clean.
7488 #[test]
7489 fn empty_registration_is_omitted_on_serialize() {
7490 let src = r#"
7491name = "m"
7492provider = "static"
7493mesh_tags = []
7494"#;
7495 let cfg: MachineConfig = toml::from_str(src).unwrap();
7496 assert!(cfg.registration.is_empty());
7497 let out = toml::to_string_pretty(&cfg).unwrap();
7498 assert!(!out.contains("[registration]"), "{out}");
7499 }
7500
7501 #[test]
7502 fn driver_provider_without_location_fails_validate() {
7503 // A driver-backed provider (hetzner/vultr) still MUST carry location +
7504 // server_type — the driver can't create a server without them. The
7505 // contract moved from load-time (required field) to provision-time
7506 // (validate), so the TOML loads but validate() rejects it.
7507 let src = r#"
7508name = "us-west-001"
7509provider = "hetzner"
7510mesh_tags = []
7511"#;
7512 let cfg: MachineConfig = toml::from_str(src).unwrap();
7513 assert!(provider_has_machine_driver(&cfg.provider));
7514 let err = cfg.validate().unwrap_err().to_string();
7515 assert!(
7516 err.contains("location"),
7517 "expected location complaint: {err}"
7518 );
7519 }
7520
7521 /// Helper for the new-tree integration tests below: lay out
7522 /// `<workspace>/.yah/{infra,services}/` with `dev-yah` + its mirrors and
7523 /// the three Phase-A providers (cloudflare, hetzner, orbstack).
7524 fn make_new_tree_with_dev_yah(root: &std::path::Path) {
7525 let infra = root.join(".yah").join("infra");
7526 let providers = infra.join("providers");
7527 std::fs::create_dir_all(&providers).unwrap();
7528 std::fs::write(
7529 providers.join("cloudflare.toml"),
7530 r#"schema_version = 1
7531id = "cloudflare"
7532kind = "cloudflare"
7533credentials = "keystore://cloudflare/yah"
7534default_zone = "yah.dev"
7535"#,
7536 )
7537 .unwrap();
7538 std::fs::write(
7539 providers.join("hetzner.toml"),
7540 r#"schema_version = 1
7541id = "hetzner"
7542kind = "hetzner"
7543credentials = "keystore://hetzner/yah"
7544default_location = "pdx"
7545default_server_type = "cpx11"
7546ssh_keys = []
7547"#,
7548 )
7549 .unwrap();
7550 std::fs::write(
7551 providers.join("orbstack.toml"),
7552 r#"schema_version = 1
7553id = "orbstack"
7554kind = "local-container"
7555runtime = "auto"
7556
7557[discovery]
7558orbstack = "~/.orbstack/run/docker.sock"
7559"#,
7560 )
7561 .unwrap();
7562
7563 let svc = root.join(".yah").join("services").join("dev-yah");
7564 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
7565 std::fs::write(
7566 svc.join("service.toml"),
7567 r#"schema_version = 1
7568name = "dev-yah"
7569domain = "yah.dev"
7570
7571[[components]]
7572id = "site"
7573kind = "mesofact-static"
7574path = "app/yah/web"
7575role = "static"
7576"#,
7577 )
7578 .unwrap();
7579 std::fs::write(
7580 svc.join("mirrors/prod.toml"),
7581 r#"schema_version = 1
7582shape = "single-machine"
7583
7584[providers.static]
7585use = "cloudflare"
7586bucket = "yah-dev"
7587zone = "yah.dev"
7588"#,
7589 )
7590 .unwrap();
7591 std::fs::write(
7592 svc.join("mirrors/local.toml"),
7593 r#"schema_version = 1
7594shape = "local"
7595
7596[providers.static]
7597kind = "local-static"
7598port = 4321
7599
7600[providers.compute]
7601use = "orbstack"
7602"#,
7603 )
7604 .unwrap();
7605 }
7606
7607 #[test]
7608 fn cloud_config_load_new_tree_populates_providers_and_services() {
7609 let tmp = tempfile::TempDir::new().unwrap();
7610 let root = tmp.path();
7611 make_new_tree_with_dev_yah(root);
7612
7613 let cfg = CloudConfig::load(root).unwrap();
7614 assert_eq!(cfg.providers.len(), 3, "three providers loaded");
7615 assert!(cfg.provider("cloudflare").is_some());
7616 assert!(cfg.provider("hetzner").is_some());
7617 assert!(cfg.provider("orbstack").is_some());
7618
7619 let dev = cfg.service("dev-yah").expect("dev-yah service");
7620 assert_eq!(dev.service.domain, "yah.dev");
7621 assert_eq!(dev.service.components.len(), 1);
7622 assert_eq!(dev.mirrors.len(), 2);
7623 // Legacy file stems "prod" and "local" are normalised to canonical tier names.
7624 assert!(dev.mirrors.contains_key("cloud"), "prod.toml → cloud tier");
7625 assert!(dev.mirrors.contains_key("dev"), "local.toml → dev tier");
7626 assert_eq!(dev.mirrors["cloud"].shape, MirrorShape::SingleMachine);
7627 assert_eq!(dev.mirrors["dev"].shape, MirrorShape::Local);
7628
7629 // Legacy fields stay empty when no .yah/cloud/ exists.
7630 assert!(cfg.legacy_mirrors.is_empty());
7631 assert!(cfg.legacy_services.is_empty());
7632 assert!(cfg.workloads.is_empty());
7633 }
7634
7635 #[test]
7636 fn cloud_config_cross_ref_fails_on_missing_provider() {
7637 // Mirror references a provider id that doesn't exist.
7638 let tmp = tempfile::TempDir::new().unwrap();
7639 let root = tmp.path();
7640 let svc = root.join(".yah").join("services").join("dev-yah");
7641 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
7642 std::fs::write(
7643 svc.join("service.toml"),
7644 "schema_version = 1\nname = \"dev-yah\"\ndomain = \"yah.dev\"\n",
7645 )
7646 .unwrap();
7647 std::fs::write(
7648 svc.join("mirrors/prod.toml"),
7649 "schema_version = 1\nshape = \"single-machine\"\n\n[providers.static]\nuse = \"fly-io\"\n",
7650 ).unwrap();
7651
7652 let err = CloudConfig::load(root).unwrap_err();
7653 let msg = err.to_string();
7654 assert!(
7655 msg.contains("fly-io"),
7656 "error should name the missing provider id, got: {msg}"
7657 );
7658 assert!(
7659 msg.contains("providers/fly-io.toml") || msg.contains("no such provider"),
7660 "error should hint at remedy, got: {msg}"
7661 );
7662 }
7663
7664 #[test]
7665 fn cloud_config_cross_ref_fails_on_missing_provider_named_by_an_ingress_edge() {
7666 // R845: the edge's own `use` is a provider reference like any other, so
7667 // a typo has to fail here rather than at the Cloudflare arm of apply.
7668 let tmp = tempfile::TempDir::new().unwrap();
7669 let root = tmp.path();
7670 let svc = root.join(".yah").join("services").join("dev-yah");
7671 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
7672 std::fs::write(
7673 svc.join("service.toml"),
7674 "schema_version = 1\nname = \"dev-yah\"\ndomain = \"yah.dev\"\n",
7675 )
7676 .unwrap();
7677 std::fs::write(
7678 svc.join("mirrors/prod.toml"),
7679 "schema_version = 1\nshape = \"single-machine\"\n\n\
7680 [providers.compute]\nkind = \"static\"\nmachine = \"borrowed-01\"\n\
7681 zone = \"a.yah.dev\"\nport = 8080\n\n\
7682 [[ingress]]\nprovider = \"cloudflare-tunnel\"\nuse = \"cloudflar\"\n",
7683 )
7684 .unwrap();
7685
7686 let msg = CloudConfig::load(root).unwrap_err().to_string();
7687 assert!(
7688 msg.contains("ingress[0].use") && msg.contains("cloudflar"),
7689 "error should name the edge and the typo'd id, got: {msg}"
7690 );
7691 }
7692
7693 #[test]
7694 fn cloud_config_cross_ref_passes_on_inline_only_mirror() {
7695 // Inline `kind = "local-static"` doesn't require an infra provider.
7696 let tmp = tempfile::TempDir::new().unwrap();
7697 let root = tmp.path();
7698 let svc = root.join(".yah").join("services").join("local-only");
7699 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
7700 std::fs::write(
7701 svc.join("service.toml"),
7702 "schema_version = 1\nname = \"local-only\"\ndomain = \"local.test\"\n",
7703 )
7704 .unwrap();
7705 std::fs::write(
7706 svc.join("mirrors/local.toml"),
7707 "schema_version = 1\nshape = \"local\"\n\n[providers.static]\nkind = \"local-static\"\nport = 8080\n",
7708 ).unwrap();
7709
7710 // Should load fine: no `use=` references, no providers required.
7711 let cfg = CloudConfig::load(root).unwrap();
7712 assert!(cfg.service("local-only").is_some());
7713 }
7714
7715 fn mirror(src: &str) -> MirrorConfig {
7716 toml::from_str(&format!("schema_version = 1\nshape = \"single-machine\"\n{src}"))
7717 .expect("parse mirror")
7718 }
7719
7720 #[test]
7721 fn passway_machines_reads_both_ingress_spellings_the_same_way() {
7722 // The whole reason this is derived in Rust rather than read off a field
7723 // by the UI: these two mirrors say the identical thing, and a consumer
7724 // that reaches for `ingress_machines` sees the second one as empty.
7725 let scalar = mirror("ingress = \"passway\"\ningress_machines = [\"us-east-001\"]\n");
7726 let edges = mirror(
7727 "[[ingress]]\nprovider = \"passway\"\nmachines = [\"us-east-001\"]\n",
7728 );
7729 assert_eq!(scalar.passway_machines(), Some(vec!["us-east-001".into()]));
7730 assert_eq!(scalar.passway_machines(), edges.passway_machines());
7731 }
7732
7733 #[test]
7734 fn passway_machines_skips_a_cloudflare_tunnel_edge() {
7735 // A cloudflared node publishes through Cloudflare's DNS and does not
7736 // serve `GET /domains/{d}/onboarding`, so naming it here would point
7737 // the custom-domain UI at a node that cannot answer.
7738 let cf_only =
7739 mirror("[[ingress]]\nprovider = \"cloudflare-tunnel\"\nmachines = [\"cf-01\"]\n");
7740 assert_eq!(cf_only.passway_machines(), None);
7741
7742 let mixed = mirror(
7743 "[[ingress]]\nprovider = \"cloudflare-tunnel\"\nmachines = [\"cf-01\"]\n\
7744 slots = [\"static\"]\n\n\
7745 [[ingress]]\nprovider = \"passway\"\nmachines = [\"us-east-001\"]\n\
7746 slots = [\"bundle\"]\n",
7747 );
7748 assert_eq!(mixed.passway_machines(), Some(vec!["us-east-001".into()]));
7749 }
7750
7751 #[test]
7752 fn passway_machines_separates_declared_but_unplaced_from_undeclared() {
7753 // Some(vec![]) means "a passway front door exists, but its placement
7754 // falls back to the fronted slot's and is not knowable from the mirror".
7755 // None means there is no passway front door at all. Collapsing the two
7756 // would make a co-located edge indistinguishable from no edge.
7757 assert_eq!(mirror("ingress = \"passway\"\n").passway_machines(), Some(vec![]));
7758 assert_eq!(mirror("").passway_machines(), None);
7759 assert_eq!(mirror("ingress = \"none\"\n").passway_machines(), None);
7760 }
7761
7762 #[test]
7763 fn passway_machines_is_none_for_a_declaration_that_cannot_mean_anything() {
7764 // `ingress_machines` with no `ingress` is an error `ingress_edges` names
7765 // properly; swallowing it to None here is deliberate, because this is
7766 // read while loading every service in the workspace and hard-failing
7767 // would report an unrelated mirror's shape error from the wrong place.
7768 let orphaned = mirror("ingress_machines = [\"us-east-001\"]\n");
7769 assert!(orphaned.ingress_edges().is_err());
7770 assert_eq!(orphaned.passway_machines(), None);
7771 }
7772
7773 #[test]
7774 fn cloud_config_load_derives_passway_machines_only_for_passway_envs() {
7775 let tmp = tempfile::TempDir::new().unwrap();
7776 let root = tmp.path();
7777 let svc = root.join(".yah").join("services").join("dev-yah");
7778 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
7779 std::fs::write(
7780 svc.join("service.toml"),
7781 "schema_version = 1\nname = \"dev-yah\"\ndomain = \"yah.dev\"\n",
7782 )
7783 .unwrap();
7784 std::fs::write(
7785 svc.join("mirrors/cloud.toml"),
7786 "schema_version = 1\nshape = \"single-machine\"\n\
7787 ingress = \"passway\"\ningress_machines = [\"us-east-001\", \"us-west-001\"]\n\n\
7788 [providers.static]\nkind = \"local-static\"\nport = 8080\n",
7789 )
7790 .unwrap();
7791 std::fs::write(
7792 svc.join("mirrors/local.toml"),
7793 "schema_version = 1\nshape = \"local\"\n\n\
7794 [providers.static]\nkind = \"local-static\"\nport = 8080\n",
7795 )
7796 .unwrap();
7797
7798 let cfg = CloudConfig::load(root).unwrap();
7799 let svc = cfg.service("dev-yah").unwrap();
7800 assert_eq!(
7801 svc.passway_machines.get("cloud"),
7802 Some(&vec!["us-east-001".to_string(), "us-west-001".to_string()])
7803 );
7804 assert!(
7805 !svc.passway_machines.contains_key("local"),
7806 "an env with no front door must be absent, not empty: {:?}",
7807 svc.passway_machines
7808 );
7809 }
7810
7811 #[test]
7812 fn cloud_config_load_coexists_legacy_and_new_trees() {
7813 // Both trees present — both fields populated independently.
7814 let tmp = tempfile::TempDir::new().unwrap();
7815 let root = tmp.path();
7816 make_new_tree_with_dev_yah(root);
7817
7818 let cloud_dir = make_legacy_cloud_dir(root);
7819 std::fs::create_dir_all(cloud_dir.join("mirrors")).unwrap();
7820 std::fs::write(
7821 cloud_dir.join("mirrors/noisetable.toml"),
7822 "camp = \"noisetable\"\nregions = [\"pdx\"]\nworkloads = []\n",
7823 )
7824 .unwrap();
7825
7826 let cfg = CloudConfig::load(root).unwrap();
7827 assert_eq!(cfg.providers.len(), 3);
7828 assert!(cfg.service("dev-yah").is_some());
7829 assert_eq!(cfg.legacy_mirrors.len(), 1);
7830 assert!(cfg.legacy_mirror("noisetable").is_some());
7831 }
7832
7833 #[test]
7834 fn web_workload_round_trips() {
7835 // app/yah/web/workload.toml is parsed as a WorkloadSpec via the
7836 // workload-spec crate. The minimum-viable manifest here exercises
7837 // schema_version + kind + build fields.
7838 //
7839 // The on-disk file uses the abbreviated v1 form (kind + build); the
7840 // full WorkloadSpec is verbose, so this test asserts the new
7841 // mesofact-static abbreviated form parses as raw TOML (B3 will plumb
7842 // it through WorkloadSpec proper).
7843 // `routes` above [build] — it is a top-level field, and TOML would
7844 // scope it into that table if written below the header (R658-B1).
7845 let src = r#"
7846schema_version = 1
7847kind = "mesofact-static"
7848
7849routes = "./routes.ts"
7850
7851[build]
7852command = "bun run build"
7853out_dir = "dist"
7854"#;
7855 let v: toml::Value = toml::from_str(src).unwrap();
7856 assert_eq!(
7857 v.get("schema_version").and_then(|x| x.as_integer()),
7858 Some(1)
7859 );
7860 assert_eq!(
7861 v.get("kind").and_then(|x| x.as_str()),
7862 Some("mesofact-static")
7863 );
7864 let build = v
7865 .get("build")
7866 .and_then(|x| x.as_table())
7867 .expect("build table");
7868 assert_eq!(
7869 build.get("command").and_then(|x| x.as_str()),
7870 Some("bun run build")
7871 );
7872 assert_eq!(build.get("out_dir").and_then(|x| x.as_str()), Some("dist"));
7873 }
7874
7875 // ─── Canonical CRUD: ServiceConfig/MirrorConfig save + delete (R323-F1) ──
7876
7877 #[test]
7878 fn service_config_save_creates_canonical_toml_and_round_trips() {
7879 let tmp = tempfile::TempDir::new().unwrap();
7880 let root = tmp.path();
7881
7882 let svc = ServiceConfig {
7883 schema_version: 1,
7884 name: "dev-yah".into(),
7885 domain: "yah.dev".into(),
7886 db: DbCatalog::default(),
7887 components: vec![ServiceComponent {
7888 mount: None,
7889 id: "site".into(),
7890 kind: "mesofact-static".into(),
7891 path: "app/yah/web".into(),
7892 role: "static".into(),
7893 publishes: Some("static".into()),
7894 wave: 0,
7895 git: None,
7896 }],
7897 };
7898 svc.save(root).unwrap();
7899
7900 // Landed at the canonical path.
7901 let path = crate::paths::service_toml(root, "dev-yah");
7902 assert!(
7903 path.exists(),
7904 "service.toml should exist at {}",
7905 path.display()
7906 );
7907
7908 // Reloads through the full CloudConfig loader (no mirrors yet).
7909 let cfg = CloudConfig::load(root).unwrap();
7910 let loaded = cfg.service("dev-yah").expect("dev-yah service");
7911 assert_eq!(loaded.service.domain, "yah.dev");
7912 assert_eq!(loaded.service.components.len(), 1);
7913 assert_eq!(
7914 loaded.service.components[0].publishes.as_deref(),
7915 Some("static")
7916 );
7917 assert!(loaded.mirrors.is_empty());
7918 }
7919
7920 #[test]
7921 fn service_config_save_overwrites_in_place() {
7922 let tmp = tempfile::TempDir::new().unwrap();
7923 let root = tmp.path();
7924
7925 let mut svc = ServiceConfig {
7926 schema_version: 1,
7927 name: "dev-yah".into(),
7928 domain: "yah.dev".into(),
7929 components: vec![],
7930 db: DbCatalog::default(),
7931 };
7932 svc.save(root).unwrap();
7933 svc.domain = "yah.example".into();
7934 svc.save(root).unwrap();
7935
7936 let cfg = CloudConfig::load(root).unwrap();
7937 assert_eq!(
7938 cfg.service("dev-yah").unwrap().service.domain,
7939 "yah.example"
7940 );
7941 }
7942
7943 #[test]
7944 fn mirror_config_save_round_trips_reference_and_inline_slots() {
7945 let tmp = tempfile::TempDir::new().unwrap();
7946 let root = tmp.path();
7947
7948 // A service must exist so the loader walks the mirrors/ dir.
7949 ServiceConfig {
7950 schema_version: 1,
7951 name: "dev-yah".into(),
7952 domain: "yah.dev".into(),
7953 components: vec![],
7954 db: DbCatalog::default(),
7955 }
7956 .save(root)
7957 .unwrap();
7958
7959 // The cloudflare provider the reference slot points at must resolve,
7960 // or CloudConfig::load's cross-ref check rejects the tree.
7961 let providers = crate::paths::providers_dir(root);
7962 std::fs::create_dir_all(&providers).unwrap();
7963 std::fs::write(
7964 providers.join("cloudflare.toml"),
7965 "schema_version = 1\nid = \"cloudflare\"\nkind = \"cloudflare\"\n",
7966 )
7967 .unwrap();
7968
7969 let mut providers_map = BTreeMap::new();
7970 providers_map.insert(
7971 "static".to_string(),
7972 MirrorProviderSlot::Reference {
7973 provider_id: "cloudflare".into(),
7974 fields: {
7975 let mut f = BTreeMap::new();
7976 f.insert("bucket".to_string(), toml::Value::String("yah-dev".into()));
7977 f
7978 },
7979 },
7980 );
7981 providers_map.insert(
7982 "compute".to_string(),
7983 MirrorProviderSlot::Inline {
7984 kind: Provider::LocalStatic,
7985 fields: {
7986 let mut f = BTreeMap::new();
7987 f.insert("port".to_string(), toml::Value::Integer(4321));
7988 f
7989 },
7990 },
7991 );
7992 let mirror = MirrorConfig {
7993 schema_version: 1,
7994 shape: MirrorShape::SingleMachine,
7995 providers: providers_map,
7996 ingress: Default::default(),
7997 ingress_machines: Vec::new(),
7998 drivers: Default::default(),
7999 asset_aliases: Default::default(),
8000 };
8001 // Save with canonical name; legacy "prod" is normalised to "cloud" on load.
8002 mirror.save(root, "dev-yah", "cloud").unwrap();
8003
8004 let path = crate::paths::service_mirror_toml(root, "dev-yah", "cloud");
8005 assert!(
8006 path.exists(),
8007 "mirror toml should exist at {}",
8008 path.display()
8009 );
8010
8011 let cfg = CloudConfig::load(root).unwrap();
8012 let loaded = &cfg.service("dev-yah").unwrap().mirrors["cloud"];
8013 assert_eq!(loaded.shape, MirrorShape::SingleMachine);
8014 assert_eq!(loaded.providers["static"].provider_id(), Some("cloudflare"));
8015 assert_eq!(
8016 loaded.providers["compute"].inline_kind(),
8017 Some(Provider::LocalStatic)
8018 );
8019 }
8020
8021 #[test]
8022 fn service_delete_removes_dir_and_mirrors() {
8023 let tmp = tempfile::TempDir::new().unwrap();
8024 let root = tmp.path();
8025
8026 let svc = ServiceConfig {
8027 schema_version: 1,
8028 name: "dev-yah".into(),
8029 domain: "yah.dev".into(),
8030 components: vec![],
8031 db: DbCatalog::default(),
8032 };
8033 svc.save(root).unwrap();
8034 MirrorConfig {
8035 schema_version: 1,
8036 shape: MirrorShape::Local,
8037 providers: BTreeMap::new(),
8038 ingress: Default::default(),
8039 ingress_machines: Vec::new(),
8040 drivers: Default::default(),
8041 asset_aliases: Default::default(),
8042 }
8043 .save(root, "dev-yah", "local")
8044 .unwrap();
8045
8046 assert!(
8047 ServiceConfig::delete(root, "dev-yah").unwrap(),
8048 "first delete reports true"
8049 );
8050 assert!(!crate::paths::service_dir(root, "dev-yah").exists());
8051 // Idempotent: deleting again is a no-op that reports false.
8052 assert!(!ServiceConfig::delete(root, "dev-yah").unwrap());
8053
8054 let cfg = CloudConfig::load(root).unwrap();
8055 assert!(cfg.service("dev-yah").is_none());
8056 }
8057
8058 #[test]
8059 fn mirror_delete_leaves_other_mirrors_and_service_intact() {
8060 let tmp = tempfile::TempDir::new().unwrap();
8061 let root = tmp.path();
8062
8063 ServiceConfig {
8064 schema_version: 1,
8065 name: "dev-yah".into(),
8066 domain: "yah.dev".into(),
8067 components: vec![],
8068 db: DbCatalog::default(),
8069 }
8070 .save(root)
8071 .unwrap();
8072 for env in ["prod", "local"] {
8073 MirrorConfig {
8074 schema_version: 1,
8075 shape: MirrorShape::Local,
8076 providers: BTreeMap::new(),
8077 ingress: Default::default(),
8078 ingress_machines: Vec::new(),
8079 drivers: Default::default(),
8080 asset_aliases: Default::default(),
8081 }
8082 .save(root, "dev-yah", env)
8083 .unwrap();
8084 }
8085
8086 assert!(MirrorConfig::delete(root, "dev-yah", "prod").unwrap());
8087 assert!(!MirrorConfig::delete(root, "dev-yah", "prod").unwrap());
8088
8089 let cfg = CloudConfig::load(root).unwrap();
8090 let svc = cfg
8091 .service("dev-yah")
8092 .expect("service survives mirror delete");
8093 // Legacy file stems are normalised on load: "prod" → "cloud", "local" → "dev".
8094 assert!(!svc.mirrors.contains_key("cloud"));
8095 assert!(svc.mirrors.contains_key("dev"));
8096 }
8097
8098 // ─── DomainConfig (R347-F2) ────────────────────────────────────────────
8099
8100 fn write_marketing_service(root: &Path) {
8101 let svc = ServiceConfig {
8102 schema_version: 1,
8103 name: "yah-marketing".into(),
8104 domain: "yah.dev".into(),
8105 db: DbCatalog::default(),
8106 components: vec![ServiceComponent {
8107 mount: None,
8108 id: "site".into(),
8109 kind: "mesofact-static".into(),
8110 path: "app/yah/web".into(),
8111 role: "static".into(),
8112 publishes: None,
8113 wave: 0,
8114 git: None,
8115 }],
8116 };
8117 svc.save(root).unwrap();
8118 }
8119
8120 #[test]
8121 fn round_trip_domain_with_each_route_mode() {
8122 let dom = DomainConfig {
8123 schema_version: 1,
8124 name: "yah-dev".into(),
8125 domain: "yah.dev".into(),
8126 front_door: FrontDoor::Worker,
8127 cdn_bucket: "yah-dev".into(),
8128 worker_bundle_path: Some(".yah/workers/yah-dev/".into()),
8129 routes: vec![
8130 DomainRoute {
8131 headers: Default::default(),
8132 path: "/".into(),
8133 mode: RouteMode::Static {
8134 component: "yah-marketing/site".into(),
8135 },
8136 },
8137 DomainRoute {
8138 headers: Default::default(),
8139 path: "/dashboard/api/*".into(),
8140 mode: RouteMode::Backend {
8141 component: "yah-dashboard/api".into(),
8142 origin: "https://api.dashboard.yah.dev".into(),
8143 },
8144 },
8145 DomainRoute {
8146 headers: Default::default(),
8147 path: "/old".into(),
8148 mode: RouteMode::Redirect {
8149 target: "https://yah.dev/blog".into(),
8150 status: 308,
8151 },
8152 },
8153 ],
8154 };
8155 let s = toml::to_string(&dom).unwrap();
8156 let back: DomainConfig = toml::from_str(&s).unwrap();
8157 assert_eq!(back.name, "yah-dev");
8158 assert_eq!(back.routes.len(), 3);
8159 assert!(matches!(back.routes[0].mode, RouteMode::Static { .. }));
8160 assert!(matches!(back.routes[1].mode, RouteMode::Backend { .. }));
8161 assert!(matches!(back.routes[2].mode, RouteMode::Redirect { .. }));
8162 }
8163
8164 #[test]
8165 fn redirect_status_defaults_to_308() {
8166 let src = r#"
8167schema_version = 1
8168name = "yah-dev"
8169domain = "yah.dev"
8170front_door = "worker"
8171cdn_bucket = "yah-dev"
8172
8173[[routes]]
8174path = "/old"
8175mode = "redirect"
8176target = "https://yah.dev/blog"
8177"#;
8178 let dom: DomainConfig = toml::from_str(src).unwrap();
8179 let RouteMode::Redirect { status, .. } = &dom.routes[0].mode else {
8180 panic!("expected redirect");
8181 };
8182 assert_eq!(*status, 308);
8183 }
8184
8185 #[test]
8186 fn missing_domains_dir_is_empty() {
8187 let tmp = tempfile::TempDir::new().unwrap();
8188 // R844-B7: `.yah/` must exist or this is a wrong-root error rather
8189 // than an empty tree. The absent directory under test is `domains/`.
8190 std::fs::create_dir_all(tmp.path().join(".yah")).unwrap();
8191 let cfg = CloudConfig::load(tmp.path()).unwrap();
8192 assert!(cfg.domains.is_empty());
8193 }
8194
8195 #[test]
8196 fn save_reload_roundtrip() {
8197 let tmp = tempfile::TempDir::new().unwrap();
8198 let root = tmp.path();
8199 write_marketing_service(root);
8200
8201 let dom = DomainConfig {
8202 schema_version: 1,
8203 name: "yah-dev".into(),
8204 domain: "yah.dev".into(),
8205 front_door: FrontDoor::Worker,
8206 cdn_bucket: "yah-dev".into(),
8207 worker_bundle_path: None,
8208 routes: vec![DomainRoute {
8209 headers: Default::default(),
8210 path: "/".into(),
8211 mode: RouteMode::Static {
8212 component: "yah-marketing/site".into(),
8213 },
8214 }],
8215 };
8216 dom.save(root).unwrap();
8217
8218 let cfg = CloudConfig::load(root).unwrap();
8219 let loaded = cfg.domain("yah-dev").expect("yah-dev domain");
8220 assert_eq!(loaded.domain, "yah.dev");
8221 assert_eq!(loaded.routes.len(), 1);
8222 }
8223
8224 #[test]
8225 fn delete_returns_false_when_absent() {
8226 let tmp = tempfile::TempDir::new().unwrap();
8227 assert!(!DomainConfig::delete(tmp.path(), "no-such-domain").unwrap());
8228 }
8229
8230 #[test]
8231 fn delete_returns_true_first_time() {
8232 let tmp = tempfile::TempDir::new().unwrap();
8233 let root = tmp.path();
8234 let dom = DomainConfig {
8235 schema_version: 1,
8236 name: "yah-dev".into(),
8237 domain: "yah.dev".into(),
8238 front_door: FrontDoor::BucketDirect,
8239 cdn_bucket: "yah-dev".into(),
8240 worker_bundle_path: None,
8241 routes: vec![],
8242 };
8243 dom.save(root).unwrap();
8244 assert!(DomainConfig::delete(root, "yah-dev").unwrap());
8245 assert!(!DomainConfig::delete(root, "yah-dev").unwrap());
8246 }
8247
8248 // ---- R594-F12: front-door discriminator ------------------------------
8249
8250 /// Write a raw domain manifest so the tests exercise the deserialize +
8251 /// validate path, not a hand-built struct that skipped serde.
8252 fn write_domain_toml(root: &Path, stem: &str, body: &str) {
8253 let dir = root.join(".yah").join("domains");
8254 std::fs::create_dir_all(&dir).unwrap();
8255 std::fs::write(dir.join(format!("{stem}.toml")), body).unwrap();
8256 }
8257
8258 #[test]
8259 fn front_door_is_required() {
8260 let tmp = tempfile::TempDir::new().unwrap();
8261 let root = tmp.path();
8262 write_marketing_service(root);
8263 write_domain_toml(
8264 root,
8265 "yah-dev",
8266 r#"
8267schema_version = 1
8268name = "yah-dev"
8269domain = "yah.dev"
8270cdn_bucket = "yah-dev"
8271[[routes]]
8272path = "/*"
8273mode = "static"
8274component = "yah-marketing/site"
8275"#,
8276 );
8277 let err = CloudConfig::load(root).unwrap_err().to_string();
8278 // serde's own missing-field message; the point is that omitting the
8279 // discriminator is not a silently-defaulted state.
8280 assert!(err.contains("yah-dev.toml"), "{err}");
8281 }
8282
8283 #[test]
8284 fn bucket_direct_with_routes_is_rejected() {
8285 let tmp = tempfile::TempDir::new().unwrap();
8286 let root = tmp.path();
8287 write_marketing_service(root);
8288 write_domain_toml(
8289 root,
8290 "cdn-yah-dev",
8291 r#"
8292schema_version = 1
8293name = "cdn-yah-dev"
8294domain = "cdn.yah.dev"
8295front_door = "bucket-direct"
8296cdn_bucket = "yah-dev"
8297[[routes]]
8298path = "/docs/*"
8299mode = "static"
8300component = "yah-marketing/site"
8301"#,
8302 );
8303 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8304 assert!(err.contains("front_door"), "{err}");
8305 assert!(err.contains("/docs/*"), "{err}");
8306 }
8307
8308 #[test]
8309 fn bucket_direct_with_worker_bundle_path_is_rejected() {
8310 let tmp = tempfile::TempDir::new().unwrap();
8311 let root = tmp.path();
8312 write_domain_toml(
8313 root,
8314 "cdn-yah-dev",
8315 r#"
8316schema_version = 1
8317name = "cdn-yah-dev"
8318domain = "cdn.yah.dev"
8319front_door = "bucket-direct"
8320cdn_bucket = "yah-dev"
8321worker_bundle_path = ".yah/workers/cdn-yah-dev/"
8322"#,
8323 );
8324 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8325 assert!(err.contains("worker_bundle_path"), "{err}");
8326 }
8327
8328 // ── R746: per-route response headers + component mounts ──────────────────
8329
8330 /// A two-component service: `site` at the root, `app` mounted at `/app`
8331 /// with isolation headers on its route. This is the noisetable.com shape
8332 /// the primitive was built for.
8333 fn write_two_component_service(root: &Path) {
8334 let svc = ServiceConfig {
8335 schema_version: 1,
8336 name: "yah-marketing".into(),
8337 domain: "yah.dev".into(),
8338 db: DbCatalog::default(),
8339 components: vec![
8340 ServiceComponent {
8341 mount: None,
8342 id: "site".into(),
8343 kind: "mesofact-static".into(),
8344 path: "app/yah/web".into(),
8345 role: "static".into(),
8346 publishes: None,
8347 wave: 0,
8348 git: None,
8349 },
8350 ServiceComponent {
8351 mount: Some("/app".into()),
8352 id: "app".into(),
8353 kind: "mesofact-static".into(),
8354 path: "app/browser".into(),
8355 role: "static".into(),
8356 publishes: None,
8357 wave: 0,
8358 git: None,
8359 },
8360 ],
8361 };
8362 svc.save(root).unwrap();
8363 }
8364
8365 const MOUNTED_DOMAIN: &str = r#"
8366schema_version = 1
8367name = "yah-dev"
8368domain = "yah.dev"
8369front_door = "worker"
8370cdn_bucket = "yah-dev"
8371
8372[[routes]]
8373path = "/app/*"
8374mode = "static"
8375component = "yah-marketing/app"
8376headers = { "Cross-Origin-Opener-Policy" = "same-origin", "Cross-Origin-Embedder-Policy" = "require-corp" }
8377
8378[[routes]]
8379path = "/*"
8380mode = "static"
8381component = "yah-marketing/site"
8382"#;
8383
8384 #[test]
8385 fn a_mounted_component_routed_at_its_mount_loads() {
8386 let tmp = tempfile::TempDir::new().unwrap();
8387 let root = tmp.path();
8388 write_two_component_service(root);
8389 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
8390 let cfg = CloudConfig::load(root).unwrap();
8391 let dom = cfg.domain("yah-dev").unwrap();
8392 assert_eq!(dom.routes.len(), 2);
8393 assert_eq!(
8394 dom.routes[0].headers.get("Cross-Origin-Opener-Policy").map(String::as_str),
8395 Some("same-origin")
8396 );
8397 assert!(dom.routes[1].headers.is_empty());
8398 }
8399
8400 /// The header table reaches the Worker in MANIFEST order with headerless
8401 /// routes dropped. Order is the whole contract — the front door applies the
8402 /// first match, so `/app/*` before `/*` is what isolates the app without
8403 /// isolating the marketing site.
8404 #[test]
8405 fn route_headers_json_preserves_order_and_drops_headerless_routes() {
8406 let tmp = tempfile::TempDir::new().unwrap();
8407 let root = tmp.path();
8408 write_two_component_service(root);
8409 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
8410 let cfg = CloudConfig::load(root).unwrap();
8411 let json = cfg.domain("yah-dev").unwrap().route_headers_json();
8412
8413 let parsed: serde_json::Value = serde_json::from_str(&json).unwrap();
8414 let rules = parsed.as_array().unwrap();
8415 assert_eq!(rules.len(), 1, "the headerless catch-all is dropped: {json}");
8416 assert_eq!(rules[0]["path"], "/app/*");
8417 assert_eq!(rules[0]["headers"]["Cross-Origin-Embedder-Policy"], "require-corp");
8418 }
8419
8420 #[test]
8421 fn route_headers_json_is_an_empty_array_when_nothing_declares_headers() {
8422 let tmp = tempfile::TempDir::new().unwrap();
8423 let root = tmp.path();
8424 write_marketing_service(root);
8425 write_domain_toml(
8426 root,
8427 "yah-dev",
8428 r#"
8429schema_version = 1
8430name = "yah-dev"
8431domain = "yah.dev"
8432front_door = "worker"
8433cdn_bucket = "yah-dev"
8434
8435[[routes]]
8436path = "/*"
8437mode = "static"
8438component = "yah-marketing/site"
8439"#,
8440 );
8441 let cfg = CloudConfig::load(root).unwrap();
8442 assert_eq!(cfg.domain("yah-dev").unwrap().route_headers_json(), "[]");
8443 }
8444
8445 /// The reconciler's own entry point: given a workspace root and a service
8446 /// name, produce the binding value. `"[]"` when nothing routes the service.
8447 #[test]
8448 fn route_headers_for_service_reads_the_workspace_domains() {
8449 let tmp = tempfile::TempDir::new().unwrap();
8450 let root = tmp.path();
8451 write_two_component_service(root);
8452 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
8453 assert!(route_headers_for_service(root, "yah-marketing")
8454 .unwrap()
8455 .contains("require-corp"));
8456 assert_eq!(route_headers_for_service(root, "some-other-svc").unwrap(), "[]");
8457 }
8458
8459 // ---- R749-T5: a broken table fails the DEPLOY, not the edge -----------
8460
8461 /// The manifest's `headers` map is hand-written TOML, so a header name with
8462 /// spaces in it is one keystroke away — and it survives serialization into
8463 /// a structurally-valid table that neither front door can apply. Fail at
8464 /// load, naming the domain, the route and the header, instead of shipping a
8465 /// binding the Worker throws on and an origin that refuses to boot.
8466 #[test]
8467 fn a_route_header_name_that_is_not_a_header_name_fails_the_load() {
8468 let tmp = tempfile::TempDir::new().unwrap();
8469 let root = tmp.path();
8470 write_marketing_service(root);
8471 write_domain_toml(
8472 root,
8473 "yah-dev",
8474 r#"
8475schema_version = 1
8476name = "yah-dev"
8477domain = "yah.dev"
8478front_door = "worker"
8479cdn_bucket = "yah-dev"
8480
8481[[routes]]
8482path = "/*"
8483mode = "static"
8484component = "yah-marketing/site"
8485headers = { "Cross Origin Opener Policy" = "same-origin" }
8486"#,
8487 );
8488 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8489 assert!(err.contains("yah-dev"), "{err}");
8490 assert!(err.contains("/*"), "{err}");
8491 assert!(err.contains("Cross Origin Opener Policy"), "{err}");
8492 assert!(err.contains("not a valid HTTP header name"), "{err}");
8493 }
8494
8495 /// A newline in a value is header injection if it ever reached the wire, so
8496 /// both doors reject it and so does this.
8497 #[test]
8498 fn a_route_header_value_that_is_not_a_header_value_fails_the_load() {
8499 let tmp = tempfile::TempDir::new().unwrap();
8500 let root = tmp.path();
8501 write_marketing_service(root);
8502 write_domain_toml(
8503 root,
8504 "yah-dev",
8505 r#"
8506schema_version = 1
8507name = "yah-dev"
8508domain = "yah.dev"
8509front_door = "worker"
8510cdn_bucket = "yah-dev"
8511
8512[[routes]]
8513path = "/*"
8514mode = "static"
8515component = "yah-marketing/site"
8516headers = { "X-Frame-Options" = "DENY\nSet-Cookie: pwned=1" }
8517"#,
8518 );
8519 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8520 assert!(err.contains("X-Frame-Options"), "{err}");
8521 assert!(err.contains("not a valid HTTP header value"), "{err}");
8522 }
8523
8524 /// The invariant this gate exists to hold: everything `route_headers_json`
8525 /// emits is applicable. A headerless route contributes no rule, so its path
8526 /// is not the table's business — only rules that ship are checked.
8527 #[test]
8528 fn a_headerless_route_is_not_subject_to_the_route_header_gate() {
8529 let tmp = tempfile::TempDir::new().unwrap();
8530 let root = tmp.path();
8531 write_two_component_service(root);
8532 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
8533 let cfg = CloudConfig::load(root).unwrap();
8534 cfg.domain("yah-dev")
8535 .unwrap()
8536 .validate_route_headers()
8537 .unwrap();
8538 }
8539
8540 /// A `bucket-direct` domain has no front door to set headers on, so it must
8541 /// not be picked up as a service's header source.
8542 #[test]
8543 fn route_headers_ignores_domains_that_are_not_route_driven() {
8544 let doms: BTreeMap<String, DomainConfig> = [(
8545 "cdn".to_string(),
8546 DomainConfig {
8547 schema_version: 1,
8548 name: "cdn".into(),
8549 domain: "cdn.yah.dev".into(),
8550 front_door: FrontDoor::BucketDirect,
8551 cdn_bucket: "yah-dev".into(),
8552 worker_bundle_path: None,
8553 routes: vec![],
8554 },
8555 )]
8556 .into_iter()
8557 .collect();
8558 assert!(domain_serving_service(&doms, "yah-marketing").is_none());
8559 }
8560
8561 #[test]
8562 fn a_mount_that_disagrees_with_its_route_path_is_rejected() {
8563 let tmp = tempfile::TempDir::new().unwrap();
8564 let root = tmp.path();
8565 write_two_component_service(root);
8566 write_domain_toml(
8567 root,
8568 "yah-dev",
8569 r#"
8570schema_version = 1
8571name = "yah-dev"
8572domain = "yah.dev"
8573front_door = "worker"
8574cdn_bucket = "yah-dev"
8575
8576[[routes]]
8577path = "/studio/*"
8578mode = "static"
8579component = "yah-marketing/app"
8580"#,
8581 );
8582 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8583 assert!(err.contains("mount = \"/app\""), "{err}");
8584 assert!(err.contains("/studio/*"), "{err}");
8585 }
8586
8587 /// The other direction: routing an unmounted component under a sub-path
8588 /// points requests at a prefix nothing published to.
8589 #[test]
8590 fn routing_an_unmounted_component_under_a_subpath_is_rejected() {
8591 let tmp = tempfile::TempDir::new().unwrap();
8592 let root = tmp.path();
8593 write_marketing_service(root);
8594 write_domain_toml(
8595 root,
8596 "yah-dev",
8597 r#"
8598schema_version = 1
8599name = "yah-dev"
8600domain = "yah.dev"
8601front_door = "worker"
8602cdn_bucket = "yah-dev"
8603
8604[[routes]]
8605path = "/docs/*"
8606mode = "static"
8607component = "yah-marketing/site"
8608"#,
8609 );
8610 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8611 assert!(err.contains("no `mount`"), "{err}");
8612 assert!(err.contains("/docs"), "{err}");
8613 }
8614
8615 #[test]
8616 fn mount_and_route_prefix_normalization_agree() {
8617 for m in ["/app", "app", "app/", "/app/"] {
8618 assert_eq!(normalize_mount(m), "app", "mount {m:?}");
8619 }
8620 assert_eq!(normalize_mount("/"), "");
8621 assert_eq!(route_path_prefix("/*"), "");
8622 assert_eq!(route_path_prefix("/app/*"), "app");
8623 assert_eq!(route_path_prefix("/app"), "app");
8624 assert_eq!(route_path_prefix("/"), "");
8625 }
8626
8627 // ── R870-B11: a mount is owned by exactly one bundle-tier component ────
8628
8629 /// Two bundle-tier components at the same explicit mount would stage into
8630 /// the same `app/dist/<mount>/` prefix inside one assembled bundle and
8631 /// silently clobber each other — reject at load, before that happens.
8632 #[test]
8633 fn two_bundle_components_at_the_same_mount_are_rejected() {
8634 let tmp = tempfile::TempDir::new().unwrap();
8635 let root = tmp.path();
8636 let svc = ServiceConfig {
8637 schema_version: 1,
8638 name: "noisetable-marketing".into(),
8639 domain: "noisetable.com".into(),
8640 db: DbCatalog::default(),
8641 components: vec![
8642 ServiceComponent {
8643 mount: Some("/app".into()),
8644 id: "app".into(),
8645 kind: "mesofact-static".into(),
8646 path: "app/browser".into(),
8647 role: "static".into(),
8648 publishes: None,
8649 wave: 0,
8650 git: None,
8651 },
8652 ServiceComponent {
8653 mount: Some("app/".into()),
8654 id: "app2".into(),
8655 kind: "mesofact-spa".into(),
8656 path: "app/other".into(),
8657 role: "static".into(),
8658 publishes: None,
8659 wave: 0,
8660 git: None,
8661 },
8662 ],
8663 };
8664 svc.save(root).unwrap();
8665 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8666 assert!(err.contains("\"app\""), "{err}");
8667 assert!(err.contains("\"app2\""), "{err}");
8668 assert!(err.contains("mount = \"/app\""), "{err}");
8669 }
8670
8671 /// The unmounted case: two bundle-tier components both leaving `mount`
8672 /// unset both claim the service root, which collides exactly the same
8673 /// way — this is the noisetable shape the ticket was filed against, if
8674 /// `app`'s mount had been forgotten instead of declared.
8675 #[test]
8676 fn two_bundle_components_with_no_mount_are_rejected() {
8677 let tmp = tempfile::TempDir::new().unwrap();
8678 let root = tmp.path();
8679 let svc = ServiceConfig {
8680 schema_version: 1,
8681 name: "noisetable-marketing".into(),
8682 domain: "noisetable.com".into(),
8683 db: DbCatalog::default(),
8684 components: vec![
8685 ServiceComponent {
8686 mount: None,
8687 id: "site".into(),
8688 kind: "mesofact-spa".into(),
8689 path: "web/landing".into(),
8690 role: "static".into(),
8691 publishes: None,
8692 wave: 0,
8693 git: None,
8694 },
8695 ServiceComponent {
8696 mount: None,
8697 id: "app".into(),
8698 kind: "mesofact-static".into(),
8699 path: "app/browser".into(),
8700 role: "static".into(),
8701 publishes: None,
8702 wave: 0,
8703 git: None,
8704 },
8705 ],
8706 };
8707 svc.save(root).unwrap();
8708 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8709 assert!(err.contains("\"site\""), "{err}");
8710 assert!(err.contains("\"app\""), "{err}");
8711 assert!(err.contains("the service root"), "{err}");
8712 }
8713
8714 /// The legitimate shape (distinct mounts) is untouched — regression guard
8715 /// so the new check does not become the next silent-overwrite bug.
8716 #[test]
8717 fn bundle_components_at_distinct_mounts_still_load() {
8718 let tmp = tempfile::TempDir::new().unwrap();
8719 let root = tmp.path();
8720 write_two_component_service(root);
8721 assert!(CloudConfig::load(root).is_ok());
8722 }
8723
8724 #[test]
8725 fn worker_with_no_routes_is_rejected() {
8726 let tmp = tempfile::TempDir::new().unwrap();
8727 let root = tmp.path();
8728 write_domain_toml(
8729 root,
8730 "yah-dev",
8731 r#"
8732schema_version = 1
8733name = "yah-dev"
8734domain = "yah.dev"
8735front_door = "worker"
8736cdn_bucket = "yah-dev"
8737"#,
8738 );
8739 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8740 assert!(err.contains("front_door = \"worker\""), "{err}");
8741 assert!(err.contains("404"), "{err}");
8742 }
8743
8744 #[test]
8745 fn passway_with_no_routes_is_rejected_too() {
8746 let tmp = tempfile::TempDir::new().unwrap();
8747 let root = tmp.path();
8748 write_domain_toml(
8749 root,
8750 "yah-dev",
8751 r#"
8752schema_version = 1
8753name = "yah-dev"
8754domain = "yah.dev"
8755front_door = "passway"
8756cdn_bucket = "yah-dev"
8757"#,
8758 );
8759 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
8760 assert!(err.contains("front_door = \"passway\""), "{err}");
8761 }
8762
8763 #[test]
8764 fn bucket_direct_without_routes_loads() {
8765 let tmp = tempfile::TempDir::new().unwrap();
8766 let root = tmp.path();
8767 // Exactly the shape .yah/domains/cdn-yah-dev.toml ships (W175: a pure
8768 // asset tier deliberately has no Worker behaviours).
8769 write_domain_toml(
8770 root,
8771 "cdn-yah-dev",
8772 r#"
8773schema_version = 1
8774name = "cdn-yah-dev"
8775domain = "cdn.yah.dev"
8776front_door = "bucket-direct"
8777cdn_bucket = "yah-dev"
8778"#,
8779 );
8780 let cfg = CloudConfig::load(root).unwrap();
8781 let dom = cfg.domain("cdn-yah-dev").expect("cdn-yah-dev domain");
8782 assert_eq!(dom.front_door, FrontDoor::BucketDirect);
8783 assert!(!dom.front_door.is_route_driven());
8784 }
8785
8786 #[test]
8787 fn front_door_round_trips_through_save() {
8788 let tmp = tempfile::TempDir::new().unwrap();
8789 let root = tmp.path();
8790 write_marketing_service(root);
8791 let dom = DomainConfig {
8792 schema_version: 1,
8793 name: "yah-dev".into(),
8794 domain: "yah.dev".into(),
8795 front_door: FrontDoor::Passway,
8796 cdn_bucket: "yah-dev".into(),
8797 worker_bundle_path: None,
8798 routes: vec![DomainRoute {
8799 headers: Default::default(),
8800 path: "/*".into(),
8801 mode: RouteMode::Static {
8802 component: "yah-marketing/site".into(),
8803 },
8804 }],
8805 };
8806 dom.save(root).unwrap();
8807 let cfg = CloudConfig::load(root).unwrap();
8808 assert_eq!(
8809 cfg.domain("yah-dev").unwrap().front_door,
8810 FrontDoor::Passway
8811 );
8812 }
8813
8814 // The four manifests this repo actually ships are asserted in
8815 // `tests/live_workspace_smoke.rs` — that's the only place with a
8816 // depth-agnostic path to the live `.yah/` tree and a skip path for the
8817 // standalone mirror checkout.
8818
8819 #[test]
8820 fn cross_ref_bails_on_missing_service() {
8821 let tmp = tempfile::TempDir::new().unwrap();
8822 let root = tmp.path();
8823 // No services declared at all — component ref must fail to resolve.
8824 let dom = DomainConfig {
8825 schema_version: 1,
8826 name: "yah-dev".into(),
8827 domain: "yah.dev".into(),
8828 front_door: FrontDoor::Worker,
8829 cdn_bucket: "yah-dev".into(),
8830 worker_bundle_path: None,
8831 routes: vec![DomainRoute {
8832 headers: Default::default(),
8833 path: "/".into(),
8834 mode: RouteMode::Static {
8835 component: "yah-marketing/site".into(),
8836 },
8837 }],
8838 };
8839 dom.save(root).unwrap();
8840
8841 let err = CloudConfig::load(root).unwrap_err();
8842 let msg = format!("{err:#}");
8843 assert!(msg.contains("no such service"), "got: {msg}");
8844 assert!(msg.contains("yah-marketing"), "got: {msg}");
8845 }
8846
8847 #[test]
8848 fn cross_ref_bails_on_missing_component() {
8849 let tmp = tempfile::TempDir::new().unwrap();
8850 let root = tmp.path();
8851 write_marketing_service(root); // has component id "site", not "elsewhere"
8852
8853 let dom = DomainConfig {
8854 schema_version: 1,
8855 name: "yah-dev".into(),
8856 domain: "yah.dev".into(),
8857 front_door: FrontDoor::Worker,
8858 cdn_bucket: "yah-dev".into(),
8859 worker_bundle_path: None,
8860 routes: vec![DomainRoute {
8861 headers: Default::default(),
8862 path: "/".into(),
8863 mode: RouteMode::Static {
8864 component: "yah-marketing/elsewhere".into(),
8865 },
8866 }],
8867 };
8868 dom.save(root).unwrap();
8869
8870 let err = CloudConfig::load(root).unwrap_err();
8871 let msg = format!("{err:#}");
8872 assert!(msg.contains("no component with id"), "got: {msg}");
8873 assert!(msg.contains("elsewhere"), "got: {msg}");
8874 }
8875
8876 #[test]
8877 fn cross_ref_bails_on_malformed_ref() {
8878 let tmp = tempfile::TempDir::new().unwrap();
8879 let root = tmp.path();
8880 write_marketing_service(root);
8881
8882 let dom = DomainConfig {
8883 schema_version: 1,
8884 name: "yah-dev".into(),
8885 domain: "yah.dev".into(),
8886 front_door: FrontDoor::Worker,
8887 cdn_bucket: "yah-dev".into(),
8888 worker_bundle_path: None,
8889 routes: vec![DomainRoute {
8890 headers: Default::default(),
8891 path: "/".into(),
8892 mode: RouteMode::Static {
8893 component: "no-slash-here".into(),
8894 },
8895 }],
8896 };
8897 dom.save(root).unwrap();
8898
8899 let err = CloudConfig::load(root).unwrap_err();
8900 let msg = format!("{err:#}");
8901 assert!(msg.contains("expected"), "got: {msg}");
8902 }
8903
8904 #[test]
8905 fn redirect_routes_skip_component_validation() {
8906 let tmp = tempfile::TempDir::new().unwrap();
8907 let root = tmp.path();
8908 // No services at all — redirect must still load cleanly because it
8909 // references nothing.
8910 let dom = DomainConfig {
8911 schema_version: 1,
8912 name: "yah-dev".into(),
8913 domain: "yah.dev".into(),
8914 front_door: FrontDoor::Worker,
8915 cdn_bucket: "yah-dev".into(),
8916 worker_bundle_path: None,
8917 routes: vec![DomainRoute {
8918 headers: Default::default(),
8919 path: "/old".into(),
8920 mode: RouteMode::Redirect {
8921 target: "https://yah.dev/blog".into(),
8922 status: 308,
8923 },
8924 }],
8925 };
8926 dom.save(root).unwrap();
8927
8928 let cfg = CloudConfig::load(root).unwrap();
8929 assert!(cfg.domain("yah-dev").is_some());
8930 }
8931
8932 #[test]
8933 fn name_must_match_file_stem() {
8934 let tmp = tempfile::TempDir::new().unwrap();
8935 let root = tmp.path();
8936 // Hand-write a file whose stem disagrees with its `name`.
8937 let dir = root.join(".yah").join("domains");
8938 std::fs::create_dir_all(&dir).unwrap();
8939 std::fs::write(
8940 dir.join("yah-dev.toml"),
8941 r#"schema_version = 1
8942name = "different-name"
8943domain = "yah.dev"
8944front_door = "bucket-direct"
8945cdn_bucket = "yah-dev"
8946"#,
8947 )
8948 .unwrap();
8949
8950 let err = CloudConfig::load(root).unwrap_err();
8951 let msg = format!("{err:#}");
8952 assert!(msg.contains("must match the file stem"), "got: {msg}");
8953 }
8954
8955 #[test]
8956 fn net_alias_tier_subdomain_manifest_loads_and_cross_refs() {
8957 // R561-F2: a per-tenant subdomain manifest on the net.yah.dev wildcard
8958 // alias tier is just a DomainConfig whose `domain` is `<name>.net.yah.dev`
8959 // and whose static route cross-refs the tenant's service component.
8960 // This is exactly the shape .yah/domains/scrabcake-net-yah-dev.toml ships.
8961 let tmp = tempfile::TempDir::new().unwrap();
8962 let root = tmp.path();
8963 write_marketing_service(root); // service "yah-marketing", component "site"
8964
8965 let dom = DomainConfig {
8966 schema_version: 1,
8967 name: "tenant-net-yah-dev".into(),
8968 domain: "tenant.net.yah.dev".into(),
8969 front_door: FrontDoor::Worker,
8970 cdn_bucket: "net-yah-dev".into(), // shared per-tier bucket
8971 worker_bundle_path: None,
8972 routes: vec![DomainRoute {
8973 headers: Default::default(),
8974 path: "/*".into(),
8975 mode: RouteMode::Static {
8976 component: "yah-marketing/site".into(),
8977 },
8978 }],
8979 };
8980 dom.save(root).unwrap();
8981
8982 let cfg = CloudConfig::load(root).unwrap();
8983 let dom = cfg
8984 .domain("tenant-net-yah-dev")
8985 .expect("net-tier subdomain manifest should load");
8986 assert_eq!(dom.domain, "tenant.net.yah.dev");
8987 assert_eq!(dom.cdn_bucket, "net-yah-dev");
8988 }
8989
8990 // ─── R572-F3: NodeAllocatable + taints ──────────────────────────────────
8991
8992 #[test]
8993 fn machine_allocatable_round_trips() {
8994 let toml_src = r#"
8995name = "us-west-001"
8996provider = "static"
8997mesh_tags = ["tag:cloud-runner"]
8998[allocatable]
8999memory_mb = 3800
9000cpu_millis = 2000
9001"#;
9002 let m: MachineConfig = toml::from_str(toml_src).unwrap();
9003 let a = m.allocatable.as_ref().expect("allocatable should parse");
9004 assert_eq!(a.memory_mb, 3800);
9005 assert_eq!(a.cpu_millis, 2000);
9006
9007 let s = toml::to_string(&m).unwrap();
9008 let back: MachineConfig = toml::from_str(&s).unwrap();
9009 let a2 = back.allocatable.as_ref().unwrap();
9010 assert_eq!(a2.memory_mb, 3800);
9011 assert_eq!(a2.cpu_millis, 2000);
9012 }
9013
9014 #[test]
9015 fn machine_taints_round_trips() {
9016 let toml_src = r#"
9017name = "us-south-001"
9018provider = "static"
9019mesh_tags = ["tag:cloud-runner"]
9020taints = ["no-appliance"]
9021"#;
9022 let m: MachineConfig = toml::from_str(toml_src).unwrap();
9023 assert_eq!(m.taints, vec!["no-appliance"]);
9024
9025 let s = toml::to_string(&m).unwrap();
9026 let back: MachineConfig = toml::from_str(&s).unwrap();
9027 assert_eq!(back.taints, vec!["no-appliance"]);
9028 }
9029
9030 #[test]
9031 fn machine_allocatable_absent_is_none() {
9032 let toml_src = "name = \"node\"\nprovider = \"static\"\nmesh_tags = []\n";
9033 let m: MachineConfig = toml::from_str(toml_src).unwrap();
9034 assert!(m.allocatable.is_none());
9035 assert!(m.taints.is_empty());
9036 }
9037
9038 #[test]
9039 fn machine_allocatable_skipped_when_none() {
9040 let m = make_machine("node", vec![]);
9041 let s = toml::to_string(&m).unwrap();
9042 assert!(
9043 !s.contains("allocatable"),
9044 "None allocatable must be omitted: {s}"
9045 );
9046 assert!(!s.contains("taints"), "empty taints must be omitted: {s}");
9047 }
9048
9049 #[test]
9050 fn machine_multiple_taints_round_trip() {
9051 let toml_src = r#"
9052name = "quarantined"
9053provider = "static"
9054mesh_tags = ["tag:build-worker"]
9055taints = ["no-server", "no-appliance", "no-job"]
9056"#;
9057 let m: MachineConfig = toml::from_str(toml_src).unwrap();
9058 assert_eq!(m.taints.len(), 3);
9059 assert!(m.taints.contains(&"no-server".to_string()));
9060 assert!(m.taints.contains(&"no-appliance".to_string()));
9061 assert!(m.taints.contains(&"no-job".to_string()));
9062 // R742-T4: every key here is one the scheduler reads. This fixture
9063 // used to carry `no-voter`, which none of them is.
9064 assert!(m.inert_taints().is_empty());
9065 }
9066
9067 // ─── R742-T4 (W305): inert-taint classification ─────────────────────────
9068
9069 #[test]
9070 fn every_archetype_repel_key_is_live() {
9071 for arch in LifecycleArchetype::ALL {
9072 let key = format!("no-{}", arch.taint_key());
9073 assert_eq!(
9074 taint_effect(&key),
9075 TaintEffect::Repels(arch),
9076 "{key} must repel {arch:?}"
9077 );
9078 }
9079 }
9080
9081 #[test]
9082 fn public_ip_is_an_affinity_key_not_an_inert_one() {
9083 assert_eq!(
9084 taint_effect(workload_spec::PUBLIC_IP_TAINT),
9085 TaintEffect::Attracts
9086 );
9087 }
9088
9089 #[test]
9090 fn a_free_form_taint_is_inert_and_says_so() {
9091 // W305's headline example: `taints = ["qa"]` parsed clean and did
9092 // nothing. Environment is not expressible as a taint.
9093 assert_eq!(taint_effect("qa"), TaintEffect::Inert);
9094 // And the one that actually cost fleet state: `no-voter` reads as an
9095 // exclusion and excludes nothing — "voter" is not an archetype.
9096 assert_eq!(taint_effect("no-voter"), TaintEffect::Inert);
9097 // A near-miss on a real key is inert too, not silently forgiven.
9098 assert_eq!(taint_effect("no-servers"), TaintEffect::Inert);
9099
9100 let m = make_machine_with_capacity(
9101 "dev-pi",
9102 8192,
9103 4000,
9104 vec!["no-appliance", "no-voter", "qa"],
9105 );
9106 assert_eq!(m.inert_taints(), vec!["no-voter", "qa"]);
9107 }
9108
9109 // ─── R876-B7: repel-by-default + declarable toleration ──────────────────
9110
9111 /// The headline inversion. A bare `RequiredSpec` — which is exactly what
9112 /// deserializing a mirror's `required = { regions, mesh_tags }` produces,
9113 /// since no TOML in the tree writes `tolerates` — is now repelled by a
9114 /// repelling taint. Before B7 it matched, because repulsion was conditional
9115 /// on a `#[serde(skip)]` field that this path could never fill.
9116 #[test]
9117 fn an_undeclared_spec_is_repelled_by_a_repelling_taint() {
9118 let tainted = make_machine_with_capacity("n", 8192, 4000, vec!["no-server"]);
9119 assert!(!RequiredSpec::default().matches(&tainted));
9120
9121 // And it is the DESERIALIZED shape that matters, not a hand-built one:
9122 // this is the mirror path reproduced exactly.
9123 let from_toml: RequiredSpec =
9124 toml::from_str("regions = [\"us-east\"]\n").expect("a mirror-shaped required parses");
9125 assert!(from_toml.tolerates.is_empty());
9126 let mut in_region = tainted.clone();
9127 in_region.region = Some("us-east".to_string());
9128 assert!(
9129 !from_toml.matches(&in_region),
9130 "a mirror-declared placement must now read machine.taints"
9131 );
9132 }
9133
9134 /// The opt-back-in half, and the one an operator writes by hand.
9135 #[test]
9136 fn an_explicit_toleration_admits_the_tainted_machine_again() {
9137 let tainted = make_machine_with_capacity("n", 8192, 4000, vec!["no-server"]);
9138 let spec = RequiredSpec {
9139 tolerates: vec!["no-server".to_string()],
9140 ..Default::default()
9141 };
9142 assert!(spec.matches(&tainted));
9143
9144 // Per-key, not a blanket pass: tolerating one repelling key says nothing
9145 // about another.
9146 let both = make_machine_with_capacity("n", 8192, 4000, vec!["no-server", "no-appliance"]);
9147 assert!(!spec.matches(&both));
9148
9149 // And it deserializes — the whole point of replacing a `#[serde(skip)]`
9150 // field is that a mirror can now declare this.
9151 let from_toml: RequiredSpec = toml::from_str("tolerates = [\"no-server\"]\n")
9152 .expect("a slot can declare a toleration");
9153 assert!(from_toml.matches(&tainted));
9154 }
9155
9156 /// An untainted machine is unaffected, which is what makes the migration
9157 /// bounded: six of the nine fleet machines carry no repelling taint at all.
9158 #[test]
9159 fn an_untainted_machine_matches_exactly_as_before() {
9160 let clean = make_machine_with_capacity("n", 8192, 4000, vec![]);
9161 assert!(RequiredSpec::default().matches(&clean));
9162 assert!(RequiredSpec {
9163 tolerates: vec!["no-server".to_string()],
9164 ..Default::default()
9165 }
9166 .matches(&clean));
9167 }
9168
9169 /// THE MIGRATION'S LOAD-BEARING FACT. `public-ip` is on three fleet nodes
9170 /// including us-east-001, the only origin serving the yah.dev apex. It is an
9171 /// *affinity* key, so repel-by-default must not touch it — reading every
9172 /// taint as repulsion would evict the apex on the next apply.
9173 #[test]
9174 fn an_affinity_taint_does_not_repel() {
9175 let public = make_machine_with_capacity("us-east-001", 8192, 4000, vec!["public-ip"]);
9176 assert!(
9177 RequiredSpec::default().matches(&public),
9178 "public-ip attracts; it must never be read as repulsion"
9179 );
9180 }
9181
9182 /// `select_matching` filters on the same predicate, so a tainted machine
9183 /// leaves the candidate set rather than being silently placed onto.
9184 #[test]
9185 fn select_matching_drops_a_tainted_candidate_and_keeps_the_rest() {
9186 let drained = make_machine_with_capacity("drained", 8192, 4000, vec!["no-server"]);
9187 let healthy = make_machine_with_capacity("healthy", 8192, 4000, vec![]);
9188 let pool = [&drained, &healthy];
9189
9190 let picked = select_matching(&pool, &RequiredSpec::default(), 1, "test pool", "empty")
9191 .expect("one candidate remains");
9192 assert_eq!(
9193 picked.iter().map(|m| m.name.as_str()).collect::<Vec<_>>(),
9194 vec!["healthy"],
9195 "tainting the first candidate moves the placement to the second"
9196 );
9197
9198 // Asking for both is a shortfall, not a half-placement.
9199 let err = select_matching(&pool, &RequiredSpec::default(), 2, "test pool", "empty")
9200 .expect_err("only one of two matches");
9201 assert!(format!("{err:#}").contains("only 1 of 2 machines match"));
9202 }
9203
9204 /// The `admit_workload` path must be behaviourally unchanged: its spec is
9205 /// built by `admission_spec`, which now emits the complementary tolerations.
9206 #[test]
9207 fn admission_preserves_archetype_scoped_repulsion_across_the_inversion() {
9208 let ws = minimal_spec("srv", 1); // a Server
9209 let req = admission_spec(&ws, &[]);
9210 assert_eq!(ws.effective_archetype(), LifecycleArchetype::Server);
9211
9212 let no_server = make_machine_with_capacity("n", 8192, 4000, vec!["no-server"]);
9213 let no_appliance = make_machine_with_capacity("n", 8192, 4000, vec!["no-appliance"]);
9214 assert!(!req.matches(&no_server), "its own class still repels it");
9215 assert!(
9216 req.matches(&no_appliance),
9217 "another class's taint still does not — this is the pre-B7 answer"
9218 );
9219 }
9220
9221 #[test]
9222 fn describe_names_the_toleration_so_a_refusal_is_readable() {
9223 let spec = RequiredSpec {
9224 regions: vec!["us-west".to_string()],
9225 tolerates: vec!["no-appliance".to_string()],
9226 ..Default::default()
9227 };
9228 assert_eq!(
9229 spec.describe(),
9230 "required.regions=[us-west] + required.tolerates=[no-appliance]"
9231 );
9232 }
9233
9234 /// A toleration widens; it must not make an underspecified slot look
9235 /// specified, or the deploy side stops refusing one.
9236 #[test]
9237 fn a_toleration_alone_is_still_an_unconstrained_spec() {
9238 assert!(RequiredSpec {
9239 tolerates: vec!["no-server".to_string()],
9240 ..Default::default()
9241 }
9242 .is_unconstrained());
9243 }
9244
9245 #[test]
9246 fn an_inert_taint_changes_no_placement_decision() {
9247 // The reason this is a lint and not a behaviour change: the guard's
9248 // whole premise is that these keys are invisible to `matches`.
9249 let clean = make_machine_with_capacity("n", 8192, 4000, vec![]);
9250 let noisy = make_machine_with_capacity("n", 8192, 4000, vec!["no-voter", "qa"]);
9251 // R876-B7: still true under repel-by-default, and for a sharper reason —
9252 // `matches` now walks `machine.taints` itself, so an unclassifiable key
9253 // is skipped by `taint_effect` rather than merely never looked up.
9254 for arch in LifecycleArchetype::ALL {
9255 let req = RequiredSpec {
9256 tolerates: tolerations_excluding(&[arch]),
9257 ..Default::default()
9258 };
9259 assert_eq!(req.matches(&clean), req.matches(&noisy));
9260 assert!(req.matches(&noisy), "neither key repels");
9261 }
9262 }
9263
9264 /// The toleration set [`admission_spec`] derives for a group of `archetypes`
9265 /// — every repelling key that is not the group's own class.
9266 fn tolerations_excluding(archetypes: &[LifecycleArchetype]) -> Vec<String> {
9267 LifecycleArchetype::ALL
9268 .into_iter()
9269 .filter(|a| !archetypes.contains(a))
9270 .map(|a| format!("no-{}", a.taint_key()))
9271 .collect()
9272 }
9273
9274 #[test]
9275 fn live_taint_keys_lists_the_whole_legal_vocabulary() {
9276 assert_eq!(
9277 live_taint_keys(),
9278 vec!["no-appliance", "no-job", "no-server", "public-ip"]
9279 );
9280 }
9281
9282 // ─── R742-F1 (W305): sovereign groups ───────────────────────────────────
9283
9284 /// A machine in `group`, with the role left unwritten — which is the state
9285 /// of every machine TOML that predates R605-F12 and resolves to `voter`.
9286 fn in_group(name: &str, group: Option<&str>) -> MachineConfig {
9287 MachineConfig {
9288 sovereign_group: group.map(String::from),
9289 ..make_machine(name, vec![])
9290 }
9291 }
9292
9293 /// A machine in `group` with its quorum eligibility stated (R605-F12).
9294 fn in_group_as(name: &str, group: &str, role: SovereignRole) -> MachineConfig {
9295 MachineConfig {
9296 sovereign_group: Some(group.to_string()),
9297 sovereign_role: Some(role),
9298 ..make_machine(name, vec![])
9299 }
9300 }
9301
9302 #[test]
9303 fn a_join_within_one_sovereign_group_is_permitted() {
9304 assert_eq!(
9305 judge_join(
9306 &in_group("us-west-013", Some("dev")),
9307 &in_group("us-west-011", Some("dev")),
9308 ),
9309 JoinVerdict::Permit
9310 );
9311 }
9312
9313 /// The case the field exists for: before it, the only thing standing
9314 /// between a dev Pi and the prod quorum was a comment in a TOML.
9315 #[test]
9316 fn a_cross_group_join_is_refused_naming_both_groups() {
9317 let verdict = judge_join(
9318 &in_group("us-west-011", Some("dev")),
9319 &in_group("us-west-001", Some("prod")),
9320 );
9321 let JoinVerdict::Refuse(msg) = verdict else {
9322 panic!("a dev node joining prod must be refused: {verdict:?}");
9323 };
9324 // A refusal that does not name what it saw is one the operator has to
9325 // go and reconstruct, so it gets worked around instead of fixed.
9326 assert!(msg.contains("us-west-011") && msg.contains("us-west-001"), "{msg}");
9327 assert!(msg.contains("dev") && msg.contains("prod"), "{msg}");
9328 }
9329
9330 /// `None` is a declaration ("standalone, in no group"), not a gap — so
9331 /// growing prod with an unstamped box is a cross-group join too, and the
9332 /// refusal has to say which file makes it legal.
9333 #[test]
9334 fn an_undeclared_node_cannot_join_a_declared_group() {
9335 let verdict = judge_join(
9336 &in_group("us-west-002", None),
9337 &in_group("us-west-001", Some("prod")),
9338 );
9339 let JoinVerdict::Refuse(msg) = verdict else {
9340 panic!("an unstamped node joining prod must be refused: {verdict:?}");
9341 };
9342 assert!(
9343 msg.contains(".yah/infra/machines/us-west-002.toml"),
9344 "the refusal must name the file to stamp: {msg}"
9345 );
9346 }
9347
9348 #[test]
9349 fn a_declared_node_cannot_join_a_standalone_target() {
9350 // us-west-003 is `mode: standalone` on purpose; it is not a group of
9351 // one waiting to be grown.
9352 let verdict = judge_join(
9353 &in_group("us-west-001", Some("prod")),
9354 &in_group("us-west-003", None),
9355 );
9356 assert!(matches!(verdict, JoinVerdict::Refuse(msg) if msg.contains("us-west-003")));
9357 }
9358
9359 #[test]
9360 fn two_undeclared_nodes_cannot_form_an_undeclared_group() {
9361 let verdict = judge_join(
9362 &in_group("us-west-002", None),
9363 &in_group("us-west-015", None),
9364 );
9365 assert!(
9366 matches!(&verdict, JoinVerdict::Refuse(msg) if msg.contains("us-west-002")
9367 && msg.contains("us-west-015")),
9368 "forming a group nobody declared must be refused, naming both: {verdict:?}"
9369 );
9370 }
9371
9372 // ─── R605-F12: the voting axis ──────────────────────────────────────────
9373
9374 /// The whole ticket in one assertion. us-west-003 is a member of prod —
9375 /// same secrets, same upgrade cadence, same destruction — and must never
9376 /// hold a prod raft seat. Before the role axis, the only thing refusing it
9377 /// was its *absent* group stamp, so writing down the truth above would have
9378 /// removed the guard.
9379 #[test]
9380 fn a_non_voting_member_is_refused_into_its_own_group() {
9381 let verdict = judge_join(
9382 &in_group_as("us-west-003", "prod", SovereignRole::NonVoter),
9383 &in_group_as("us-west-001", "prod", SovereignRole::Voter),
9384 );
9385 let JoinVerdict::Refuse(msg) = verdict else {
9386 panic!("a non-voting prod member must not join the prod quorum: {verdict:?}");
9387 };
9388 assert!(msg.contains("us-west-003") && msg.contains("NON-VOTING"), "{msg}");
9389 // The refusal must not blame the group: both sides say "prod", and a
9390 // cross-group message here would read as a bug in the check itself.
9391 assert!(!msg.contains("cross-group"), "{msg}");
9392 assert!(
9393 msg.contains(".yah/infra/machines/us-west-003.toml"),
9394 "the refusal must name the file that decides it: {msg}"
9395 );
9396 }
9397
9398 /// Read from the other end: a box declared non-voting has no quorum seat to
9399 /// be grown, so it cannot be a join target either.
9400 #[test]
9401 fn a_non_voting_target_has_no_quorum_to_grow() {
9402 let verdict = judge_join(
9403 &in_group_as("us-west-001", "prod", SovereignRole::Voter),
9404 &in_group_as("us-west-003", "prod", SovereignRole::NonVoter),
9405 );
9406 assert!(
9407 matches!(&verdict, JoinVerdict::Refuse(msg) if msg.contains("the target")
9408 && msg.contains("us-west-003")),
9409 "{verdict:?}"
9410 );
9411 }
9412
9413 /// A non-voter joining a *standalone* target is refused for two reasons at
9414 /// once, and the message must pick the one whose fix would actually work.
9415 /// Naming the role here would send the operator to flip `sovereign_role`
9416 /// and come back to the same refusal.
9417 #[test]
9418 fn a_refusal_names_the_group_when_fixing_the_role_would_not_help() {
9419 let verdict = judge_join(
9420 &in_group_as("us-west-003", "prod", SovereignRole::NonVoter),
9421 &in_group("us-west-002", None),
9422 );
9423 let JoinVerdict::Refuse(msg) = verdict else {
9424 panic!("a standalone target has no group to join: {verdict:?}");
9425 };
9426 assert!(
9427 msg.contains(".yah/infra/machines/us-west-002.toml"),
9428 "the refusal must point at the target's missing group stamp: {msg}"
9429 );
9430 assert!(!msg.contains("NON-VOTING"), "{msg}");
9431 }
9432
9433 /// The back-compat seam, pinned: the six nodes stamped before R605-F12
9434 /// write no role, and an absent role means what declaring a group has
9435 /// always meant. If this flips, the live prod and dev quorums stop being
9436 /// growable on a config the operator never edited.
9437 #[test]
9438 fn an_unwritten_role_still_joins_its_group() {
9439 let joiner = in_group("us-west-013", Some("dev"));
9440 assert_eq!(joiner.sovereign_role, None);
9441 assert_eq!(
9442 judge_join(&joiner, &in_group("us-west-011", Some("dev"))),
9443 JoinVerdict::Permit
9444 );
9445 assert_eq!(
9446 judge_join(
9447 &joiner,
9448 &in_group_as("us-west-011", "dev", SovereignRole::Voter)
9449 ),
9450 JoinVerdict::Permit
9451 );
9452 }
9453
9454 /// A non-voting member is still a *member*, and the two claims must not be
9455 /// conflated: `sovereign_membership()` reports the group either way, so a
9456 /// consumer asking "is this box in prod's blast radius" gets yes.
9457 #[test]
9458 fn a_non_voter_is_still_in_the_group_it_names() {
9459 let m = in_group_as("us-west-003", "prod", SovereignRole::NonVoter);
9460 assert_eq!(m.sovereign_membership().group, Some("prod"));
9461 assert!(!m.sovereign_membership().role.is_voter());
9462
9463 // …and the group-membership query the fleet reads is unaffected by it.
9464 let cfg = make_empty_cfg(vec![
9465 m,
9466 in_group_as("us-west-001", "prod", SovereignRole::Voter),
9467 ]);
9468 assert_eq!(
9469 cfg.machines_in_group("prod")
9470 .iter()
9471 .map(|m| m.name.as_str())
9472 .collect::<Vec<_>>(),
9473 vec!["us-west-003", "us-west-001"]
9474 );
9475 }
9476
9477 /// The role travels through TOML in one spelling, and an absent one stays
9478 /// absent on the way back out — otherwise every machine file would grow a
9479 /// `sovereign_role = "voter"` line the operator never wrote, and the
9480 /// unroled-member lint would have nothing left to find.
9481 #[test]
9482 fn sovereign_role_round_trips_and_is_omitted_when_unwritten() {
9483 let m: MachineConfig = toml::from_str(
9484 r#"
9485name = "us-west-003"
9486provider = "static"
9487region = "us-west"
9488arch = "x86_64"
9489mesh_tags = []
9490sovereign_group = "prod"
9491sovereign_role = "non-voter"
9492"#,
9493 )
9494 .unwrap();
9495 assert_eq!(m.sovereign_role, Some(SovereignRole::NonVoter));
9496 assert!(toml::to_string(&m)
9497 .unwrap()
9498 .contains(r#"sovereign_role = "non-voter""#));
9499
9500 let unwritten = MachineConfig {
9501 sovereign_role: None,
9502 ..m
9503 };
9504 assert!(!toml::to_string(&unwritten)
9505 .unwrap()
9506 .contains("sovereign_role"));
9507 }
9508
9509 /// The invariant the ticket is most explicit about: a sovereign group is a
9510 /// blast radius, not a filter. If this ever fails, `matches` has grown an
9511 /// axis it must not have and dev-mode workloads have silently become
9512 /// unschedulable on the dev group.
9513 #[test]
9514 fn sovereign_group_is_not_a_placement_input() {
9515 let standalone = in_group("n", None);
9516 let grouped = in_group("n", Some("dev"));
9517 let other = in_group("n", Some("prod"));
9518
9519 for spec in [
9520 RequiredSpec::default(),
9521 RequiredSpec {
9522 regions: vec!["us-west".into()],
9523 ..Default::default()
9524 },
9525 RequiredSpec {
9526 tolerates: tolerations_excluding(&[LifecycleArchetype::Appliance]),
9527 ..Default::default()
9528 },
9529 ] {
9530 let baseline = spec.matches(&standalone);
9531 assert_eq!(spec.matches(&grouped), baseline);
9532 assert_eq!(spec.matches(&other), baseline);
9533 }
9534 }
9535
9536 // ─── R742-F3 (W305): group → machine set, and group-scoped admission ────
9537
9538 /// The primitive `migrate --to` needs and `rollout plan` still lacks
9539 /// (W314 gap 1): a group exists only as the set of machines naming it, so
9540 /// membership has to be derived rather than declared anywhere.
9541 #[test]
9542 fn machines_in_group_derives_membership_from_the_declarations() {
9543 let cfg = make_empty_cfg(vec![
9544 in_group("us-west-001", Some("prod")),
9545 in_group("us-west-011", Some("dev")),
9546 in_group("us-west-013", Some("dev")),
9547 in_group("us-west-002", None),
9548 ]);
9549
9550 let dev: Vec<&str> = cfg
9551 .machines_in_group("dev")
9552 .iter()
9553 .map(|m| m.name.as_str())
9554 .collect();
9555 assert_eq!(dev, vec!["us-west-011", "us-west-013"]);
9556 assert_eq!(cfg.machines_in_group("prod").len(), 1);
9557
9558 // Standalone is "in no group", not "in a group called none" — so an
9559 // unstamped box is never swept into a migration target.
9560 assert!(cfg.machines_in_group("").is_empty());
9561 assert!(cfg.machines_in_group("staging").is_empty());
9562 }
9563
9564 #[test]
9565 fn declared_sovereign_groups_is_the_vocabulary_a_bad_target_is_named_against() {
9566 let cfg = make_empty_cfg(vec![
9567 in_group("a", Some("prod")),
9568 in_group("b", Some("dev")),
9569 in_group("c", Some("prod")),
9570 in_group("d", None),
9571 ]);
9572 // Sorted + deduped, and standalone contributes nothing.
9573 assert_eq!(cfg.declared_sovereign_groups(), vec!["dev", "prod"]);
9574 assert!(make_empty_cfg(vec![in_group("a", None)])
9575 .declared_sovereign_groups()
9576 .is_empty());
9577 }
9578
9579 /// Group-scoped admission must be the SAME predicate as unscoped
9580 /// admission, only over fewer candidates. If it ever diverges, `migrate`
9581 /// becomes a way to place a workload somewhere `yah cloud apply` would
9582 /// refuse — which is exactly the silent routing-around W305 exists to stop.
9583 #[test]
9584 fn admit_workload_in_group_narrows_candidates_without_changing_the_predicate() {
9585 let mut prod = in_group("us-west-001", Some("prod"));
9586 prod.mesh_tags = vec!["tag:cloud-runner".into()];
9587 let mut dev_repels = in_group("us-west-011", Some("dev"));
9588 dev_repels.taints = vec!["no-appliance".into()];
9589 let mut dev_ok = in_group("us-west-013", Some("dev"));
9590 dev_ok.mesh_tags = vec!["tag:cloud-runner".into()];
9591
9592 let cfg = make_empty_cfg(vec![prod, dev_repels, dev_ok]);
9593
9594 let mut ws = ws_with_selector(None);
9595 ws.archetype = Some(LifecycleArchetype::Appliance);
9596
9597 // Unscoped picks the first match in declaration order.
9598 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "us-west-001");
9599 // Scoped skips the repelling dev node and lands on the other one —
9600 // the taint is honoured, not bypassed.
9601 assert_eq!(
9602 cfg.admit_workload_in_group(&ws, "dev").unwrap().name,
9603 "us-west-013"
9604 );
9605 }
9606
9607 #[test]
9608 fn admit_workload_in_group_distinguishes_an_empty_group_from_a_repelling_one() {
9609 let mut dev = in_group("us-west-011", Some("dev"));
9610 dev.taints = vec!["no-appliance".into()];
9611 let cfg = make_empty_cfg(vec![in_group("us-west-001", Some("prod")), dev]);
9612
9613 let mut ws = ws_with_selector(None);
9614 ws.archetype = Some(LifecycleArchetype::Appliance);
9615
9616 // A group nobody declares names the legal vocabulary, because a typo
9617 // is the realistic cause and "no candidates" would send the operator
9618 // hunting for a placement problem that does not exist.
9619 let missing = cfg.admit_workload_in_group(&ws, "stagng").unwrap_err().to_string();
9620 assert!(missing.contains("no machine declares"), "{missing}");
9621 assert!(missing.contains("dev") && missing.contains("prod"), "{missing}");
9622
9623 // A group that exists but refuses names the machines it tried.
9624 let repelled = cfg.admit_workload_in_group(&ws, "dev").unwrap_err().to_string();
9625 assert!(repelled.contains("us-west-011"), "{repelled}");
9626 }
9627
9628 #[test]
9629 fn sovereign_group_round_trips_and_is_omitted_when_standalone() {
9630 let src = r#"
9631name = "us-west-011"
9632provider = "static"
9633mesh_tags = []
9634sovereign_group = "dev"
9635"#;
9636 let m: MachineConfig = toml::from_str(src).unwrap();
9637 assert_eq!(m.sovereign_group.as_deref(), Some("dev"));
9638 assert!(toml::to_string(&m).unwrap().contains("sovereign_group"));
9639
9640 // A machine that predates the field parses as standalone and does not
9641 // grow the key back on write.
9642 let legacy: MachineConfig =
9643 toml::from_str("name = \"us-west-002\"\nprovider = \"static\"\nmesh_tags = []\n")
9644 .unwrap();
9645 assert_eq!(legacy.sovereign_group, None);
9646 assert!(!toml::to_string(&legacy).unwrap().contains("sovereign_group"));
9647 }
9648
9649 // ─── R572-F5: capacity floor + absolute (untolerable) taints ────────────
9650
9651 fn make_machine_with_capacity(
9652 name: &str,
9653 memory_mb: u32,
9654 cpu_millis: u32,
9655 taints: Vec<&str>,
9656 ) -> MachineConfig {
9657 MachineConfig {
9658 allocatable: Some(NodeAllocatable {
9659 memory_mb,
9660 cpu_millis,
9661 }),
9662 taints: taints.into_iter().map(String::from).collect(),
9663 ..make_machine(name, vec![])
9664 }
9665 }
9666
9667 fn server_spec(memory_mb: u32, cpu_millis: u32) -> WorkloadSpec {
9668 use workload_spec::{ImageRef, LifecycleArchetype, ResourceLimits, TierTag};
9669 let mut ws = WorkloadSpec::for_forge(
9670 "f5-test",
9671 ImageRef {
9672 registry: "localhost".into(),
9673 repository: "test".into(),
9674 tag: "latest".into(),
9675 digest: workload_spec::testing::test_digest(),
9676 },
9677 TierTag("infra".into()),
9678 vec![],
9679 );
9680 ws.archetype = Some(LifecycleArchetype::Server);
9681 ws.resources = ResourceLimits {
9682 memory_mb,
9683 cpu_millis,
9684 ephemeral_storage_mb: 0,
9685 };
9686 // These are SERVER specs that borrow `for_forge` as a constructor
9687 // shortcut, so drop the forge memory request it stamps on — otherwise
9688 // every spec here silently requests the forge default instead of the
9689 // `memory_mb` the caller passed, and the capacity-floor tests below
9690 // stop testing their own argument. A server workload declares no
9691 // request, which is the documented fall-back-to-`resources.memory_mb`
9692 // path (`WorkloadSpec::memory_request_mb`).
9693 ws.annotations
9694 .remove(workload_spec::MEMORY_REQUEST_ANNOTATION);
9695 ws
9696 }
9697
9698 fn appliance_spec_ws(memory_mb: u32, cpu_millis: u32) -> WorkloadSpec {
9699 use workload_spec::LifecycleArchetype;
9700 let mut ws = server_spec(memory_mb, cpu_millis);
9701 ws.archetype = Some(LifecycleArchetype::Appliance);
9702 ws
9703 }
9704
9705 #[test]
9706 fn capacity_floor_rejects_undersized_node() {
9707 let cfg = make_empty_cfg(vec![make_machine_with_capacity("small", 256, 500, vec![])]);
9708 let ws = server_spec(512, 1000); // demands more than available
9709 assert!(cfg.admit_workload(&ws).is_err());
9710 }
9711
9712 #[test]
9713 fn capacity_floor_accepts_exact_fit() {
9714 let cfg = make_empty_cfg(vec![make_machine_with_capacity("exact", 512, 1000, vec![])]);
9715 let ws = server_spec(512, 1000);
9716 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "exact");
9717 }
9718
9719 #[test]
9720 fn capacity_floor_passes_when_allocatable_absent() {
9721 // A machine with no allocatable block skips the capacity check (no data).
9722 let cfg = make_empty_cfg(vec![make_machine("no-alloc", vec![])]);
9723 let ws = server_spec(99999, 99999); // would exceed any real node
9724 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "no-alloc");
9725 }
9726
9727 #[test]
9728 fn taint_repulsion_blocks_appliance_on_no_appliance_node() {
9729 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
9730 "south",
9731 1024,
9732 2000,
9733 vec!["no-appliance"],
9734 )]);
9735 let ws = appliance_spec_ws(256, 500);
9736 assert!(
9737 cfg.admit_workload(&ws).is_err(),
9738 "appliance must be repelled by no-appliance taint"
9739 );
9740 }
9741
9742 #[test]
9743 fn taint_repulsion_allows_server_on_no_appliance_node() {
9744 // "no-appliance" only repels Appliance workloads; servers are unaffected.
9745 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
9746 "south",
9747 1024,
9748 2000,
9749 vec!["no-appliance"],
9750 )]);
9751 let ws = server_spec(256, 500);
9752 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "south");
9753 }
9754
9755 #[test]
9756 fn taint_repulsion_job_not_blocked_by_no_server() {
9757 use workload_spec::LifecycleArchetype;
9758 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
9759 "build-box",
9760 8192,
9761 4000,
9762 vec!["no-server", "no-appliance"],
9763 )]);
9764 let mut ws = server_spec(256, 500);
9765 ws.archetype = Some(LifecycleArchetype::Job);
9766 // Job only repelled by "no-job"; "no-server" and "no-appliance" don't affect it.
9767 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "build-box");
9768 }
9769
9770 #[test]
9771 fn requires_taint_affinity_blocks_placement_without_it() {
9772 use workload_spec::{LifecycleArchetype, PUBLIC_IP_TAINT, REQUIRES_TAINT_ANNOTATION};
9773 // Simulate the passway ingress appliance: requires "public-ip" taint.
9774 let mut ws = appliance_spec_ws(256, 512);
9775 ws.archetype = Some(LifecycleArchetype::Appliance);
9776 ws.annotations
9777 .insert(REQUIRES_TAINT_ANNOTATION.into(), PUBLIC_IP_TAINT.into());
9778
9779 // Node without the taint: rejected.
9780 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
9781 "no-pip",
9782 2048,
9783 2000,
9784 vec![],
9785 )]);
9786 assert!(cfg.admit_workload(&ws).is_err());
9787
9788 // Node with the taint: accepted.
9789 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
9790 "pub-node",
9791 2048,
9792 2000,
9793 vec!["public-ip"],
9794 )]);
9795 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "pub-node");
9796 }
9797
9798 #[test]
9799 fn w244_fleet_scenario_appliance_rejected_from_south_and_west002() {
9800 // Full W244 fleet table scenario:
9801 // us-west-001/east-001: no taints, large capacity → appliance lands here
9802 // us-south-001: no-appliance taint → appliance rejected
9803 // us-west-002: no-server, no-appliance → appliance rejected
9804 let cfg = make_empty_cfg(vec![
9805 make_machine_with_capacity("us-south-001", 512, 1000, vec!["no-appliance"]),
9806 make_machine_with_capacity(
9807 "us-west-002",
9808 16384,
9809 8000,
9810 vec!["no-server", "no-appliance"],
9811 ),
9812 make_machine_with_capacity("us-west-001", 4096, 4000, vec![]),
9813 ]);
9814 let ws = appliance_spec_ws(256, 500);
9815 // Skips south (no-appliance) and west-002 (no-appliance), lands on west-001.
9816 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "us-west-001");
9817 }
9818
9819 #[test]
9820 fn w244_fleet_scenario_job_lands_on_west002_first() {
9821 use workload_spec::LifecycleArchetype;
9822 // Jobs should prefer (or at least land on) the job-only box.
9823 let cfg = make_empty_cfg(vec![
9824 make_machine_with_capacity("us-west-001", 4096, 4000, vec![]),
9825 make_machine_with_capacity(
9826 "us-west-002",
9827 16384,
9828 8000,
9829 vec!["no-server", "no-appliance"],
9830 ),
9831 ]);
9832 let mut ws = server_spec(256, 500);
9833 ws.archetype = Some(LifecycleArchetype::Job);
9834 // No fleet node declares `no-job`, so a Job is repelled by nothing;
9835 // west-001 comes first in declaration order (greedy, no preference),
9836 // which is the expected tie-break. Note this is *absence of a repel
9837 // key*, not toleration — no workload can tolerate a taint (W305).
9838 let picked = cfg.admit_workload(&ws).unwrap();
9839 // Both are eligible.
9840 assert!(
9841 picked.name == "us-west-001" || picked.name == "us-west-002",
9842 "job must land on an eligible node, got {}",
9843 picked.name
9844 );
9845 }
9846
9847 #[test]
9848 fn r569_f4_macos_node_taints_keep_cloud_critical_off_but_admit_build_jobs() {
9849 use workload_spec::LifecycleArchetype;
9850 // R569-F4: the headless M2 (us-west-015) joins the fleet as a
9851 // build-worker but must never take cloud-critical load. It carries the
9852 // same repel set as the x86 build-worker (`no-server, no-appliance` —
9853 // see .yah/infra/machines/us-west-015.toml). This pins that intent:
9854 // with a plain cloud node available beside the Mac, every
9855 // cloud-critical archetype lands on the cloud node and never the Mac;
9856 // build Jobs (the Mac's actual purpose) remain eligible on it.
9857 //
9858 // R742-T4: `no-voter` used to sit in this set and in the TOML. It was
9859 // never read here — there is no "voter" workload archetype — and
9860 // R569-F3's learner-only join is what actually keeps the box out of
9861 // quorum. It is now rejected by `yah cloud validate` as inert.
9862 let mac_taints = vec!["no-server", "no-appliance"];
9863 let fleet = || {
9864 make_empty_cfg(vec![
9865 make_machine_with_capacity("us-west-015", 24576, 8000, mac_taints.clone()),
9866 make_machine_with_capacity("us-west-001", 4096, 4000, vec![]),
9867 ])
9868 };
9869
9870 // A cloud-critical Server workload is repelled from the Mac and lands
9871 // on the untainted cloud node.
9872 let cfg = fleet();
9873 assert_eq!(
9874 cfg.admit_workload(&server_spec(256, 500)).unwrap().name,
9875 "us-west-001",
9876 "a Server workload must never land on the no-server Mac node"
9877 );
9878
9879 // Same for an Appliance (pinned/stateful cloud-critical) workload.
9880 let cfg = fleet();
9881 assert_eq!(
9882 cfg.admit_workload(&appliance_spec_ws(256, 500))
9883 .unwrap()
9884 .name,
9885 "us-west-001",
9886 "an Appliance workload must never land on the no-appliance Mac node"
9887 );
9888
9889 // Sharpest repulsion proof: with ONLY the Mac in the fleet, a
9890 // cloud-critical Server workload is rejected outright — the taint keeps
9891 // it off even when that means nowhere to run.
9892 let mac_only = make_empty_cfg(vec![make_machine_with_capacity(
9893 "us-west-015",
9894 24576,
9895 8000,
9896 mac_taints.clone(),
9897 )]);
9898 assert!(
9899 mac_only.admit_workload(&server_spec(256, 500)).is_err(),
9900 "a Server workload must be repelled from a Mac-only fleet, not admitted"
9901 );
9902
9903 // But the Mac's real job — build/forge workloads — IS admitted on it:
9904 // it tolerates every fleet taint (there is no `no-job`).
9905 let mut job = server_spec(256, 500);
9906 job.archetype = Some(LifecycleArchetype::Job);
9907 assert_eq!(
9908 mac_only.admit_workload(&job).unwrap().name,
9909 "us-west-015",
9910 "a build Job must still be admitted on the Mac build-worker"
9911 );
9912 }
9913
9914 // ─── R615-F1: linked infra sources (`.yah/infra/sources.toml`) ─────────
9915
9916 #[test]
9917 fn sources_load_is_empty_when_the_file_is_absent() {
9918 // "Every camp without linked infra has none" — which today is every
9919 // camp — must not be an error.
9920 let tmp = tempfile::TempDir::new().unwrap();
9921 let cfg = SourcesConfig::load(tmp.path()).unwrap();
9922 assert_eq!(cfg, SourcesConfig::default());
9923 assert!(cfg.source.is_empty());
9924 assert_eq!(cfg.schema_version, 1);
9925 }
9926
9927 #[test]
9928 fn sources_parses_a_path_kind_exactly_like_w274s_example() {
9929 let tmp = tempfile::TempDir::new().unwrap();
9930 std::fs::write(
9931 tmp.path().join("sources.toml"),
9932 r#"
9933schema_version = 1
9934
9935[[source]]
9936owner = "yah"
9937kind = "path"
9938path = "../yah"
9939mode = "read-only"
9940"#,
9941 )
9942 .unwrap();
9943 let cfg = SourcesConfig::load(tmp.path()).unwrap();
9944 assert_eq!(cfg.source.len(), 1);
9945 let s = &cfg.source[0];
9946 assert_eq!(s.owner, "yah");
9947 assert_eq!(s.mode, SourceMode::ReadOnly);
9948 assert!(s.select.is_empty());
9949 match &s.kind {
9950 InfraSourceKind::Path { path } => assert_eq!(path, "../yah"),
9951 other => panic!("expected Path, got {other:?}"),
9952 }
9953 }
9954
9955 #[test]
9956 fn sources_parses_a_git_kind_reusing_gitsource_verbatim() {
9957 let tmp = tempfile::TempDir::new().unwrap();
9958 std::fs::write(
9959 tmp.path().join("sources.toml"),
9960 r#"
9961schema_version = 1
9962
9963[[source]]
9964owner = "yah"
9965kind = "git"
9966repo = "git@github.com:yah-ai/infra.git"
9967ref = "main"
9968subdir = "infra"
9969select = ["tag:cloud-runner"]
9970mode = "read-only"
9971"#,
9972 )
9973 .unwrap();
9974 let cfg = SourcesConfig::load(tmp.path()).unwrap();
9975 assert_eq!(cfg.source.len(), 1);
9976 let s = &cfg.source[0];
9977 assert_eq!(s.select, vec!["tag:cloud-runner".to_string()]);
9978 match &s.kind {
9979 InfraSourceKind::Git(git) => {
9980 assert_eq!(git.repo, "git@github.com:yah-ai/infra.git");
9981 assert_eq!(git.r#ref, "main");
9982 assert_eq!(git.subdir.as_deref(), Some("infra"));
9983 }
9984 other => panic!("expected Git, got {other:?}"),
9985 }
9986 }
9987
9988 #[test]
9989 fn sources_mode_defaults_to_read_only_and_manage_is_explicit() {
9990 let tmp = tempfile::TempDir::new().unwrap();
9991 std::fs::write(
9992 tmp.path().join("sources.toml"),
9993 r#"
9994schema_version = 1
9995
9996[[source]]
9997owner = "a"
9998kind = "path"
9999path = "../a"
10000
10001[[source]]
10002owner = "b"
10003kind = "path"
10004path = "../b"
10005mode = "manage"
10006"#,
10007 )
10008 .unwrap();
10009 let cfg = SourcesConfig::load(tmp.path()).unwrap();
10010 assert_eq!(cfg.source[0].mode, SourceMode::ReadOnly, "omitted mode = read-only");
10011 assert_eq!(cfg.source[1].mode, SourceMode::Manage);
10012 }
10013
10014 #[test]
10015 fn sources_preserves_declaration_order() {
10016 // Overlay order matters (R615-F2) when two sources name the same
10017 // machine — the list must round-trip in file order, not be reordered
10018 // by owner or kind.
10019 let tmp = tempfile::TempDir::new().unwrap();
10020 std::fs::write(
10021 tmp.path().join("sources.toml"),
10022 r#"
10023schema_version = 1
10024
10025[[source]]
10026owner = "second"
10027kind = "path"
10028path = "../second"
10029
10030[[source]]
10031owner = "first"
10032kind = "path"
10033path = "../first"
10034"#,
10035 )
10036 .unwrap();
10037 let cfg = SourcesConfig::load(tmp.path()).unwrap();
10038 let owners: Vec<&str> = cfg.source.iter().map(|s| s.owner.as_str()).collect();
10039 assert_eq!(owners, vec!["second", "first"]);
10040 }
10041
10042 #[test]
10043 fn sources_round_trips_through_serialize() {
10044 let cfg = SourcesConfig {
10045 schema_version: 1,
10046 source: vec![
10047 InfraSource {
10048 owner: "yah".into(),
10049 kind: InfraSourceKind::Path {
10050 path: "../yah".into(),
10051 },
10052 mode: SourceMode::ReadOnly,
10053 select: vec![],
10054 },
10055 InfraSource {
10056 owner: "yah".into(),
10057 kind: InfraSourceKind::Git(GitSource {
10058 repo: "git@github.com:yah-ai/infra.git".into(),
10059 r#ref: "main".into(),
10060 subdir: Some("infra".into()),
10061 }),
10062 mode: SourceMode::Manage,
10063 select: vec!["tag:cloud-runner".into()],
10064 },
10065 ],
10066 };
10067 let toml_str = toml::to_string_pretty(&cfg).unwrap();
10068 let reloaded: SourcesConfig = toml::from_str(&toml_str).unwrap();
10069 assert_eq!(reloaded, cfg, "round-trip through TOML must be lossless:\n{toml_str}");
10070 }
10071
10072 // ─── R615-F2: overlay loader in CloudConfig::load ───────────────────────
10073
10074 fn write_min_machine(dir: &Path, name: &str, extra_toml: &str) {
10075 std::fs::create_dir_all(dir).unwrap();
10076 // `extra_toml` supplies `mesh_tags` when the caller cares about it;
10077 // otherwise default to the empty list. Never hardcode `mesh_tags`
10078 // here as well as in `extra_toml` -- TOML rejects a duplicate key.
10079 let mesh_tags = if extra_toml.contains("mesh_tags") {
10080 String::new()
10081 } else {
10082 "mesh_tags = []\n".to_string()
10083 };
10084 std::fs::write(
10085 dir.join(format!("{name}.toml")),
10086 format!("name = \"{name}\"\nprovider = \"static\"\n{mesh_tags}{extra_toml}"),
10087 )
10088 .unwrap();
10089 }
10090
10091 fn write_min_provider(dir: &Path, id: &str) {
10092 std::fs::create_dir_all(dir).unwrap();
10093 std::fs::write(
10094 dir.join(format!("{id}.toml")),
10095 format!("schema_version = 1\nid = \"{id}\"\nkind = \"static\"\n"),
10096 )
10097 .unwrap();
10098 }
10099
10100 fn write_sources_toml(camp_root: &Path, body: &str) {
10101 let dir = camp_root.join(".yah/infra");
10102 std::fs::create_dir_all(&dir).unwrap();
10103 std::fs::write(dir.join("sources.toml"), body).unwrap();
10104 }
10105
10106 #[test]
10107 fn load_with_no_sources_toml_is_unchanged() {
10108 let tmp = tempfile::TempDir::new().unwrap();
10109 write_min_machine(&tmp.path().join(".yah/infra/machines"), "local-1", "");
10110 let cfg = CloudConfig::load(tmp.path()).unwrap();
10111 assert_eq!(cfg.machines.len(), 1);
10112 assert!(cfg.machine_origins.is_empty());
10113 assert!(cfg.provider_origins.is_empty());
10114 }
10115
10116 #[test]
10117 fn path_source_overlays_machines_and_providers_tagged_with_origin() {
10118 let camp = tempfile::TempDir::new().unwrap();
10119 let other = tempfile::TempDir::new().unwrap();
10120 write_min_machine(&other.path().join(".yah/infra/machines"), "borrowed-1", "");
10121 write_min_provider(&other.path().join(".yah/infra/providers"), "borrowed-provider");
10122 write_sources_toml(
10123 camp.path(),
10124 &format!(
10125 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
10126 other.path().display()
10127 ),
10128 );
10129
10130 let cfg = CloudConfig::load(camp.path()).unwrap();
10131 assert_eq!(cfg.machines.len(), 1);
10132 assert_eq!(cfg.machines[0].name, "borrowed-1");
10133 assert_eq!(cfg.providers.len(), 1);
10134 assert_eq!(cfg.providers[0].id, "borrowed-provider");
10135
10136 let origin = cfg.machine_origins.get("borrowed-1").expect("origin recorded");
10137 assert_eq!(origin.owner, "other");
10138 assert_eq!(origin.mode, SourceMode::ReadOnly);
10139 assert!(origin.source.starts_with("path:"));
10140 assert_eq!(
10141 cfg.provider_origins.get("borrowed-provider").unwrap().owner,
10142 "other"
10143 );
10144 }
10145
10146 #[test]
10147 fn camp_local_wins_on_name_collision_and_carries_no_origin() {
10148 let camp = tempfile::TempDir::new().unwrap();
10149 let other = tempfile::TempDir::new().unwrap();
10150 // Both declare a machine named "shared" -- camp-local's copy must win,
10151 // and it must never gain an origin tag.
10152 write_min_machine(&camp.path().join(".yah/infra/machines"), "shared", "");
10153 write_min_machine(
10154 &other.path().join(".yah/infra/machines"),
10155 "shared",
10156 "nickname = \"the borrowed one\"\n",
10157 );
10158 write_sources_toml(
10159 camp.path(),
10160 &format!(
10161 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
10162 other.path().display()
10163 ),
10164 );
10165
10166 let cfg = CloudConfig::load(camp.path()).unwrap();
10167 assert_eq!(cfg.machines.len(), 1, "the name collides, so exactly one entry");
10168 assert_eq!(cfg.machines[0].nickname, None, "camp-local's copy, not the borrowed one");
10169 assert!(
10170 !cfg.machine_origins.contains_key("shared"),
10171 "camp-local entries never carry an origin tag"
10172 );
10173 }
10174
10175 #[test]
10176 fn an_earlier_source_wins_over_a_later_one_on_collision() {
10177 let camp = tempfile::TempDir::new().unwrap();
10178 let first = tempfile::TempDir::new().unwrap();
10179 let second = tempfile::TempDir::new().unwrap();
10180 write_min_machine(&first.path().join(".yah/infra/machines"), "dup", "");
10181 write_min_machine(&second.path().join(".yah/infra/machines"), "dup", "");
10182 write_sources_toml(
10183 camp.path(),
10184 &format!(
10185 "schema_version = 1\n\n[[source]]\nowner = \"first\"\nkind = \"path\"\npath = \"{}\"\n\n[[source]]\nowner = \"second\"\nkind = \"path\"\npath = \"{}\"\n",
10186 first.path().display(),
10187 second.path().display()
10188 ),
10189 );
10190
10191 let cfg = CloudConfig::load(camp.path()).unwrap();
10192 assert_eq!(cfg.machines.len(), 1);
10193 assert_eq!(cfg.machine_origins.get("dup").unwrap().owner, "first");
10194 }
10195
10196 #[test]
10197 fn select_filters_borrowed_machines_by_name_or_mesh_tag() {
10198 let camp = tempfile::TempDir::new().unwrap();
10199 let other = tempfile::TempDir::new().unwrap();
10200 write_min_machine(&other.path().join(".yah/infra/machines"), "runner-1", "mesh_tags = [\"tag:cloud-runner\"]\n");
10201 write_min_machine(&other.path().join(".yah/infra/machines"), "excluded-1", "");
10202 write_sources_toml(
10203 camp.path(),
10204 &format!(
10205 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\nselect = [\"tag:cloud-runner\"]\n",
10206 other.path().display()
10207 ),
10208 );
10209
10210 let cfg = CloudConfig::load(camp.path()).unwrap();
10211 assert_eq!(cfg.machines.len(), 1);
10212 assert_eq!(cfg.machines[0].name, "runner-1");
10213 }
10214
10215 #[test]
10216 fn one_unparseable_foreign_machine_does_not_sink_the_rest_of_the_directory_or_the_load() {
10217 let camp = tempfile::TempDir::new().unwrap();
10218 let other = tempfile::TempDir::new().unwrap();
10219 let dir = other.path().join(".yah/infra/machines");
10220 write_min_machine(&dir, "good", "");
10221 // Schema-skew gotcha: a foreign machine this binary's MachineConfig
10222 // can't parse at all (not just an unknown field -- MachineConfig has
10223 // no deny_unknown_fields, so this has to fail on a TYPE, not a name).
10224 std::fs::write(dir.join("bad.toml"), "name = 1\nprovider = 2\n").unwrap();
10225 write_sources_toml(
10226 camp.path(),
10227 &format!(
10228 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
10229 other.path().display()
10230 ),
10231 );
10232
10233 // Must not error at all -- camp-local load must never fail because a
10234 // source it doesn't own has one bad file.
10235 let cfg = CloudConfig::load(camp.path()).unwrap();
10236 assert_eq!(cfg.machines.len(), 1, "the good entry still loads");
10237 assert_eq!(cfg.machines[0].name, "good");
10238 }
10239
10240 #[test]
10241 fn an_unsynced_git_source_overlays_nothing_and_is_not_an_error() {
10242 // No `yah infra sync` (R615-T3) has ever run, so the cache dir this
10243 // resolves to doesn't exist. Must be silent, not fatal.
10244 let camp = tempfile::TempDir::new().unwrap();
10245 write_sources_toml(
10246 camp.path(),
10247 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"git\"\nrepo = \"git@github.com:yah-ai/infra.git\"\nref = \"main\"\n",
10248 );
10249 let cfg = CloudConfig::load(camp.path()).unwrap();
10250 assert!(cfg.machines.is_empty());
10251 assert!(cfg.machine_origins.is_empty());
10252 }
10253
10254 #[test]
10255 fn a_synced_git_source_reads_from_the_cache_dir_not_the_repo_path() {
10256 // No `subdir` declared -- the checkout ROOT is the infra root.
10257 let camp = tempfile::TempDir::new().unwrap();
10258 let cache = crate::paths::infra_source_cache_dir(camp.path(), "yah");
10259 write_min_machine(&cache.join("machines"), "synced-1", "");
10260 write_sources_toml(
10261 camp.path(),
10262 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"git\"\nrepo = \"git@github.com:yah-ai/infra.git\"\nref = \"main\"\n",
10263 );
10264 let cfg = CloudConfig::load(camp.path()).unwrap();
10265 assert_eq!(cfg.machines.len(), 1);
10266 assert_eq!(cfg.machines[0].name, "synced-1");
10267 assert!(cfg.machine_origins.get("synced-1").unwrap().source.starts_with("git:"));
10268 }
10269
10270 #[test]
10271 fn a_git_sources_subdir_is_honoured_like_the_component_case() {
10272 // W274's own example declares `subdir = "infra"` for a monorepo whose
10273 // registry lives under a subdirectory of the clone rather than at its
10274 // root -- prove `infra_root` actually reads it, not just `.subdir` on
10275 // GitSource parsing (R615-F1 already covers that half).
10276 let camp = tempfile::TempDir::new().unwrap();
10277 let cache = crate::paths::infra_source_cache_dir(camp.path(), "yah");
10278 write_min_machine(&cache.join("infra").join("machines"), "subdir-1", "");
10279 // Also plant a decoy at the checkout root to prove the root itself is
10280 // NOT read when a subdir is declared.
10281 write_min_machine(&cache.join("machines"), "root-decoy", "");
10282 write_sources_toml(
10283 camp.path(),
10284 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"git\"\nrepo = \"git@github.com:yah-ai/infra.git\"\nref = \"main\"\nsubdir = \"infra\"\n",
10285 );
10286 let cfg = CloudConfig::load(camp.path()).unwrap();
10287 assert_eq!(cfg.machines.len(), 1);
10288 assert_eq!(cfg.machines[0].name, "subdir-1");
10289 }
10290
10291 #[test]
10292 fn load_from_config_dir_never_applies_sources_overlay() {
10293 // R615-F2's explicit decision: multi-root sibling trees don't inherit
10294 // the classic .yah/infra/sources.toml. Prove it rather than assert it
10295 // silently -- a sources.toml sitting at workspace_root/.yah/infra/
10296 // must NOT leak into a load_from_config_dir call even though both
10297 // share the same workspace_root.
10298 let camp = tempfile::TempDir::new().unwrap();
10299 let other = tempfile::TempDir::new().unwrap();
10300 write_min_machine(&other.path().join(".yah/infra/machines"), "borrowed-1", "");
10301 write_sources_toml(
10302 camp.path(),
10303 &format!(
10304 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
10305 other.path().display()
10306 ),
10307 );
10308 let sibling_config_dir = camp.path().join(".noisetable");
10309 std::fs::create_dir_all(&sibling_config_dir).unwrap();
10310
10311 let cfg = CloudConfig::load_from_config_dir(&sibling_config_dir, camp.path()).unwrap();
10312 assert!(cfg.machines.is_empty(), "sources.toml must not apply here");
10313 assert!(cfg.machine_origins.is_empty());
10314 }
10315
10316 // ─── R615-T5: `inherit_machines` retirement — cutover proof ────────────
10317
10318 /// The successor to R615-T5's parity proof. That earlier pair of tests
10319 /// asserted the legacy `[infra].inherit_machines` redirect and an
10320 /// equivalent `kind = "path"` source resolved the same machine set, and
10321 /// that the two coexisted without duplicating rows. Both claims were about
10322 /// a mechanism that no longer exists, so they retired with it — what has
10323 /// to hold *now* is the other half of the same guarantee: a camp that
10324 /// declares only `sources.toml` resolves the shared root exactly as the
10325 /// redirect used to, and a stale `inherit_machines` key left behind in
10326 /// `camp.toml` changes nothing.
10327 ///
10328 /// That stale-key case is not hypothetical: it is precisely the state a
10329 /// camp is in between the code cutover and someone tidying its
10330 /// `camp.toml`, and a silent re-resolution there would double-count the
10331 /// borrowed nodes or hide their origin badge.
10332 #[test]
10333 fn a_stale_inherit_machines_key_does_not_change_what_sources_toml_resolves() {
10334 let shared = tempfile::TempDir::new().unwrap();
10335 write_min_machine(&shared.path().join(".yah/infra/machines"), "shared-node-1", "");
10336 write_min_machine(&shared.path().join(".yah/infra/machines"), "shared-node-2", "");
10337
10338 let sources_toml = format!(
10339 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"path\"\npath = \"{}\"\nmode = \"read-only\"\n",
10340 shared.path().display()
10341 );
10342
10343 // Camp A: migrated cleanly — sources.toml only.
10344 let clean = tempfile::TempDir::new().unwrap();
10345 write_sources_toml(clean.path(), &sources_toml);
10346
10347 // Camp B: mid-migration — same source, plus the retired key still
10348 // sitting in camp.toml pointing at the same root.
10349 let stale = tempfile::TempDir::new().unwrap();
10350 std::fs::create_dir_all(stale.path().join(".yah")).unwrap();
10351 std::fs::write(
10352 stale.path().join(".yah/camp.toml"),
10353 format!(
10354 "[infra]\ninherit_machines = \"{}\"\n",
10355 shared.path().display()
10356 ),
10357 )
10358 .unwrap();
10359 write_sources_toml(stale.path(), &sources_toml);
10360
10361 let via_clean = CloudConfig::load(clean.path()).unwrap();
10362 let via_stale = CloudConfig::load(stale.path()).unwrap();
10363
10364 let names = |cfg: &CloudConfig| {
10365 let mut v: Vec<String> = cfg.machines.iter().map(|m| m.name.clone()).collect();
10366 v.sort();
10367 v
10368 };
10369 assert_eq!(
10370 names(&via_clean),
10371 names(&via_stale),
10372 "a leftover inherit_machines key must be inert — the retired redirect is gone"
10373 );
10374 assert_eq!(names(&via_clean), vec!["shared-node-1", "shared-node-2"]);
10375
10376 // And both are *borrowed*, not camp-local. This is the operator-facing
10377 // win the stopgap could never deliver: under the old redirect these
10378 // resolved with no origin at all, indistinguishable from locally-owned
10379 // nodes.
10380 assert_eq!(via_clean.machine_origins.len(), 2);
10381 assert_eq!(via_stale.machine_origins.len(), 2);
10382 for origin in via_stale.machine_origins.values() {
10383 assert_eq!(origin.owner, "yah");
10384 assert_eq!(origin.mode, SourceMode::ReadOnly);
10385 }
10386 }
10387
10388 // ─── R860-T4 (W338): placement groups ───────────────────────────────────
10389
10390 /// One requirement edge, written the way a spec author writes it.
10391 fn requirement(ident: &str, locality: Locality) -> workload_spec::Requirement {
10392 workload_spec::Requirement {
10393 ident: workload_spec::MeshIdent(ident.into()),
10394 locality,
10395 supply: workload_spec::Supply::Wait,
10396 provides: None,
10397 }
10398 }
10399
10400 /// A `minimal_spec` (256 MiB / 250 millicores, Server by inference) that
10401 /// requires the given edges.
10402 fn spec_requiring(name: &str, requires: Vec<workload_spec::Requirement>) -> WorkloadSpec {
10403 WorkloadSpec {
10404 requires,
10405 ..minimal_spec(name, 1)
10406 }
10407 }
10408
10409 /// The declared inventory an ident is resolved against — `.yah/infra/workloads/`.
10410 fn declared(specs: Vec<WorkloadSpec>) -> Vec<WorkloadConfig> {
10411 specs.into_iter().map(|spec| WorkloadConfig { spec }).collect()
10412 }
10413
10414 fn member_names(group: &[WorkloadSpec]) -> Vec<&str> {
10415 group.iter().map(|s| s.name.as_str()).collect()
10416 }
10417
10418 /// The headline case: `local` means "same node", so the two specs are one
10419 /// placement unit and admission has to reason about both.
10420 #[test]
10421 fn a_local_edge_binds_the_provider_into_the_placement_group() {
10422 let inventory = declared(vec![minimal_spec("headscale-replicator", 1)]);
10423 let requirer = spec_requiring(
10424 "headscale",
10425 vec![requirement("headscale-replicator", Locality::Local)],
10426 );
10427
10428 assert_eq!(
10429 member_names(&placement_group(&requirer, &inventory)),
10430 vec!["headscale", "headscale-replicator"]
10431 );
10432 }
10433
10434 /// The edge that must NOT bind. `prefer-local` "never blocks placement"
10435 /// (W338's locality table), and `anywhere` — which is what every legacy
10436 /// `depends_on` folds into — is an ordinary service dependency. Binding
10437 /// either would silently make every dependency in the tree a co-scheduling
10438 /// constraint and start summing unrelated workloads into the capacity floor.
10439 #[test]
10440 fn prefer_local_and_anywhere_edges_do_not_bind_the_group() {
10441 let inventory = declared(vec![
10442 minimal_spec("headscale-db", 1),
10443 minimal_spec("metrics", 1),
10444 minimal_spec("legacy-dep", 1),
10445 ]);
10446
10447 let requirer = WorkloadSpec {
10448 depends_on: vec![workload_spec::MeshIdent("legacy-dep".into())],
10449 ..spec_requiring(
10450 "headscale",
10451 vec![
10452 requirement("headscale-db", Locality::PreferLocal),
10453 requirement("metrics", Locality::Anywhere),
10454 ],
10455 )
10456 };
10457
10458 assert_eq!(
10459 member_names(&placement_group(&requirer, &inventory)),
10460 vec!["headscale"]
10461 );
10462 }
10463
10464 /// Transitive, and via the inline spec a `supply = "self"` requirement
10465 /// carries rather than via an ident lookup — the sidecar shape W338's
10466 /// worked example is built on.
10467 #[test]
10468 fn the_group_is_the_transitive_closure_and_traverses_inline_provides() {
10469 let inline = workload_spec::Requirement {
10470 supply: workload_spec::Supply::SelfProvision,
10471 provides: Some(Box::new(minimal_spec("headscale-restore", 1))),
10472 ..requirement("headscale-restore", Locality::Local)
10473 };
10474 let middle = WorkloadSpec {
10475 requires: vec![requirement("wal-shipper", Locality::Local)],
10476 ..minimal_spec("headscale-replicator", 1)
10477 };
10478 let inventory = declared(vec![middle, minimal_spec("wal-shipper", 1)]);
10479
10480 let requirer = spec_requiring(
10481 "headscale",
10482 vec![
10483 inline,
10484 requirement("headscale-replicator", Locality::Local),
10485 ],
10486 );
10487
10488 assert_eq!(
10489 member_names(&placement_group(&requirer, &inventory)),
10490 vec![
10491 "headscale",
10492 "headscale-restore",
10493 "headscale-replicator",
10494 "wal-shipper"
10495 ]
10496 );
10497 }
10498
10499 /// `validate::check_requires` bounds `provides` nesting to depth 1 but
10500 /// cannot stop two separately-declared specs from naming each other. Without
10501 /// the visited set this closure never terminates, so admission would hang
10502 /// rather than refuse — the worst failure shape for a deploy gate.
10503 #[test]
10504 fn an_ident_cycle_closes_the_group_instead_of_looping_forever() {
10505 let b = spec_requiring("b", vec![requirement("a", Locality::Local)]);
10506 let a = spec_requiring("a", vec![requirement("b", Locality::Local)]);
10507 let inventory = declared(vec![a.clone(), b]);
10508
10509 assert_eq!(member_names(&placement_group(&a, &inventory)), vec!["a", "b"]);
10510 }
10511
10512 /// An unresolvable ident is skipped, not fatal: admission is a pure function
10513 /// of the declared inventory, and refusing every deploy whose provider is
10514 /// not yet declared would make `requires` unusable before R860-T6 lands.
10515 #[test]
10516 fn an_unresolvable_local_ident_is_skipped_rather_than_refused() {
10517 let requirer = spec_requiring("headscale", vec![requirement("not-declared", Locality::Local)]);
10518 assert_eq!(
10519 member_names(&placement_group(&requirer, &[])),
10520 vec!["headscale"]
10521 );
10522 }
10523
10524 /// W338 §Placement consequences 1: the capacity floor is the group's sum.
10525 /// A node that fits the requirer alone must refuse the group — placing it
10526 /// there would oversubscribe the node the moment the provider follows.
10527 #[test]
10528 fn the_capacity_floor_is_the_sum_of_the_group_not_the_requirer_alone() {
10529 let provider = minimal_spec("headscale-replicator", 1);
10530 let requirer = spec_requiring(
10531 "headscale",
10532 vec![requirement("headscale-replicator", Locality::Local)],
10533 );
10534 // Two `minimal_spec`s: 256 MiB + 250 millicores each.
10535 let inventory = declared(vec![provider]);
10536
10537 let too_small = CloudConfig {
10538 workloads: inventory.clone(),
10539 ..make_empty_cfg(vec![make_machine_with_capacity("small", 300, 4000, vec![])])
10540 };
10541 let err = too_small.admit_workload(&requirer).unwrap_err().to_string();
10542 assert!(
10543 err.contains("memory_mb>=512"),
10544 "the floor must name the group's summed demand, got: {err}"
10545 );
10546
10547 let big_enough = CloudConfig {
10548 workloads: inventory,
10549 ..make_empty_cfg(vec![make_machine_with_capacity("roomy", 512, 4000, vec![])])
10550 };
10551 assert_eq!(
10552 big_enough.admit_workload(&requirer).unwrap().name,
10553 "roomy",
10554 "a node covering the sum must still admit the group"
10555 );
10556 }
10557
10558 /// W338 §Placement consequences 2, and the reason repulsion is computed over
10559 /// a set at all: the requirer is a `Server`, so the pre-R860 axis would have
10560 /// let it onto a `no-appliance` dev Pi and dragged its Appliance provider
10561 /// there with it.
10562 #[test]
10563 fn a_server_requiring_an_appliance_locally_is_repelled_by_no_appliance() {
10564 let appliance = WorkloadSpec {
10565 archetype: Some(LifecycleArchetype::Appliance),
10566 ..minimal_spec("headscale", 1)
10567 };
10568 let requirer = spec_requiring("headscale-ui", vec![requirement("headscale", Locality::Local)]);
10569 assert_eq!(
10570 requirer.effective_archetype(),
10571 LifecycleArchetype::Server,
10572 "precondition: the requirer itself must not be an Appliance"
10573 );
10574
10575 let cfg = CloudConfig {
10576 workloads: declared(vec![appliance]),
10577 ..make_empty_cfg(vec![
10578 make_machine_with_capacity("dev-pi", 8192, 4000, vec!["no-appliance"]),
10579 make_machine_with_capacity("us-west-001", 8192, 4000, vec![]),
10580 ])
10581 };
10582
10583 assert_eq!(
10584 cfg.admit_workload(&requirer).unwrap().name,
10585 "us-west-001",
10586 "the dev Pi repels the group's Appliance member"
10587 );
10588
10589 // And with the Appliance gone from the group, the same requirer is
10590 // admissible on the same Pi — proving the repulsion came from the edge.
10591 let alone = minimal_spec("headscale-ui", 1);
10592 assert_eq!(cfg.admit_workload(&alone).unwrap().name, "dev-pi");
10593 }
10594
10595 /// The set-valued form of the per-workload drain skip the node makes in
10596 /// `drain_workloads`: one Appliance member pins the whole group.
10597 #[test]
10598 fn a_group_containing_an_appliance_is_not_drainable() {
10599 let server = minimal_spec("headscale-ui", 1);
10600 let appliance = WorkloadSpec {
10601 archetype: Some(LifecycleArchetype::Appliance),
10602 ..minimal_spec("headscale", 1)
10603 };
10604
10605 assert!(group_is_drainable(std::slice::from_ref(&server)));
10606 assert!(!group_is_drainable(&[server, appliance]));
10607 }
10608
10609 /// The regression that matters most: nothing in the tree declares
10610 /// `requires` yet, so every existing spec's group is exactly itself and its
10611 /// admission axes must be bit-identical to the pre-R860 derivation.
10612 #[test]
10613 fn a_spec_with_no_local_edges_admits_exactly_as_it_did_before() {
10614 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
10615 let req = admission_spec(&ws, &[]);
10616
10617 assert_eq!(req.mesh_tags, vec!["tag:build-worker", "arch:x86"]);
10618 assert_eq!(req.memory_mb, ws.memory_request_mb());
10619 assert_eq!(req.cpu_millis, ws.resources.cpu_millis);
10620 // R876-B7: the axis is now the complement — every repelling key EXCEPT
10621 // this spec's own class, which is the same predicate stated from the
10622 // other side. Asserted against the derivation rather than a literal so
10623 // it stays true if a fourth archetype is added.
10624 assert_eq!(
10625 req.tolerates,
10626 tolerations_excluding(&[ws.effective_archetype()])
10627 );
10628 let own = format!("no-{}", ws.effective_archetype().taint_key());
10629 assert!(
10630 !req.tolerates.contains(&own),
10631 "a spec never tolerates the taint aimed at its own class"
10632 );
10633 }
10634
10635 // ─── R860-T5 (W338 §Placement consequences 3): native-exec capability ────
10636
10637 /// A `minimal_spec` carrying the `yah.exec = native` marker — the only way
10638 /// a workload says "fork+exec me on the host" (`WorkloadSpec::
10639 /// wants_native_exec`). It stays a Container workload on the wire; the
10640 /// marker is the whole difference.
10641 fn native_spec(name: &str) -> WorkloadSpec {
10642 let mut ws = minimal_spec(name, 1);
10643 ws.annotations.insert(
10644 workload_spec::NATIVE_EXEC_ANNOTATION.to_string(),
10645 workload_spec::NATIVE_EXEC_VALUE.to_string(),
10646 );
10647 assert!(ws.wants_native_exec(), "precondition: the marker must read back");
10648 ws
10649 }
10650
10651 /// The R858 failure, now caught at placement instead of at dispatch: a node
10652 /// whose kamaji has no `--native-exec-dir` accepted the election and then
10653 /// refused the deploy, and nothing upstream could see it coming.
10654 #[test]
10655 fn a_node_without_the_native_exec_capability_cannot_host_a_native_workload() {
10656 let native = native_spec("headscale");
10657
10658 let incapable = make_empty_cfg(vec![make_machine("us-south-001", vec![])]);
10659 let err = incapable.admit_workload(&native).unwrap_err().to_string();
10660 assert!(
10661 err.contains(NATIVE_EXEC_MESH_TAG),
10662 "the refusal must name the missing capability, got: {err}"
10663 );
10664
10665 let capable = make_empty_cfg(vec![
10666 make_machine("us-south-001", vec![]),
10667 make_machine("us-west-001", vec![NATIVE_EXEC_MESH_TAG]),
10668 ]);
10669 assert_eq!(
10670 capable.admit_workload(&native).unwrap().name,
10671 "us-west-001",
10672 "a node declaring the capability admits the native workload"
10673 );
10674 }
10675
10676 /// W338's actual sentence: a `supply = "self"` spec "must be placeable where
10677 /// its requirer lands". The requirer here is an ordinary container workload
10678 /// — it is the *provider* reached by a `local` edge that needs the host
10679 /// backend, so the capability has to be required of the group, not of the
10680 /// spec being deployed.
10681 #[test]
10682 fn a_local_edge_to_a_native_provider_makes_the_requirer_need_the_capability() {
10683 let requirer = spec_requiring(
10684 "headscale-ui",
10685 vec![requirement("headscale", Locality::Local)],
10686 );
10687 assert!(
10688 !requirer.wants_native_exec(),
10689 "precondition: the requirer itself is an ordinary container workload"
10690 );
10691
10692 let cfg = CloudConfig {
10693 workloads: declared(vec![native_spec("headscale")]),
10694 ..make_empty_cfg(vec![
10695 make_machine("plain", vec![]),
10696 make_machine("us-west-001", vec![NATIVE_EXEC_MESH_TAG]),
10697 ])
10698 };
10699
10700 assert_eq!(
10701 cfg.admit_workload(&requirer).unwrap().name,
10702 "us-west-001",
10703 "the group's native member pulls the requirer onto a capable node"
10704 );
10705
10706 // Without the edge the same requirer is admissible on the plain node,
10707 // so the constraint provably came from the group and not from the spec.
10708 let alone = minimal_spec("headscale-ui", 1);
10709 assert_eq!(cfg.admit_workload(&alone).unwrap().name, "plain");
10710 }
10711
10712 /// The regression guard: nothing in the tree is native-marked today, so
10713 /// every existing spec's axes must be untouched by this ticket.
10714 #[test]
10715 fn a_group_with_no_native_member_does_not_require_the_capability() {
10716 let inventory = declared(vec![minimal_spec("headscale-replicator", 1)]);
10717 let requirer = spec_requiring(
10718 "headscale",
10719 vec![requirement("headscale-replicator", Locality::Local)],
10720 );
10721
10722 let req = admission_spec(&requirer, &inventory);
10723 assert!(
10724 !req.mesh_tags.iter().any(|t| t == NATIVE_EXEC_MESH_TAG),
10725 "no native member ⇒ no capability axis, got: {:?}",
10726 req.mesh_tags
10727 );
10728
10729 // And it still lands on a node that declares nothing at all.
10730 let cfg = CloudConfig {
10731 workloads: inventory,
10732 ..make_empty_cfg(vec![make_machine("plain", vec![])])
10733 };
10734 assert_eq!(cfg.admit_workload(&requirer).unwrap().name, "plain");
10735 }
10736}