cloud/config.rs
1//! @yah:ticket(R040-F16, "pg-on-mesh service recipe: bind tailscale0 + pg_hba.conf snippet + ufw rules")
2//! @yah:at(2026-05-05T00:32:34Z)
3//! @yah:assignee(agent:claude)
4//! @yah:status(review)
5//! @yah:parent(R040)
6//! @yah:handoff("Companion to R040-F15. Inter-node TCP (Postgres primary↔replica, NATS clusters, anything raw-protocol) lives on the Headscale mesh, not on Hetzner public IPs. Each node has a stable 100.64.x.x mesh IP that survives replacement of the underlying box, so DNS / config / pg_hba never churn when a CPX-11 is rebuilt. WireGuard already encrypts the wire — TLS becomes defense-in-depth, not load-bearing. This ticket carries the concrete pg-shaped recipe so the first stateful service deploy doesn't have to re-derive the pattern; subsequent services (redis, NATS, etc.) cargo-cult from it.")
7//! @yah:next("ServiceConfig gains a `bind_interface: Option<String>` field (e.g. `Some(\"tailscale0\")` for mesh-only services). The cloud-init/podman compose renderer translates this into either `--network host` + `pg listen_addresses = '<mesh-ip>'` OR a podman macvlan/host-binding pattern that achieves the same.")
8//! @yah:next("Generated pg_hba.conf snippet: allow the mesh subnet (100.64.0.0/10) for replication + app users. Postgres binds to the node's tailscale0 mesh IP only — `listen_addresses` is templated from the node's `tailscale ip --4` at first boot.")
9//! @yah:next("Generated ufw rules: `ufw allow in on tailscale0 to any port 5432; ufw deny 5432` — mirrors the existing yah-yubaba 7443 pattern in mirror.yml. Same shape works for any mesh-only port.")
10//! @yah:next("Replica connection string uses primary's mesh IP, NOT its public IP. Stable across box replacement.")
11//! @yah:next("Out of scope: pg_basebackup orchestration, failover, WAL archiving — those belong in noisetable's domain; this ticket only standardizes the binding/firewall/auth shape so noisetable's pg deployment doesn't reinvent it.")
12//!
13//!
14//! @yah:ticket(R323-F9, "Add sync-wave ordering to ServiceComponent (deploy-panel wave order)")
15//! @yah:assignee(agent:claude)
16//! @yah:at(2026-05-26T15:20:25Z)
17//! @yah:status(review)
18//! @yah:phase(P2)
19//! @yah:parent(R323)
20//! @yah:next("ServiceComponent gains a wave/order field (or depends_on between components) so the deploy panel (R323-F4) can group workload rollout rows into sync waves (wave 0 parallel, wait healthy, wave 1, …). Today all components are implicitly wave 0.")
21//! @yah:next("compute_service/compute_cell in reconciler/sync_status.rs surface the wave per workload so F4 doesn't re-derive it.")
22//! @yah:gotcha("Until this lands, F4 should render every workload as wave 0 (no ordering).")
23//! @yah:handoff("Added wave: u32 (serde default=0, skip_serializing_if zero) to ServiceComponent in config.rs. Added is_zero_u32 helper. Fixed the three struct literal call-sites that now need wave: 0 (config.rs test, local_sim.rs x2, mesofact_static.rs). Added wave?: number to the TS ServiceComponent interface with a doc comment. Deploy panel now reads c.wave ?? 0 for each WorkloadRow instead of hardcoded 0. SyncFooter computes maxWave from the components array and renders 'wave 0' (all-zero case) or 'waves 0–N' (multi-wave). All 218 cloud lib tests pass; bun run typecheck clean.")
24//! @yah:verify("cargo test -p cloud --lib # 218 passed")
25//! @yah:verify("cd packages/yah/ui && bun run typecheck # no new errors")
26//! @yah:verify("In service.toml: add wave = 1 to a component, rebuild, open the deploy panel — that workload row shows 'w1' badge; SyncFooter shows 'waves 0–1'")
27//! @yah:verify("Component with no wave field in TOML deserializes as wave=0 (default). Saving a wave=0 component omits the field from the output TOML (skip_serializing_if).")
28//!
29//! @arch:see(.yah/docs/working/W142-pond.md)
30//!
31//! @yah:relay(R615, "Linked infra sources: sources.toml overlay so a camp can borrow another camp's substrate")
32//! @yah:at(2026-07-20T18:18:05Z)
33//! @yah:status(open)
34//! @arch:see(.yah/docs/working/W274-linked-infra-sources.md)
35//!
36//! @yah:ticket(R615-F1, "InfraSource types + SourcesConfig::load(infra_dir) parsing .yah/infra/sources.toml")
37//! @yah:status(review)
38//! @yah:assignee(agent:bundle-anthropic-miravel)
39//! @yah:at(2026-08-08T19:55:57Z)
40//! @yah:phase(P1)
41//! @yah:parent(R615)
42//! @yah:next("Add InfraSourceKind { Path { path }, Git(GitSource) } + InfraSource { owner, kind, mode, select } to cloud/src/config.rs. Reuse the existing GitSource (config.rs:1205, { repo, ref, subdir }) verbatim — do not invent a second git-source shape.")
43//! @yah:next("SourcesConfig::load(infra_dir) reads .yah/infra/sources.toml (schema_version = 1, ordered [[source]] array). Absent file = empty list, never an error — every existing camp has no sources.toml.")
44//! @yah:next("mode is the write-gate: read-only (borrower cannot mutate) vs owner-manages. Model it as an enum, not a bool, so a future read-write-with-approval tier is additive.")
45//! @yah:verify("cargo check -p cloud && cargo test -p cloud")
46//! @arch:see(.yah/docs/working/W274-linked-infra-sources.md)
47//! @yah:tier(Cleric)
48//! @yah:handoff("InfraSourceKind{Path{path},Git(GitSource)} + SourceMode{ReadOnly,Manage} + InfraSource{owner,kind,mode,select} + SourcesConfig{schema_version,source} all landed in oss/yubaba/crates/cloud/src/config.rs (after default_git_ref, ~line 1550). GitSource reused verbatim -- Git(GitSource) wraps the existing R561 type unchanged, no second git-source shape. InfraSourceKind is internally tagged (#[serde(tag=\"kind\", rename_all=\"kebab-case\")]) and flattened into InfraSource so a [[source]] table reads exactly like W274's example: owner/kind/path-or-repo+ref+subdir/mode/select all at one table level. mode: SourceMode defaults ReadOnly via #[serde(default)] on the field (enum, not bool, per the ticket's own instruction -- Manage is the explicit escape hatch). SourcesConfig::load(infra_dir) returns Ok(default()) -- schema_version=1, empty source list -- when sources.toml is absent; only parses+errors when the file exists and is malformed.")
49//! @yah:handoff("Tree anchor 85801e7f. Pathspec: oss/yubaba/crates/cloud/src/config.rs (only file touched). Tests: cargo test -p yah-cloud --lib (from oss/yubaba) 710 passed / 0 failed / 4 ignored, +6 new over the 704 baseline your R707-T6 verification recorded (sources_load_is_empty_when_the_file_is_absent, sources_parses_a_path_kind_exactly_like_w274s_example, sources_parses_a_git_kind_reusing_gitsource_verbatim, sources_mode_defaults_to_read_only_and_manage_is_explicit, sources_preserves_declaration_order, sources_round_trips_through_serialize). cargo check -p cloud also green (implied by the test build).")
50//! @yah:handoff("Tree anchor at handoff: 85801e7f6b76b369c0c8ecd2e5c7874990cd9286 — the shared tree as I left it. Diff against it (`git diff 85801e7f6b76b369c0c8ecd2e5c7874990cd9286..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
51//! @yah:next("R615-F2 picks this straight up: overlay these sources into CloudConfig::load, tagging origin{owner,source} and merging camp-local-wins-on-collision.")
52//! @yah:handoff("Verified pre-existing work: InfraSourceKind{Path,Git(GitSource)} + SourceMode + InfraSource + SourcesConfig all present in oss/yubaba/crates/cloud/src/config.rs at tree anchor 871fde1c, matching the inline @yah:handoff notes already on this ticket. GitSource reused verbatim, no second git-source shape. This session added no new code -- only ran verification and closed the board state, which a prior session left stuck in `open` despite the work being done (code + handoff notes landed, but board.review/handoff was never called).")
53//! @yah:verify("cargo check -p yah-cloud -- clean (2 pre-existing unrelated warnings)")
54//! @yah:verify("cargo test -p yah-cloud --lib -- 723 passed; 0 failed; 4 ignored (from oss/yubaba)")
55//!
56//! @yah:ticket(R615-F2, "Overlay loader: resolve sources in CloudConfig::load, tag origin, camp-local wins on collision")
57//! @yah:status(review)
58//! @yah:assignee(agent:bundle-anthropic-miravel)
59//! @yah:at(2026-08-08T19:56:05Z)
60//! @yah:phase(P1)
61//! @yah:parent(R615)
62//! @yah:next("In CloudConfig::load, after loading camp-local machines/providers/rules, resolve each source to an infra root (git sources read from the .yah/cache/infra/ sync cache — load stays offline), load that root's machines/providers/rules, tag each entry with origin { owner, source }, and overlay UNDER camp-local. Camp-local wins on name collision.")
63//! @yah:next("The machine load site is config.rs:533 (load_dir::<MachineConfig>(paths::machines_dir(...))). Note config.rs:575 load_from_config_dir is a SECOND machine load site that deliberately skips the inherit_machines redirect for multi-root/sibling trees (W206) — decide explicitly whether sources overlay applies there too, and document the answer either way.")
64//! @yah:verify("cargo check -p cloud && cargo test -p cloud")
65//! @yah:verify("A camp with sources.toml [[source]] kind=path to a sibling camp sees that camp's machines in CloudConfig::load, each tagged with the source owner")
66//! @yah:gotcha("Cross-camp MachineConfig schema skew is real: noisetable ships an older machine schema (location/server_type/hosts_mirrors) while yah's use region/arch/[connect]. A borrowed source can carry fields the borrower's binary predates. Overlay load MUST tolerate/skip unparseable foreign entries per-file and warn — never fail the whole load.")
67//! @arch:see(.yah/docs/working/W274-linked-infra-sources.md)
68//! @yah:depends_on(R615-F1)
69//! @yah:tier(Warrior)
70//! @yah:handoff("Overlay landed in CloudConfig::load (oss/yubaba/crates/cloud/src/config.rs). After camp-local machines/providers/legacy-merge finish, SourcesConfig::load(paths::infra_dir(workspace_root)) resolves + overlay_infra_sources() merges each source's machines/providers UNDER what's already there -- camp-local wins any name collision, and among sources themselves the earlier-declared one wins (both proven by dedicated tests). Provenance is NOT a field on MachineConfig/ProviderConfig: added CloudConfig.machine_origins/provider_origins: BTreeMap<String, InfraOrigin> instead, keyed by name/id. Reason recorded in a doc comment on InfraOrigin -- MachineConfig/ProviderConfig are constructed by struct literal in test helpers across several crates (including crates/yah/agent-tools/src/cloud_tools.rs, which is fenced/live-owned this session), so widening either shape would have forced an edit there for zero semantic gain; origin is a property of the LOAD, not the machine.")
71//! @yah:handoff("GOTCHA closed: added load_dir_tolerant<T>() -- a per-file-tolerant sibling of the existing (strict) load_dir -- so one unparseable foreign machine/provider (schema skew) skips-with-a-tracing::warn! and never sinks the rest of that source's directory or this camp's own load. Proven by one_unparseable_foreign_machine_does_not_sink_the_rest_of_the_directory_or_the_load. load_dir itself is untouched -- camp-local files still hard-fail on a bad TOML, which is correct, only borrowed roots get the tolerant path.")
72//! @yah:handoff("Git sources: InfraSource::infra_root() resolves kind=path to <workspace_root>/<path>/.yah/infra (live tree, no I/O beyond building the path) and kind=git to paths::infra_source_cache_dir(workspace_root, owner)/infra -- a NEW path helper in paths.rs, also what R615-T3's `yah infra sync` target directory must be so the two line up. An unsynced git source (cache dir absent) overlays nothing and is explicitly NOT an error (test: an_unsynced_git_source_overlays_nothing_and_is_not_an_error) -- load() stays fully offline as W274 §3 requires.")
73//! @yah:handoff("select filtering implemented for machines only (name exact-match or literal mesh_tags membership -- not a glob engine, matches W274's own example verbatim) via machine_matches_select(); does NOT apply to providers -- documented as a deliberate choice, nothing in W274 or the ticket describes a provider-scoped filter.")
74//! @yah:handoff("EXPLICIT DECISION on the config.rs:575-equivalent gotcha (now load_from_config_dir): sources overlay does NOT apply there. Multi-root sibling config dirs (W206 layout (b)) are a second config root INSIDE the same camp, not a second camp -- .yah/infra/sources.toml is tied to paths::infra_dir(workspace_root) specifically, which has no well-defined meaning for an arbitrary config_dir. Documented in the function's doc comment and proven by load_from_config_dir_never_applies_sources_overlay (a sources.toml at the real workspace root does NOT leak into a load_from_config_dir call against a sibling .noisetable/ dir under that same root).")
75//! @yah:handoff("Tree anchor 85801e7f. Pathspec: oss/yubaba/crates/cloud/src/config.rs, oss/yubaba/crates/cloud/src/paths.rs (added infra_source_cache_dir + 1 test), oss/yubaba/crates/cloud/src/reconciler/mesofact_bundle.rs (CloudConfig test-literal fixed for the 2 new fields), app/yah/cli/src/cloud.rs (3 CloudConfig test-literal sites fixed, same reason). Tests: cargo test -p yah-cloud --lib (from oss/yubaba) 720 passed / 0 failed / 4 ignored, +10 over R615-F1's 710 baseline (9 overlay tests in config.rs + 1 in paths.rs). cargo build -p yah --lib (repo root) green -- confirms nothing downstream (agent-tools, cloud.rs, hub) broke from CloudConfig's two new fields.")
76//! @yah:handoff("Tree anchor at handoff: 85801e7f6b76b369c0c8ecd2e5c7874990cd9286 — the shared tree as I left it. Diff against it (`git diff 85801e7f6b76b369c0c8ecd2e5c7874990cd9286..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
77//! @yah:next("R615-T3 (yah infra sync) is unblocked and has everything it needs: paths::infra_source_cache_dir(workspace_root, owner) is the exact target directory to clone/pull git sources into, already matching what F2's overlay reads from.")
78//! @yah:next("R615-F4 (Infra tab origin badge, not in my assigned lane) can read CloudConfig.machine_origins/provider_origins directly -- no further backend plumbing needed for the badge itself.")
79//! @yah:handoff("Verified pre-existing work: overlay landed in CloudConfig::load (oss/yubaba/crates/cloud/src/config.rs) at tree anchor 871fde1c -- SourcesConfig::load resolves sources, overlay_infra_sources() merges under camp-local with camp-local-wins and earlier-source-wins collision rules, machine_origins/provider_origins BTreeMaps added to CloudConfig, load_dir_tolerant() added for per-file-tolerant foreign schema skew, InfraSource::infra_root() resolves path/git kinds, load_from_config_dir explicitly does NOT get the overlay (documented). Matches this ticket's own inline @yah:handoff notes. This session added no new code -- only ran verification and closed board state that a prior session left stuck in `open` despite the work being done.")
80//! @yah:verify("cargo check -p yah-cloud -- clean (2 pre-existing unrelated warnings)")
81//! @yah:verify("cargo test -p yah-cloud --lib -- 723 passed; 0 failed; 4 ignored (from oss/yubaba), includes overlay tests + load_dir_tolerant test + infra_source_cache_dir test in paths.rs")
82//!
83//! @yah:ticket(R605-F12, "Sovereign groups have no voting axis, so non-voting membership is inexpressible and the raft guard is enforced by an absent field")
84//! @yah:status(review)
85//! @yah:at(2026-08-20T05:15:30Z)
86//! @yah:assignee(agent:bundle-anthropic-ashguard)
87//! @yah:parent(R605)
88//! @arch:see(.yah/docs/working/W325-isolated-x86-build-capacity.md)
89//! @yah:next("OPERATOR INTENT (2026-08-19) that the model cannot currently record: us-west-003 is a NON-VOTING member of the us-west-001-based (prod) sovereign group, and us-west-011 is a DIFFERENT sovereign (dev) from 001/003. The dev/prod split is already declared correctly. The non-voting membership is not — us-west-003.toml declares no sovereign_group at all.")
90//! @yah:next("THE GAP: MachineConfig::sovereign_group is a single Option<String>, so membership is binary, and judge_join (oss/yubaba/crates/cloud/src/config.rs:459) permits a join IFF both sides declare the same non-None group. There is no way to say 'in this blast radius, but not quorum-eligible'.")
91//! @yah:next("WHY THAT IS ACTIVELY BAD, not just missing: today the ONLY thing refusing us-west-003 into the prod raft at the join gate is its ABSENT stamp. Its own file is emphatic it must never hold a raft node id ('a home-internet partition should never be able to stall the raft'), and that guarantee currently rests on a field nobody wrote. Stamping it prod to record the operator's real intent would REMOVE the guard. This is precisely the W305 failure mode that produced R742-T4: `no-voter` sat inert on three nodes asserting something nothing enforced.")
92//! @yah:next("PROPOSED SHAPE (recommended): a second axis, e.g. sovereign_role = voter | non-voter (default voter for back-compat, or make it required), with judge_join permitting a same-group join only for voters. Then us-west-003 stamps prod + non-voter, the intent is machine-readable, and the raft guard stops depending on omission. us-west-004 (R605-T7) would take the same shape.")
93//! @yah:next("TOUCHES TWO COPIES OF THE PREDICATE, do not fix only one: cloud::judge_join renders the camp-side refusal, but the predicate itself lives in workload_spec::sovereign::join_permitted because yubaba's POST /raft/add-learner gate asks the same question and there is deliberately no yubaba -> cloud edge. Also re-read `yubaba serve --sovereign-group`, whose node-side gate is narrower on purpose (an unset flag means 'declared nothing', not 'declared standalone').")
94//! @yah:gotcha("THE CODE AND THE OPERATOR CURRENTLY DISAGREE ABOUT 003, and a reader should know which is which before editing. judge_join's own doc comment asserts 'prod and dev are both stamped, and us-west-002/003/015 are deliberately not raft members' — i.e. R742-F1 modelled 003 as STANDALONE. The operator's model is that it is a NON-VOTING MEMBER of prod. Those are different claims, not a wording difference: standalone means no blast-radius relationship to 001 at all. Do not silently 'correct' either side; this ticket is the reconciliation.")
95//! @yah:gotcha("FLEET STATE AS DECLARED (2026-08-19): prod = us-west-001, us-south-001, us-east-001. dev = us-west-011, us-west-013, us-west-014. NO sovereign_group declared = us-west-002, us-west-003, us-west-015. Verify against the files rather than trusting this list — xtask/tests/fleet_sovereign_groups.rs pins the roster and will need updating in the same change (it also asserts the stamp parses as a TOP-LEVEL key, which matters because 003 has a long comment block before [allocatable] where a stamp would silently become a member of that table).")
96//! @yah:gotcha("SEPARATE AXIS, DO NOT ENTANGLE: mesh membership is not sovereign membership. The standing rule is ONE mesh for the entire fleet regardless of group (operator, 2026-08-19), so us-west-003 and us-west-011 enrolling in headscale is unrelated work with no design question in it — see R605-T10. A voting axis on sovereign_group must not become a reason to keep any node off the mesh.")
97//! @yah:gotcha("SHARED-TREE COLLISION, live 2026-08-20: R772 (Miravel:spade, session:ce6d74a9) is refactoring oss/yubaba/crates/cloud/src/validate.rs at the same time and the file is currently RED - error[E0425] cannot find function load_machines at validate.rs:753, a half-landed extraction of the machine-loading walk that check_inert_taints / check_retired_arch_tags / the new check_unroled_sovereign_members all duplicate. That error is NOT from this ticket. Told them by party.chat and asked them to absorb check_unroled_sovereign_members into load_machines rather than leave one holdout. Do not hand-fight the file.")
98//! @yah:gotcha("R772 ALSO BROKE THREE PRE-EXISTING INGRESS TESTS, again not this ticket: two_services_fronting_one_node_collate_into_one_front_door, a_cross_service_hostname_clash_is_reported_with_both_declarations, one_mirrors_broken_declaration_does_not_hide_the_rest - all failing with 'providers.compute.use = hetzner - no such provider'. Cause is their new CloudConfig::load(workspace_root) at validate.rs:750 inside collate_workspace_ingress; the fronted_mirror fixture declares the slot but never writes infra/providers/hetzner.toml, and CloudConfig::load runs cross_ref_validate. Left alone deliberately - peer-owned.")
99//! @yah:gotcha("TRAP THAT MADE THREE OF MY OWN TESTS PASS FOR THE WRONG REASON: the machine-lint sweeps SKIP unparseable TOMLs by design (a peer's half-written scaffold must not sink the sweep). So a test fixture missing a REQUIRED MachineConfig field - mesh_tags is the one that bites - is silently skipped, the lint finds nothing, and every assert-empty test passes vacuously. Only the one test asserting found.len() == 1 noticed. write_sovereign_machine now always writes mesh_tags = [] and carries a comment saying why. Check this before trusting any new test in cloud::validate.")
100//! @yah:verify("cargo test -p yah-workload-spec --lib sovereign (from oss/yah-base) -- 9 passed, 0 failed. Covers both new refusals (a_non_voting_member_does_not_join_its_own_group, a_non_voting_target_has_no_quorum_to_join), the back-compat pin (the_default_role_is_the_pre_r605_f12_meaning), and the one-spelling round-trip across TOML/CLI/JSON.")
101//! @yah:verify("cargo test -p yubaba --lib sovereign (from oss/yubaba) -- 13 passed, 0 failed. Includes a_non_voting_joiner_is_refused_by_role_not_by_group, a_non_voting_target_refuses_every_joiner, a_group_without_a_role_key_is_a_voter_not_a_refusal (the deployed-fleet back-compat seam), a_peer_reports_its_role_in_the_toml_spelling.")
102//! @yah:verify("cargo test -p yubaba --test raft_sovereign_group (from oss/yubaba) -- 11 passed, 0 failed, up from 8. Three new end-to-end against real single-node rafts: a_non_voting_member_of_the_same_group_is_refused, a_non_voting_leader_refuses_to_grow_its_quorum, a_node_publishes_its_role_and_the_leader_reads_it_there (which also proves the request body cannot vote a non-voter in - the leader dials the joiner).")
103//! @yah:verify("cargo test -p xtask --test fleet_sovereign_groups (from repo root) -- 2 passed, 0 failed. THE DECISIVE ONE: parses the real .yah/infra/machines/*.toml through the actual MachineConfig deserializer. Confirms us-west-003 = prod + non-voter on disk, all six pre-existing voters now stamped sovereign_role = voter explicitly, and neither key swallowed by a table header.")
104//! @yah:verify("cargo test -p yah-cloud --lib (from oss/yubaba) -- 891 passed, 3 failed, where all 3 failures were R772's ingress-collate tests and none were mine. A clean re-run is BLOCKED, not failing: R555's in-flight AdmissionGrant.secrets field breaks velveteen-exec, and yah-cloud is not a root workspace member so its dev-deps can only resolve from the oss/yubaba workspace. Re-run once R555 lands.")
105//! @yah:handoff("LANDED, operator chose the second-axis shape (Call 1 = A, 2026-08-20). sovereign_role = voter | non-voter now sits beside sovereign_group, and ONE predicate judges both: workload_spec::sovereign::join_permitted(Membership, Membership) where Membership { group: Option<&str>, role: SovereignRole }. Permitted iff same non-None group AND both sides Voter. Both copies of the predicate call it - cloud::judge_join (camp-side) and yubaba::sovereign_group::judge (node-side) - so the rule itself cannot drift; only the prose differs, which was already the R742-F1 split.")
106//! @yah:handoff("WHY THE ROLE IS CHECKED ON BOTH SIDES, since only the joiner half was asked for: a join grows a quorum and it takes two nodes. Refusing a non-voting JOINER is the us-west-003 case. Refusing a non-voting TARGET is the same assertion read from the other end - a box declared non-voting that is serving add-learner is already holding a raft seat its own declaration forbids, and permitting there would paper over the contradiction. Both refusals name the role rather than the group when the groups match, because a message reading 'cross-group join refused: prod and prod' reads as a bug in the check.")
107//! @yah:handoff("THE DEFAULT IS THE LOAD-BEARING DECISION AND IT IS DELIBERATELY PERMISSIVE. An absent sovereign_role resolves to Voter (MachineConfig::sovereign_membership, the ONE place the Option is resolved). Reason: before this field, declaring a group WAS declaring quorum eligibility, so absence has to keep meaning that or the change silently retires six live voters. The permissiveness is bounded at the other end by cloud::validate::check_unroled_sovereign_members, which makes `yah cloud validate` FAIL on a group stamp with no role beside it - so the default can be reached by choice but not by silence. MachineConfig::sovereign_role stays Option<SovereignRole> (not a defaulted plain field) precisely so that lint can tell 'chose voter' from 'never considered it'.")
108//! @yah:handoff("NODE-SIDE BACK-COMPAT SEAM, pinned by a test because it is a decision and not an oversight: a peer answering GET /raft/status with a sovereign_group but NO sovereign_role key - every yubaba built between R742-F1 and R605-F12, which today is the entire prod raft - is read as Voter, not refused. Refusing would freeze a stamped cluster's growth until every member was rolled, strictly worse than what the role guards against, and it is the same degrade-toward-prior-behaviour stance the module already took for the group. Residue, named rather than hidden in read_group's doc: a box whose machine.toml says non-voter but whose daemon predates the flag answers 'voter' and the node gate admits it. judge_join refuses it camp-side, which is where operator-driven joins go. Window closes per-group as its nodes carry the flag.")
109//! @yah:handoff("FILES: workload-spec/src/sovereign.rs (SovereignRole + Membership + role-aware join_permitted, +227). cloud/src/config.rs (sovereign_role field, sovereign_membership(), judge_join same-group role branch, SovereignRole re-exported from cloud::config). cloud/src/validate.rs (check_unroled_sovereign_members + UnroledSovereignMember). app/yah/cli/src/cloud.rs (lint wired: ERROR in `yah cloud validate`, WARNING in the apply preflight - same split as inert-taint/retired-arch-tag, because an unwritten role changes no placement decision and the machine may be declared in a tree this camp does not own). yubaba/src/{sovereign_group,lib,main}.rs (--sovereign-role flag, ServerState.sovereign_role, /raft/status publishes it always-never-null, gate both directions). yubaba-test-harness/src/solo_node.rs (solo_node_with_sovereign_role). .yah/infra/machines/*.toml (7 files). xtask/tests/fleet_sovereign_groups.rs + fleet_build_placement.rs. W325 section 3d.")
110//! @yah:handoff("ONE BEHAVIOUR CHANGE WORTH A SECOND OPINION: a node started with --sovereign-role non-voter AND a --raft-node-id now refuses EVERY add-learner. I judged that correct - it is a contradiction the operator should see loudly - but the symptom is 'joins mysteriously stop working' rather than a startup refusal. main.rs warns loudly at boot when that pair is present; I did NOT make it fatal, because refusing to start could brick a node mid-roll. Reconsider if it bites.")
111//! @yah:handoff("NOT DONE, and it is a HARD GATE: .yah/schema/machine.toml.schema.json has NOT been regenerated, so sovereign_role is absent from it and schema-drift-guard (scripts/check-schema-drift.sh, a step in .yah/qed/check.toml, run by CI on every push) WILL FAIL. Fix is `cargo run -p xtask -- emit-schemas` from the repo root - it was queued behind ~7 concurrent peer cargo builds for the whole session. Nothing else is required to make this pushable.")
112//! @yah:handoff("ALSO NOT RE-CONFIRMED: `cargo test -p yah-cloud --lib` needs a clean run. Its last real run was 891 passed / 3 failed with all three failures belonging to R772's ingress-collate work and none to this ticket. The re-run is BLOCKED not failing - R555's in-flight AdmissionGrant.secrets field breaks velveteen-exec, and yah-cloud is not a root workspace member so its dev-deps only resolve from the oss/yubaba workspace where that break lives. Re-run from oss/yubaba once R555 lands.")
113//! @yah:verify("cargo run -p xtask -- emit-schemas (from repo root) -- wrote 8 files, exit 0 after an 18m24s build queued behind ~7 concurrent peer cargo jobs. .yah/schema/machine.toml.schema.json now carries the sovereign_role property (anyOf SovereignRole | null, with the full doc comment) and the SovereignRole definition as a oneOf over the two string enums voter / non-voter. The schema-drift-guard gate for THIS ticket is closed.")
114//! @yah:gotcha("emit-schemas IS ALL-OR-NOTHING AND WILL PICK UP A PEER'S UNCOMMITTED WORK. Running it to close this ticket's machine-schema drift also regenerated .yah/schema/secret.toml.schema.json (+34) from R555-F5's in-flight SecretAccess::Recipes / RecipeMatch source. That output is CORRECT for the tree as it stands and was not hand-edited, but it means the schema diff in the working tree is not purely R605-F12's: machine.toml.schema.json (+32) is this ticket, secret.toml.schema.json (+34) is R555. Told Ashguard:spade by party.chat so they carry it with their commit rather than regenerating on top. Anyone splitting these commits needs to split the schema diff too.")
115//! @yah:handoff("ALL GATES CLOSED as of 2026-08-20. Both items listed as outstanding in the earlier handoff notes are done: emit-schemas ran (machine.toml.schema.json carries sovereign_role + the SovereignRole voter/non-voter enum, drift guard satisfied), and cargo test -p yah-cloud --lib is 896 passed / 0 failed once R555 and R772 settled. 45 tests green across workload-spec (9), yubaba lib (13), yubaba raft integration (11), yah-cloud lib (10 of this ticket's, within 896), xtask fleet (2). Ready for review. NOTE for whoever commits: the working tree's schema diff is not purely this ticket - .yah/schema/machine.toml.schema.json (+32) is R605-F12, .yah/schema/secret.toml.schema.json (+34) is R555-F5, both correct generated output from one emit-schemas run. Ashguard:spade has agreed to carry theirs.")
116//! @yah:verify("cargo test -p yah-cloud --lib (from oss/yubaba) -- 896 passed, 0 FAILED, 4 ignored. The blocked check from earlier is now clean: R555 landed the velveteen-exec and TransformRecipe.secrets fixes, R772's ingress-collate work settled (they replaced the CloudConfig::load in collate_workspace_ingress with a narrower machines-only loader, so cross_ref_validate can no longer fail the collate over an unrelated provider typo). All 45 R605-F12 tests across the four crates are green simultaneously on one tree.")
117//! @yah:verify("Confirmed by NAME rather than by total, since a passing count proves nothing about which tests ran: cargo test -p yah-cloud --lib -- role voter voting lists all ten of this ticket's cloud tests green - a_non_voting_member_is_refused_into_its_own_group, a_non_voting_target_has_no_quorum_to_grow, a_refusal_names_the_group_when_fixing_the_role_would_not_help, an_unwritten_role_still_joins_its_group, a_non_voter_is_still_in_the_group_it_names, sovereign_role_round_trips_and_is_omitted_when_unwritten, a_group_with_no_role_is_reported_with_the_declaring_file, either_stated_role_is_clean, a_machine_in_no_group_is_not_asked_for_a_role, unroled_findings_are_ordered_by_file_so_output_is_stable.")
118//!
119//! @yah:ticket(R876-B7, "Node taints are structurally inert for mirror-declared placements: you cannot drain a node, and it fails silently")
120//! @yah:at(2026-09-09T09:05:55Z)
121//! @yah:status(review)
122//! @yah:assignee(agent:bundle-anthropic-ashguard)
123//! @yah:parent(R876)
124//! @yah:severity(high)
125//! @yah:next("SECOND HALF, and it is what makes the relay's headline question answerable: a working taint must produce a MOVE, not a refusal. Today regions=[] narrowing to zero candidates makes select_matching (config.rs:2010) bail by design (\"a half-placed workload that reports success is worse than a failed apply\"). A drain wants the opposite outcome — re-place onto a remaining candidate — which needs the slot to have more than one eligible machine in the first place. Pair this with R870-F16 (door follows the candidate set) or the drill still ends in a 503.")
126//! @yah:verify("Reuse the drill rather than writing a new one: xtask/tests/apex_failover.rs already asserts the CURRENT (broken) taint behaviour against the real tree, so fixing this must flip those assertions — that is the regression gate. Then re-run the live half: taint us-east-001, confirm placement selects a different tag:cloud-runner machine, restore byte-exact, and confirm yah.dev stays 200 throughout.")
127//! @yah:gotcha("IT FAILS SILENTLY, WHICH IS THE SHARP EDGE. \"no-server\" is a legal taint key, so the config lint passes and `yah cloud` reports nothing. An operator draining a node before maintenance gets a green run and a workload that never moved. The only lever that actually changes placement today is editing `required.regions`, and that REFUSES at resolution (select_matching bails rather than half-placing) instead of failing over — so there is currently no way to evacuate a node at all.")
128//! @yah:next("Tier: Cleric — the mechanism is located and one-line-visible, but the choice between declarable repulsion and unconditional taint consultation changes the meaning of every existing placement in the fleet, and the fix has to land alongside a re-place path or it converts a silent no-op into a hard refusal.")
129//! @yah:gotcha("MEASURED, NOT INFERRED — R876-S2's drill, 2026-09-09. `taints = [\"public-ip\", \"no-server\"]` was written onto the REAL .yah/infra/machines/us-east-001.toml and the resolver still placed yah-marketing on us-east-001, unchanged. Restored byte-exact (diff empty, sha256 back to 17dd15e2..., git clean against blob d66ab6d8); yah.dev stayed 200 throughout and no mutating apply was run.")
130//! @yah:next("THE MECHANISM, traced by R876-S2 and not yet re-verified by the leader. Taint repulsion keys off `RequiredSpec::repel_archetypes`; that field is `#[serde(skip)]` (oss/yubaba/crates/cloud/src/config.rs:4067), so a slot declared in a mirror's `required = {...}` ALWAYS deserializes with it empty. `matches` (config.rs:4182) consequently never reads `machine.taints` at all. Confirm both line anchors before editing — the shared tree moves.")
131//! @yah:handoff("SEMANTICS LANDED — repel-by-default + declarable toleration. `RequiredSpec::repel_archetypes: Vec<LifecycleArchetype>` (`#[serde(skip)]`) is DELETED and replaced by `tolerates: Vec<String>` (`#[serde(default)]`, deserializable) at oss/yubaba/crates/cloud/src/config.rs:4319. `matches` (config.rs:4397) no longer iterates a field of `self`: it walks `machine.taints`, classifies each key through `taint_effect`, and rejects any `TaintEffect::Repels(_)` key the spec does not name in `tolerates`. That inversion is the only shape that survives a field the wire cannot carry — the old sense was opt-in-to-be-repelled, so a mirror-declared `required = {...}` always deserialized with an empty archetype set and `machine.taints` was never read at all. Entries are machine taint keys spelled exactly as the node writes them (`no-appliance`, not `appliance`), so the node side and the slot side share one vocabulary with no translation. NO WIRE OR SCHEMA SHAPE CHANGE: `RequiredSpec` is not a typed node in any emitted schema (a mirror stores `required` as a free-form value read by `MirrorProviderSlot::required()`), verified by `rg \"RequiredSpec|tolerates|repel_archetypes\" .yah/schema/*.json` — the only hits are prose inside a doc-comment description.")
132//! @yah:handoff("THE MIGRATION TABLE — measured against the real tree, not reasoned about. FLEET TAINTS, all nine machines (`grep -rE \"^\\s*taints\\s*=\" .yah/infra/machines/*.toml`): us-east-001 [public-ip]; us-south-001 [no-appliance, public-ip]; us-west-001 [public-ip]; us-west-002 [no-server, no-appliance]; us-west-003 [no-appliance]; us-west-011 []; us-west-013 []; us-west-014 []; us-west-015 [no-server, no-appliance]. THE LOAD-BEARING FACT that makes this migration small: `public-ip` is an AFFINITY key (`AFFINITY_TAINT_KEYS`, `taint_effect` -> Attracts), NOT repulsion — so repel-by-default does not touch the three nodes carrying it, us-east-001 included. Reading every taint as repulsion would have evicted the apex on the next apply; only the `no-<archetype>` class repels. Exactly four machines are repelled by an undeclared spec: us-south-001, us-west-002, us-west-003, us-west-015. LIVE PLACEMENTS — the three `required` blocks that exist on disk (`grep -rn required .yah/services/*/mirrors/*.toml`): (1) yah-marketing providers.bundle, cloud.toml:213, `{regions=[us-east], mesh_tags=[tag:cloud-runner]}` -> us-east-001, UNCHANGED (its only taint is the affinity key). (2) yah-cloud providers.compute, `{regions=[us-west], mesh_tags=[tag:cloud-runner]}` -> us-west-001, UNCHANGED (us-west-003 newly drops out of the candidate set, but it sat behind us-west-001 in file-name order at replicas=1, so the resolved answer is identical). (3) yah-cloud-admin providers.compute, same constraint -> us-west-001, UNCHANGED. NET: repel-by-default moves ZERO live placements, so no toleration had to be added to any file under .yah/services/ or .yah/infra/ and none was. No file under .yah/infra/machines/ or .yah/services/ was written by this ticket at all.")
133//! @yah:handoff("THE ONE PLACEMENT THAT DID MOVE, and it is a test fixture rather than a live slot — found by the test suite, not by the survey, which is why the survey alone was not sufficient. `xtask/tests/mirror_ingress.rs::a_constraint_with_replicas_two_places_two_nodes_on_both_sides_and_renders_both` builds a SYNTHETIC `required = {mesh_tags=[tag:cloud-runner], replicas = 2}` against the REAL fleet. Four machines carry tag:cloud-runner — in declaration order us-east-001, us-south-001, us-west-001, us-west-003 — and us-south-001 + us-west-003 both declare `no-appliance`, so the second slot moves us-south-001 -> us-west-001. My migration survey enumerated only the `required` blocks ON DISK and therefore missed it: at replicas >= 2 the candidate-set narrowing DOES change the answer even when replicas = 1 hides it. Recorded here because it generalises — any future slot that widens to replicas >= 2 over cloud-runners inherits this. Fixed at the site that caught it (mirror_ingress.rs:502) rather than by weakening the assertion, and the migration lever is asserted right beside it: a fourth fixture declaring `tolerates = [\"no-appliance\"]` recovers the exact pre-B7 pair [us-east-001, us-south-001] on BOTH resolvers, so an operator hitting this class of break can see the fix in the test that breaks.")
134//! @yah:verify("BASELINE MEASURED BEFORE EDITING, then re-measured after. `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` = 1129 passed / 0 failed / 4 ignored, exit 0 (the run completed and printed its result line before my first Edit; a deferred W298 skew advisory later named config.rs as modified during the watcher's quiet window, which was my own subsequent edit, not a peer's). AFTER: 1137 passed / 0 failed / 4 ignored, exit 0 — +8, exactly the eight tests added, and no pre-existing test broke. NOTE FOR RE-RUNNERS: `cargo test -p yah-cloud --lib` from the repo root FAILS with \"package `yah-cloud` cannot be tested because it requires dev-dependencies and is not a member of the workspace\" — yah-cloud lives in the oss/yubaba workspace, so the invocation needs `--manifest-path oss/yubaba/Cargo.toml`. Also `cargo check --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --all-targets` exit 0 and `-p yubaba --all-targets` exit 0 (yubaba consumes cloud, so it is where the field removal would have surfaced). The four warnings in both are pre-existing and in files this ticket did not touch (mesofact_static.rs unused imports, app_manifest.rs dead field, pond_door.rs unused fn, reconciler/mod.rs non-snake-case).")
135//! @yah:verify("EIGHT NEW UNIT TESTS in config.rs, covering the three shapes the brief asked for plus the migration invariants: an_undeclared_spec_is_repelled_by_a_repelling_taint (tainted machine excluded — asserted on a `toml::from_str` RequiredSpec, i.e. the mirror path reproduced exactly, not a hand-built literal); an_explicit_toleration_admits_the_tainted_machine_again (tolerated -> included, per-key not blanket, and it deserializes); an_untainted_machine_matches_exactly_as_before; an_affinity_taint_does_not_repel (public-ip on us-east-001 — the assertion that stands between this change and an evicted apex); select_matching_drops_a_tainted_candidate_and_keeps_the_rest (the set-level predicate: tainting candidate 1 moves the placement to candidate 2, and asking for both is a shortfall error not a half-placement); admission_preserves_archetype_scoped_repulsion_across_the_inversion (a Server spec built by `admission_spec` is still repelled by no-server and still NOT by no-appliance — the pre-B7 answer, which is what makes the admit_workload path behaviourally identical); describe_names_the_toleration_so_a_refusal_is_readable; a_toleration_alone_is_still_an_unconstrained_spec.")
136//! @yah:verify("REGRESSION GATE FLIPPED, not deleted. `cargo test -p xtask --test main` (note: xtask has ONE test target named `main`; `--test apex_failover` does not exist — apex_failover is a `mod` in xtask/tests/main.rs). Result 65 passed / 1 failed. xtask/tests/apex_failover.rs: the drill's finding-1 test was inverted and renamed every_repelling_taint_at_once_leaves_the_apex_bundle_exactly_where_it_was -> ..._now_makes_the_apex_node_ineligible; it now asserts that ONE repelling key is enough (checked before the all-three case so a regression handling only the union is still caught), that all three refuse, and that restoring us-east-001's real taint list [\"public-ip\"] puts the placement straight back. The module header was rewritten to say the hole is closed. ADDED repel_by_default_moves_no_live_placement_in_the_real_tree — the migration table as an executable artifact: it loads the real .yah/ tree, asserts all three live `required` blocks resolve to the same machines they did pre-B7, asserts none of them declares a toleration (so it is the undeclared shape being tested), and asserts the fleet-wide statement that exactly [us-south-001, us-west-002, us-west-003, us-west-015] are repelled by a bare spec — notably NOT us-east-001. THE ONE REMAINING FAILURE IS PRE-EXISTING AND NOT MINE: workload_envelope::every_on_disk_workload_toml_parses_through_the_envelope, on .yah/infra/state/sources/scrabcake/site/site/workload.toml (`unknown field routes`). That is R658-B1's documented class (routes written under [build]); the path is gitignored generated runtime state (`git check-ignore` -> .yah/.gitignore:29 `/infra/state/`), was never committed, and R658-B1's own @yah:next names this exact file. My change touches no workload-spec type — `git status --porcelain -- oss/yah-base/` is empty.")
137//! @yah:handoff("SCOPE BOUNDARY HELD, deliberately. yah-marketing's candidate set was NOT widened: `.yah/services/yah-marketing/mirrors/cloud.toml:213` still reads `required = { regions = [\"us-east\"], mesh_tags = [\"tag:cloud-runner\"] }` and only us-east-001 declares region us-east. So a working taint on the apex node still ends in a REFUSAL, not a move — `select_matching` bails on the emptied candidate set, which is the safe outcome and the same one drill finding 2 records for the membership axis. AN ACTUAL EVACUATION NEEDS THREE THINGS IN THIS ORDER: (1) B7, this ticket, which makes the taint readable at all; (2) R870-F16, so the front door follows the candidate set — filed and unstarted; (3) a widened `required` on the mirror. Doing (3) before (2) buys a workload that relocates and a yah.dev that 503s, which is why it was not done here. Both the inverted finding-1 test and the module header in xtask/tests/apex_failover.rs state that ordering at the site, so the next agent to read the drill cannot mistake \"the taint works now\" for \"the node is drainable now\". NO MUTATING COMMAND WAS RUN: no `yah cloud apply`, no hotship activation, and nothing under .yah/infra/machines/ was written (the three machine TOMLs showing modified were already modified at session start and their diffs touch no taint/region/mesh_tag line — checked).")
138//! @yah:handoff("GENERATED ARTIFACTS REGENERATED, and one of them is a peer's. `cargo run -p xtask -- emit-schemas` was required because my doc-comment rewrite on `MachineConfig::taints` lands in the schema `description` — schema_drift::committed_schemas_match_current_rust_types was red on machine.toml.schema.json. The regen also swept in mirror.toml.schema.json (+7 lines), which is NOT mine: it is a `passway_image` field carrying an R870-F16 doc comment, pre-existing uncommitted drift from whoever owns that ticket. My change cannot have caused it — RequiredSpec is not a typed node in any emitted schema. Regenerated per CLAUDE.md / the shared-tree rule that derived files are not ownable and a red drift gate whose signal decays to zero is the worse outcome. @Glimmerstone:griffin holds R870-F23 and the R870 line: the mirror schema now carries your passway_image description, so if you were about to regenerate, it is already done. Both schema files are the only two under .yah/schema/ that changed.")
139//! @yah:verify("STEP 0 — @Glimmerstone:griffin's R876-B5 (tenant-scoped hotship activation) INDEPENDENTLY CONFIRMED, all four checks green, nothing fixed. (1) `bash -n scripts/hotship.sh` clean. (2) `./scripts/hotship.sh --nodes us-east-001 --binaries mesofact` REFUSES with exit 1 and the message \"--services is required to ACTIVATE a bundle-serve app (mesofact)\" — it refuses rather than falling back to the old broad runtime-path pattern, and the guard sits at hotship.sh:507 ahead of the version stamp and every remote call. (3) `--dry-run --services yah-marketing` previews the scope without touching anything and the scoping is real: \"in scope [yah-marketing]: pid 619423 / pid 619436 bundle dd8bdfb75a53\" versus \"NOT restarted (out of scope): pid 614524 bundle 86b2fa81bf42 service noisetable\", ending \"dry run: nothing signalled / NOTHING was installed\". (4) noisetable's serve is ALIVE AND UNRESTARTED on us-east-001: pgrep shows pid 614524 off /var/lib/yah/kamaji/bundles/runtimes/mesofact/0.8.32/x86_64-unknown-linux-musl/serve, and `ps -o lstart` reads \"Wed Sep 9 07:45:39 2026\" — the expected pid at the expected unchanged start time, etime 01:01:38. `curl -sS -o /dev/null -w %{http_code} https://yah.dev/` = 200. No real hotship activation was run.")
140//! @yah:verify("BUILDS. `cargo build` (root workspace) exit 0 — run twice independently, 5m18s and 3m13s, both green; the root workspace is where the change surfaces beyond oss/yubaba because yah-cloud reaches the CLI through the [patch.crates-io] bridge. `cargo check --manifest-path oss/yubaba/Cargo.toml -p yubaba --all-targets` exit 0. Clean re-measure of `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` after all annotation writes: 1137 passed / 0 failed / 4 ignored, exit 0 — identical to the first post-change measurement, so the earlier W298 skew advisory naming config.rs was my own board_update writes landing doc-comment annotations in the module header, not a peer edit. A later advisory on the root build named app/yah/cli/src/cloud.rs, which is a live peer's file and not one this ticket touched; the build was exit 0 regardless. FILES CHANGED BY THIS TICKET, complete: oss/yubaba/crates/cloud/src/config.rs, xtask/tests/apex_failover.rs, xtask/tests/mirror_ingress.rs, .yah/schema/machine.toml.schema.json, .yah/schema/mirror.toml.schema.json. Nothing under .yah/infra/ or .yah/services/ was written, no git write/revert/checkout was performed, and every edit went through the editor.")
141//! @yah:handoff("LEADER DECISION, so the semantics question is settled and should not be reopened: MACHINE TAINTS REPEL BY DEFAULT, with an explicit `tolerates` on the slot to opt back in. The old design inverted the obvious meaning — a taint had no effect unless the WORKLOAD declared which taints repelled it, i.e. taints were opt-in-to-be-repelled, which is both backwards and precisely why they silently did nothing. `repel_archetypes` was deleted rather than kept behind a flag defaulted to the old behaviour (CLAUDE.md, \"break it, don't tape it\").")
142//! @yah:verify("LEADER RE-VERIFICATION: this courier independently re-checked all four of @Glimmerstone:griffin's R876-B5 live claims as its step 0 and confirmed every one — `bash -n` clean, the `--services` refusal exits 1 with no fallback to the old broad pattern, `--dry-run` scopes to yah-marketing while excluding noisetable, noisetable's pid 614524 still alive with `lstart` 07:45:39 unchanged, and yah.dev 200. Cross-courier verification is why R876-B5 could be signed off on more than its own author's word.")
143//! @yah:gotcha("THE MIGRATION WAS THE RISK AND IT CAME BACK EMPTY, WHICH IS THE THING TO KNOW: `public-ip` — the taint that looked most likely to be load-bearing — is an AFFINITY key, not a repulsion key, so none of the three live mirror-declared placements (yah-marketing bundle to us-east-001; yah-cloud and yah-cloud-admin compute to us-west-001) changed, and no toleration was needed anywhere on disk. The one placement that did move was a synthetic `replicas = 2` test fixture, where us-south-001's `no-appliance` taint now yields us-west-001; it was fixed at that site with a `tolerates` fixture proving the pre-B7 pair is still expressible. Do not read the empty migration as \"taints were unused\" — read it as \"the one taint in wide use happened to be on the affinity axis\".")
144//!
145//! @yah:ticket(R870-F23, "Render and supervise the inner door: the service.toml + domain-manifest join that feeds passway's PathRouter config")
146//! @yah:status(review)
147//! @yah:phase(P2)
148//! @yah:at(2026-09-11T00:25:14Z)
149//! @yah:assignee(agent:bundle-anthropic-ashguard)
150//! @yah:parent(R870)
151//! @yah:next("THE CONSUMER SIDE IS DONE AND ITS FORMAT IS FIXED (R870-T18, in review). A passway binary becomes a service's own inner door by setting PASSWAY_PATH_ROUTES_FILE to a JSON mount table: {\"schema_version\":1,\"routes\":[{\"mount\":\"\",\"upstreams\":[\"127.0.0.1:8081\"]},{\"mount\":\"/app\",\"upstreams\":[\"127.0.0.1:8082\"],\"headers\":{\"cross-origin-opener-policy\":\"same-origin\"}}]}. Parser + validation: oss/passway/crates/passway/src/path_routes_file.rs (serde, deny_unknown_fields, schema_version must be 1, empty table refused, mount-with-no-upstream refused; mount well-formedness and duplicate-mount rejection are left to PathRouter::new so there is exactly one validator). Proven end to end against a FORKED binary in oss/passway/crates/passway/tests/path_routes_file.rs. This ticket is the producer: write that file.")
152//! @yah:next("WHY THIS IS A SEPARATE TICKET AND NOT HALF OF R870-T18. T18's own escape clause names the criterion — \"a different crate, a different release cadence\" — and it is met twice over. (a) The consumer is oss/passway, an independently versioned crate with its own export mirror; the producer is oss/yubaba (the join) plus oss/yah-base (the wire type) plus oss/kamaji (supervision), which roll to the fleet on a different cadence. (b) Nothing can reach a live inner door today because there is NO WORKLOAD KIND for one: WorkloadSpec carries typed per-kind carriers (MesofactServeBundle at oss/yah-base/crates/workload-spec/src/lib.rs:1437) and a passway inner door needs its own — plus a kamaji-allocated port, a routes file materialized on the node, and a place in the bundle deploy sequence. Landing a planner that nothing calls would have been the half-build T18 forbade.")
153//! @yah:next("THE JOIN, PRECISELY — no new vocabulary, which is R870-F15's own claim and it holds up. Inputs: .yah/services/<svc>/service.toml (ServiceComponent { id, kind, mount, ... }, config.rs:3427) and .yah/domains/<zone>.toml (DomainRoute { path, headers, mode }, config.rs:4558, where front_door = passway). Per mount: mount = path_route::mount_from_component(component.mount) — that function already exists and is already the ONE place the \"app\"/None to \"/app\"/\"\" translation happens; headers = the DomainRoute whose route_path_prefix(path) equals normalize_mount(component.mount) (cross_ref_validate already PROVES those two agree, config.rs:1688-1725, so the join cannot silently mismatch); upstreams = the address of the deployed unit serving that mount. Only the last one is placement-time and is why this needs the workload kind above. Group by DEPLOYED UNIT, not by component: every bundle-tier component of a service shares ONE bundle workload (that is config 1, R870-B11), so config-1 mounts collapse to a single root upstream and only independently-deployed units earn their own mount.")
154//! @yah:next("THE TWO ADMISSION RULES, and where each one goes. Both belong to the GENERATOR, never to passway — passway proxies whatever PathRouter it is handed and has no view of how many components a service declares. (1) A service with ONE independently-deployed unit gets NO inner tier at all — enforce by construction: the planner returns Option<InnerDoorPlan> and answers None below two units, so there is no config to write and no process to supervise, and the negative is assertable on the ABSENCE of the plan rather than on a site staying up. (2) A component cannot be both bundle-staged (config 1) and its own workload. R870-B11 landed the config-1-internal half in CloudConfig::cross_ref_validate (config.rs:1621-1657, two bundle components at one mount are refused); put this half in the SAME loop rather than a parallel one. NOTE, checked not assumed: the second half is NOT EXPRESSIBLE TODAY — [providers.bundle] is a per-MIRROR slot, not per-component, so there is no way to say \"give this one component its own workload\" at all. The rule becomes writable in the same commit that introduces that vocabulary, which is this ticket. Do not invent the vocabulary separately.")
155//! @yah:gotcha("DESIGN WRINKLE FOUND WHILE BUILDING R870-T18, and it is an operator call, not a coding one. passway ALWAYS terminates TLS on its listener: TlsMode has exactly two variants, Manual and Acme (oss/passway/crates/passway/src/tls.rs:215), and main() unconditionally calls proxy_service.add_tls_with_settings(&listen, None, tls_settings). So an inner door on loopback still needs a cert on disk, and the outer door still needs PASSWAY_UPSTREAM_TLS=true plus an SNI to reach it. That works — T18's binary-level test does exactly this with an rcgen self-signed leaf — but it means the \"cheap inner tier\" costs a cert, a renewal story, and an upstream TLS handshake per request on loopback. The obvious fix is a plaintext listener mode, and it was deliberately NOT taken in T18: adding a way for a public-facing trust-boundary door to serve cleartext is a security decision with a blast radius past this relay. Decide it before building the supervisor, because it changes what the workload spec has to carry.")
156//! @yah:verify("A two-component service whose components deploy INDEPENDENTLY gets an inner door: one yah cloud apply leaves both https://<host>/ and https://<host>/app/ at 200, and curl -sI on /app/ carries cross-origin-opener-policy: same-origin AND cross-origin-embedder-policy: require-corp from the /app/* route in the domain manifest, while / carries neither.")
157//! @yah:verify("THE NEGATIVE, asserted on absence rather than on uptime: a single-component service (yah-marketing) produces NO inner-door config and NO inner-door process — no routes file materialized on the node, no extra supervised workload in kamaji's table, and a byte-identical workload spec to today. A unit test on the planner returning None is the cheap half; the node-side absence check is the half that matters.")
158//! @yah:gotcha("OPERATOR CALL ASKED AND NOT ANSWERED (R870 relay leader, session:abde2cbb, 2026-09-09). The TLS question in this ticket first gotcha was put to the operator as a three-way choice and the prompt timed out unanswered after 30 minutes, so it remains genuinely open — it was not skipped and not decided by default. The three options as framed, so whoever picks this up does not have to re-derive them: (A) add a plaintext listener mode gated so it is structurally impossible to combine with a public bind — refuse at config load unless the bind is loopback, keep it mutually exclusive with ACME/cert paths; this was the leader recommendation, on the grounds that it makes the inner tier actually cheap as R870-F15 design claimed while keeping the risk a bounded testable invariant rather than an operator remembering not to misconfigure it. (B) keep TLS everywhere and have F23 carry a cert-issuance plus renewal story for every inner door, which is safest by construction and already proven working in R870-T18 binary-level test with an rcgen self-signed leaf, but makes every service with 2+ independently-deployed components pay a cert, a renewal and a loopback handshake per request. (C) park the tier — nothing regresses, because config 1 (bundle staging, R870-B11, in review) already covers the deploy-together case, which is the one noisetable actually needs. THIS IS THE ONLY THING BLOCKING F23 DESIGN; the join itself, both admission rules and the workload-kind vocabulary are all specified in this ticket next entries and need no further decisions.")
159//! @yah:handoff("OPERATOR CALL ANSWERED 2026-09-09: option (A), the loopback-only plaintext listener. It was re-put with ONE fact the earlier framing did not have, and that fact inverts the safety argument the three options were weighed on: option (B) was never \"already proven working\". pingora defaults verify_cert: true (pingora-core-0.8.1/src/upstreams/peer.rs:479, read not assumed) and passway NEVER overrides it — there is no verify_cert anywhere in oss/passway/crates/passway/src. R870-T18's test drove the door from an HTTP client with danger_accept_invalid_certs, not from an outer passway, so the outer-to-inner leg was untested. Since no CA issues for 127.0.0.1, \"keep TLS everywhere\" required a SECOND unbuilt change — a way to disable or pin upstream certificate verification on a public-facing door — traded for encrypting a hop that never leaves the loopback interface. (A) is strictly the smaller security surface, not merely the cheaper one.")
160//! @yah:handoff("PASSWAY: TlsMode::Plaintext, selected by PASSWAY_TLS_MODE=plaintext (a third value on the EXISTING discriminator, not a new bool env var — one variable owns the listener's TLS mode). All guards live in ONE function, tls.rs parse_listener_tls_mode, and each is a boot failure naming what to change: the bind must parse as a LITERAL loopback SocketAddr (0.0.0.0:443 — the default — is refused, and so is a hostname this process cannot prove); PASSWAY_TLS_CERT/KEY must be unset, so a configured public door cannot go cleartext by ADDING a variable rather than removing two; LISTEN_FDS is refused outright because under socket activation PASSWAY_LISTEN is only the key pingora looks the socket up by and proves nothing about the bind. An unrecognized PASSWAY_TLS_MODE is now also a boot failure instead of a silent fall-through to manual. main() reads the mode BEFORE the cert paths (plaintext has none), branches to proxy_service.add_tcp(&listen), and build_tls_settings returns Err rather than panicking on the variant it can no longer be handed.")
161//! @yah:handoff("THE VOCABULARY, and admission rule 2 made UNREPRESENTABLE rather than refused. ServiceComponent gains deploy: DeployTier { Bundle (default), Workload } — oss/yubaba/crates/cloud/src/config.rs. That is the per-component slot [providers.bundle] could not express, and because it is ONE field with two values, \"both bundle-staged and its own workload\" has no spelling at all; there is no rule to enforce. What remained checkable — two components claiming one mount — went into R870-B11's EXISTING cross_ref_validate loop rather than a parallel one. That loop previously filtered on kind and so skipped the workload tier entirely; it now covers both tiers, and only the explanation branches (bundle/bundle = one storage prefix in one bundle; workload/workload = one prefix in the inner-door table; mixed = the mount names two things serving one prefix). skip_serializing_if on the default keeps every existing service.toml byte-identical.")
162//! @yah:handoff("THE JOIN: new module oss/yubaba/crates/cloud/src/inner_door.rs. plan(&ServiceConfig, &domains) -> Result<Option<InnerDoorPlan>>. Rule 1 is by construction — None below two DEPLOYED UNITS, so the negative is assertable on the absence of a plan. Grouping is per unit but mounts are per COMPONENT: N bundle components collapse to one DeployedUnit::Bundle yet keep N mounts, because a bundle sub-mount can carry route headers the root does not and the collapse-to-root shape would silently drop them. Headers come from the DomainRoute whose route_path_prefix equals the component's normalize_mount — cross_ref_validate already PROVES those agree, so the lookup cannot mismatch. Err is reserved for one case: two-plus units with no root mount, which would 503 every unclaimed path. routes_file() refuses an unresolved upstream instead of skipping the mount — a dropped mount does not 503, it falls through to the root and serves the WRONG component with a 200. passway_mount() composes with normalize_mount rather than trimming slashes a second time.")
163//! @yah:handoff("SUPERVISION: WorkloadSpec gains files: Vec<InlineFile { path, content, mode }> (oss/yah-base/crates/workload-spec/src/lib.rs, appended last, serde(default), no skip_serializing_if — postcard is positional, so every pre-existing spec decodes to an empty vec). kamaji's NATIVE backend writes them in spawn_child BEFORE exec and on every respawn (materialize_files, oss/kamaji/crates/kamaji/src/native.rs); containerd/docker/microvm call the new kamaji::reject_unmaterializable_files and REFUSE such a spec by name rather than starting a door against a file that is not there — a silently-skipped route table comes up healthy and routes wrongly, which is worse than not starting. InnerDoorPlan::workload(listen_port, address) renders Workload::Container: argv /usr/local/bin/passway, env PASSWAY_TLS_MODE=plaintext + PASSWAY_LISTEN=127.0.0.1:<port> (the 127.0.0.1 is literal, NOT a parameter, so a wrong port cannot make the door reachable) + PASSWAY_PATH_ROUTES_FILE, and the table itself as the one InlineFile. Not a new Workload variant: TenantPasswayWorkload earns one by carrying config kamaji acts on; an inner door's whole config is an argv, three env vars and a file, so a variant would buy only exhaustive-match churn in peer-owned kamaji-proto (the R572-F1 trade).")
164//! @yah:verify("cargo test -p passway (oss/passway) = 205 lib + 43 + 35 integration, 283 passed / 0 failed, up from the 275 baseline @Ashguard:abde2cbb recorded on R870-T21. 8 new lib tests in tls::tests and 2 new integration tests in tests/path_routes_file.rs. THE END-TO-END ONE IS THE POINT: a_cleartext_inner_door_serves_the_same_mount_table_with_no_certificate forks a REAL passway binary with no PASSWAY_TLS_CERT set at all and asserts the same two-mount split and the same per-mount COOP header over plain http:// — i.e. the tier the operator authorized actually costs a process and nothing else. Its negative, a_cleartext_door_on_a_reachable_bind_refuses_to_start, spawns the binary on 0.0.0.0:0 (the DEFAULT bind, so it is the exact misconfiguration that would make an inner door a public cleartext one) and asserts a non-zero exit whose message names the bind.")
165//! @yah:verify("cargo test -p yah-cloud --lib = 1150 passed / 0 failed (11 new in inner_door::tests, 2 new in config::tests). The cheap half of this ticket's own negative is a_single_unit_service_gets_no_inner_door plus several_bundle_components_are_one_unit_and_still_get_no_door — three components sharing one bundle are still ONE unit and still get no door, which is the case that would be easy to get wrong by counting components. cargo test -p yubaba --lib = 952/0. cargo test -p kamaji --lib --all-features = 208/0 (2 new; the materialization test asserts ORDERING by having the child cat the file into a second path, not merely that the file exists). cargo test -p yah-workload-spec --all-features = 205 + 101, 0 failed. cargo test --workspace --all-features in oss/kamaji = 18+208+5+303, all green in-package.")
166//! @yah:gotcha("ONE PRE-EXISTING FLAKE, DIAGNOSED NOT WAVED THROUGH. kamaji-bin's server::tests::tenant_passway::the_list_reports_the_digest_of_the_spec_it_was_deployed_with FAILS under `cargo test --workspace --all-features` in oss/kamaji, reproducibly, and PASSES 303/303 under `cargo test -p kamaji-bin --lib --all-features` both parallel AND --test-threads=1. So it is cross-PACKAGE contention, not in-package parallelism and not this change: the failing assertion is the second deploy failing to Ack after `free_port()` (server.rs ~:8667) handed back a port another package's test binary had taken between the probe and the bind — a TOCTOU in the helper. Nothing in this ticket adds a port or touches that path; WorkloadSpec::files cannot reach it, since Workload::TenantPassway carries a TenantPasswayWorkload and no WorkloadSpec at all. Worth a real fix (bind-and-hold instead of probe-and-release) but it is not this relay's.")
167//! @yah:gotcha("ROLL ORDER MATTERS AND IS NOT THE USUAL \"JSON IGNORES UNKNOWN KEYS\" ANSWER — flagged by @Ashguard:eclipse (session:e188ccc2, R881-T6) mid-session. All three prod voters now run kamaji+yubaba 0.8.37-h5 (us-south-001 and us-west-001 rolled 2026-09-09; us-east-001 on 0.8.37-h1/h2), all built BEFORE WorkloadSpec::files existed. WorkloadSpec has no deny_unknown_fields, so on the JSON leg an un-rolled node ignores the field exactly as R870-B6's `origin` did. The postcard leg is the one that does NOT forgive: it is positional and non-self-describing, so a new yubaba encoding a spec with a trailing `files` to an old kamaji decoder is a DESYNC, not an ignored key. Before deploying any inner door, confirm which codec that node's kamaji link uses (kamaji-proto/src/codec.rs) and roll kamaji first if it is postcard. Nothing regresses until something actually SETS files — every existing spec encodes an empty vec — but the ordering is a real constraint, not a formality.")
168//! @yah:handoff("WIDER THAN THE TITLE, all mechanical and all compiler-verified. Adding two fields to types this many call sites construct exhaustively meant ~45 initializer repairs across FOUR workspaces: oss/yah-base (workload-spec + local-driver), oss/kamaji (incl. peer-owned kamaji-proto/src/codec.rs and kamaji-containerd-core), oss/yubaba, and the root (crates/yah/hub, app/yah/cli). Each is one line — `files: Vec::new(),` or `deploy: Default::default(),` — with no semantic content; they were driven off E0063 spans, not grep, so none was guessed. NOTE the sweep needs --all-features AND `cargo test --no-run`: `cargo check --all-targets` alone missed sites behind feature gates and in examples/. ALSO REGENERATED (both are pure functions of the tree, so this is not authorship): .yah/schema/{workload,service}.toml.schema.json via `cargo run -p xtask -- emit-schemas` and packages/yah/workload-spec/index.ts via the export-ts bin. Both drift gates still report red because they compare against GIT, and this camp defers commits — they go green with the commit, and the regenerated content is correct.")
169//! @yah:handoff("PHASE 1 DONE — the tier EXISTS and every piece of it is proven in isolation: the cleartext listener (proven through a forked binary), the vocabulary, the join with both admission rules, the wire carrier, and node-side materialization + restart. What is NOT done is the last hop: nothing CALLS plan() yet, so `yah cloud apply` still produces no inner door. That is deliberate rather than abandoned — it is placement work with a live-fleet verify attached, and it is the whole of phase 2.")
170//! @yah:handoff("Tree anchor at handoff: 6f984b53a9dd1a291d29fe7d4cb544b47d4f65e6 — the shared tree as I left it. Diff against it (`git diff 6f984b53a9dd1a291d29fe7d4cb544b47d4f65e6..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
171//! @yah:next("ONE DESIGN QUESTION PHASE 1 LEFT OPEN, stated so it is not rediscovered as a bug. A bundle-tier component at a non-root mount now gets its own entry in the inner-door table pointing at the SAME bundle upstream, purely so the domain manifest's per-path response headers can be applied (see a_bundle_components_sub_mount_keeps_its_headers_and_the_bundle_upstream). That is correct for headers and harmless for routing, but it means the inner door re-states routing the bundle already does internally. If the outer door or the Worker is ALREADY applying those headers for a config-1 service, the inner door would apply them twice — check which tier owns route headers for a passway front door before wiring step 4, because R746 put ROUTE_HEADERS into the Cloudflare Worker and I did not confirm the passway-front-door equivalent.")
172//! @yah:verify("THE LIVE HALF, unrun and needing a fleet: a two-component service whose components deploy independently gets one inner door from one `yah cloud apply`, with https://<host>/ and https://<host>/app/ both 200 and `curl -sI` on /app/ carrying cross-origin-opener-policy: same-origin AND cross-origin-embedder-policy: require-corp while / carries neither. The config-side half of exactly that assertion is already green as inner_door::tests::each_mount_carries_only_its_own_routes_headers, and the transport-side half as the forked-binary cleartext test — what remains unproven is only that apply joins them. THE NEGATIVE'S node-side half is also unrun: for a single-component service (yah-marketing), assert NO routes file is materialized on the node and NO extra workload appears in kamaji's table.")
173//! @yah:next("PHASE 2 IS FIVE STEPS AND EVERY INPUT ALREADY EXISTS. (1) Call inner_door::plan(&svc.service, &cfg.domains) once per service in the apply path; Ok(None) is the common answer and means do nothing at all. (2) Allocate the loopback port. It is deliberately a PARAMETER of InnerDoorPlan::workload rather than config — which port is free is a property of the node — so this is the only genuinely new decision: either take it from kamaji's ledger (oss/kamaji/crates/kamaji/src/ports.rs) or pin one per service. (3) Resolve each DeployedUnit to an address for the `address` closure: DeployedUnit::Bundle is the service's one bundle workload (R870-B11), DeployedUnit::Component(id) is that component's own workload. (4) Deploy the rendered workload in the bundle deploy sequence — BEFORE the outer door is repointed, since the door 503s until its upstream is up. (5) Repoint the outer door's PASSWAY_UPSTREAMS at 127.0.0.1:<port> instead of at the bundle, via IngressPlan::resolve_upstreams (oss/yubaba/crates/cloud/src/reconciler/ingress.rs:397).")
174//! @yah:handoff("DEFECT IN THIS TICKET'S OWN CHANGE, CAUGHT IN REVIEW BY @Ashguard:eclipse (session:e188ccc2) AND FIXED BEFORE IT LEFT THE TREE. WorkloadSpec rides the postcard `Deploy` frame (kamaji-proto/src/messages.rs:373, and V7's own stanza names `Workload::Container(WorkloadSpec)` as what that frame carries), and kamaji-proto/src/version.rs states the rule twice: every field on a postcard message is mandatory and always encoded, and the only compatibility mechanism is a ProtocolVersion bump. V2/V4/V5/V6 were each exactly \"a field appended to a struct\" and each got one. `files` is that shape and I had not bumped. The reasoning that made me miss it is the one V6's stanza already refutes: `#[serde(default)]` makes an OLD spec decode fine, so the JSON leg really is unaffected — but `default` only affects DEserialization, so a new yubaba still ENCODES a length varint an old kamaji reads as the next field and misparses from there. Now V8, CURRENT = V8, with a stanza naming the wrong reasoning rather than only the rule. cargo test -p kamaji-proto --all-features = 33/0; oss/kamaji workspace = 18+208+5+303+2+2+2+1, 0 failed.")
175//! @yah:gotcha("CORRECTION TO THE FLAKE GOTCHA ABOVE — my characterization was too narrow and would mislead the next reader, so read this one instead. I wrote that the tenant_passway digest test is \"green 303/303 in-package, fails only workspace-wide\". @Ashguard:eclipse measured the counter-example on the same tree: `cargo test -p kamaji -p kamaji-bin --lib --all-features`, in-package and parallel, failed a DIFFERENT test in the same module — deploy_arms_the_declared_socket_and_stop_releases_it (server.rs:8590) — and it failed identically before my sweep. On my own later run the workspace-wide invocation came back 303/0. So the truth is: at least two tests in server::tests::tenant_passway are intermittently flaky in BOTH configurations, the cause is `free_port()` probe-and-release losing the port between the probe and the bind (server.rs ~:8667), and it predates R870-F23. Do NOT read an in-package red there as a regression, and do not read a single green run as proof either. The fix is bind-and-hold; it belongs to neither R870 nor R881 and is unfiled — @Ashguard:eclipse tried and board.open refused for want of a parent relay.")
176//! @yah:gotcha("CONSEQUENCE OF THE V8 BUMP FOR ANY FLEET OPERATION, not just for this relay — relayed by @Ashguard:eclipse (session:e188ccc2) who is holding the fleet on R881-T6, and worth acting on before the next roll. The tree is now ProtocolVersion::V8; EVERY node runs a pre-V8 pair (us-east-001 on 0.8.37-h1/h2, us-south-001 and us-west-001 on 0.8.37-h5, the other six on 0.8.28-0.8.34). Nothing is broken, because each node is internally matched and the protocol is a node-local UDS. What changed is that `hotship --binaries yubaba` ALONE — or `kamaji` alone — is now a footgun on every node: it puts a V8 binary against a V7 sibling, and per version.rs:71 that does not fail cleanly, it misreads every field after the desync, \"which is how a wrong image or a wrong volume mount gets deployed instead of an error\". Ship the PAIR. That was harmless before this ticket and is not now.")
177//! @yah:handoff("PHASE 2 LANDED — `yah cloud apply` now produces an inner door. All five steps, with the call sites. (1) PLAN: `service_inner_door` (app/yah/cli/src/cloud.rs:9212) calls `inner_door::plan` once per service; `Ok(None)` is the answer for every service on disk today and returns before anything else runs. (2) PORT: derived, not allocated — `inner_door::listen_port(service)` (oss/yubaba/crates/cloud/src/inner_door.rs:346). (3) RESOLVE: `InnerDoorPlan::resolve_addresses` (inner_door.rs:418) maps each unit to a mesh ident via `unit_ident` (:398) and looks it up with the new `ServiceRecordFanout::address_for_ident` (reconciler/service_discovery.rs:426). (4) DEPLOY: `deploy_inner_door` (cloud.rs:9274), called at the END of the deploy-phase closure in BOTH apply paths — `reconcile_root` (cloud.rs:11877) and `handle_mirror_up` (cloud.rs:7100) — so it is after every unit registered a record and before the front-door phase repoints anything. (5) REPOINT: `IngressPlan::point_at_inner_door` (reconciler/ingress.rs:438), called from `reconcile_ingress_edge` (cloud.rs:7726).")
178//! @yah:handoff("STEP 2 ANSWERED — the port is DERIVED from the service name, not taken from kamaji's ledger, and the three facts that decided it were read rather than assumed. (a) `LedgerPorts` is node-local (a JSON file beside the supervisor's state dir) and yubaba's HTTP surface exposes no allocation verb at all — yubaba/src/lib.rs routes /workloads/*, /services, /node/*, and nothing for ports — so an apply has no way to ask. (b) A stated number is HONOURED, not rejected, on the path this workload takes: R844-F14's pin rule bites inside `LedgerPorts::resolve_set`, and `NativeRuntime::resolve_declared_ports` (oss/kamaji/crates/kamaji/src/native.rs:280) filters `pin.is_none()` BEFORE calling it. That matters because `PASSWAY_LISTEN` must carry the number, and a number the node picks after the spec is rendered cannot be in it. (c) A collision is not representable: the ledger allocates on the workload's MESH ip, an inner door binds loopback, so 100.64.0.3:14210 and 127.0.0.1:14210 are different sockets. The window is 10000-19999, deliberately below Linux's default ephemeral floor (32768) where `pick_free_port`'s bind(:0) draws from. FNV-1a written out inline rather than `DefaultHasher`, whose stability std does not promise — this number goes into a deployed door's env AND the outer door's upstream list, and a toolchain bump silently moving it would repoint one tier and not the other.")
179//! @yah:handoff("THE OPEN HEADER QUESTION IS ANSWERED, AND THE ANSWER IS NO CHANGE — grounded by reading, not assumed. The question was whether the passway FRONT door also applies per-route response headers. It does not: `PassProxy::response_filter` (oss/passway/crates/passway/src/proxy.rs:700) iterates `ctx.route_headers`, and its own doc at :694 states that vector is empty for `RoutingStrategy::ByHost` — which is what every outer door is. So the outer tier owns no headers and there is no double-apply to resolve there. A THIRD tier the question did not name does apply them, and is worth recording: the mesofact bundle ORIGIN, via `MESOFACT_ROUTE_HEADERS` set by `add_declared_route_headers` (app/yah/cli/src/cloud.rs:8612). For a mount served by `DeployedUnit::Bundle` both that origin and the inner door apply the route's headers — but CONVERGENTLY, not duplicatively: both read the same `.yah/domains` route map, `PathRouter` does not strip the mount prefix (oss/passway/crates/passway/src/path_route.rs has no strip/rewrite), so both match the same request path, and both use insert-semantics (`HeaderMap::insert` in mesofact's `RouteHeaderTable::apply`, `insert_header` in passway) — one header, one value. Do NOT collapse it to one owner. The bundle origin's coverage is strictly WIDER: it applies headers for a declared route that has no component mount (a `/docs/*` route served out of the root bundle's dist), which the inner door has no entry for. And the inner door is the ONLY owner for a `DeployTier::Workload` mount, since nothing hands such a component a header table. The two are complementary; removing either loses headers somewhere.")
180//! @yah:handoff("PLUMBING BUILT BECAUSE STEPS 3 AND 5 NEEDED IT, all three of which did not exist. (1) `inner_door::component_workload_ident(service, component_id)` (inner_door.rs:371) — the mesh identity a workload-tier component registers under. It is a NAMING RULE stated here because nothing else states it: a bundle's ident is a mirror fact (`BundleSlot::workload_name`, renameable with `name = \"...\"`), but a workload-tier component has no slot of its own, since `[providers.*]` is per-kind-per-mirror — the exact gap `DeployTier` was added to close. Folded through `reconciler::native_support::sanitize_ident`, which I widened from private to `pub(crate) mod` (reconciler/mod.rs) rather than writing a second normalizer. Getting the ident wrong fails LOUDLY: `routes_file` refuses a mount whose unit resolved to nothing, naming the unit. (2) `ServiceRecordFanout::address_for_ident` — deliberately SINGULAR where `upstreams_for` is plural. An inner door proxies over loopback to a unit on its own node; handed a fleet-wide set it would dial across the mesh, which is not what the cleartext-listener safety argument assumed. Two nodes, two addresses is ambiguity (None), not load balancing. Port selection follows `port_for`'s discipline exactly (`kamaji::DEFAULT_PORT_NAME` first, then the sole anonymous port) so a unit resolves the same way at both tiers or neither. (3) `IngressPlan::point_at_inner_door` OVERRIDES where `resolve_upstreams`/`resolve_ports` fill in — it clears both halves and then goes through those same two methods, so this stays the only place in the crate writing those fields. The ticket's step 5 named `resolve_upstreams`; used alone it is WRONG, because it skips a rule that already has an `upstream_host` and every mirror on disk pins one. A pin names ONE unit, and fronting a two-unit service from one unit serves half the site and 503s the other half, so the pin has to lose here and nowhere else.")
181//! @yah:handoff("TWO PLACEMENT DECISIONS PHASE 2 HAD TO MAKE, both recorded at the site. (a) The inner door lands on the FRONT DOORS, not the workload nodes — the outer door dials 127.0.0.1, so a door anywhere else is a door the outer tier cannot reach. `ingress_topology` (cloud.rs:9230) recomputes `resolve_ingress_placements` + `plan_ingress` in the deploy phase to learn that set; both are pure, so this costs no network and cannot disagree with the front-door phase's own answer. (b) SELF-DISCOVERY IS TURNED OFF for an inner-door service. `PASSWAY_UPSTREAM_SOURCE=yubaba` makes the door poll for the fronted workload's records and use those INSTEAD of its static set — which would route straight past the inner door to whichever unit registered under the mirror's ident, silently undoing step 5. The rendered note says so in its own words rather than reusing R844-F20's \"NOT self-discoverable ... MANUAL step\" wording, because this is not a degradation: the address is derived and byte-identical on every apply. (c) A mirror with two units and NO declared front door SKIPS with a note rather than failing — `reconcile_mirror_ingress` already returns early on `plans.is_empty()`, so there would be no outer door to repoint and nothing that can 503. Every `shape = \"local\"` dev mirror is in that state; bailing there would have broken `yah mirror up`.")
182//! @yah:handoff("DISCOVERED WORK, FIXED IN THIS PASS, NOT FILED AS A FOLLOWUP. `cargo test -p yah-cloud --lib` was 1161/2 on arrival, and the two reds were NOT mine and NOT a flake: `cloud_init::tests::{rendered_runcmd_entries_are_all_strings, coordinator_prestage_only_for_standalone}`. Cause: oss/yubaba/crates/cloud/templates/mirror.yml:107-108, the two R858-F17 turso-backup-helper runcmd entries, were written as BARE YAML scalars containing a `: ` — which makes the whole entry parse as a Mapping, so cloud-init skips it and the helpers never land on a provisioned node. The file is committed and clean (last touched by a8f0d501, i.e. it regressed AFTER phase 1's 1150/0 measurement), no live peer owns it, and the fix is two lines: double-quote the entries and escape the inner quotes. Both tests are green and the comment at the site names the gate. This is a real provisioning defect, not just a red test — a node provisioned since a8f0d501 has no turso-backup-hydrate / turso-backup-tail, and R858-F17's own design makes durability-declaring workloads refuse to deploy without them. Worth a look at whether any node was provisioned in that window.")
183//! @yah:verify("PHASE 2 MEASURED, every number run by me and read. `cargo test -p yah-cloud --lib` = 1163 passed / 0 failed (baseline 1150; +13 — 6 in inner_door::tests, 4 in service_discovery::tests, 3 in ingress::tests). `cargo test -p yah --lib` = 1549 / 0 (+3 new in a new `inner_door_apply_tests` module). `cargo test -p xtask --test main mirror_ingress` = 13 / 0 (baseline 11; +2). `cargo test -p yubaba --lib` = 952 / 0, exactly the baseline. `cargo test -p passway` in oss/passway = 205 + 43 + 37 = 285 / 0 against the 283 baseline, and `cargo test -p kamaji --lib --all-features` = 217 / 0 against 208 — BOTH deltas are peers', not mine: I touched neither crate. Sweeps: `cargo test --workspace --all-features --no-run` clean, and the same in oss/yubaba clean (only the two pre-existing unused-import warnings in a peer's in-flight mesofact_static.rs). NO SCHEMA REGEN NEEDED — this pass added functions, constants and one module-visibility widening, and no serde-visible field on any generator input, so .yah/schema/*.json and packages/yah/workload-spec/index.ts are untouched by construction.")
184//! @yah:verify("THE NEGATIVE IS ASSERTED IN THREE PLACES, at three different altitudes, because it is the claim the live fleet rests on. (1) `xtask/tests/mirror_ingress.rs::no_service_on_disk_gets_an_inner_door` walks the REAL `.yah/services/` tree and asserts every service plans `None`. That is the strongest form available without a fleet: the only way to be wrong about it is for a service to acquire `deploy = \"workload\"`, at which point the test names the service. Sibling `every_services_derived_inner_door_port_is_distinct` pins the port derivation against the real service list. (2) `cloud::inner_door_apply_tests::a_single_unit_service_leaves_the_outer_door_exactly_as_it_was` builds a two-component fixture that is BYTE-FOR-BYTE the positive test's, with one word changed (`workload` -> `bundle`), and asserts the rendered `PASSWAY_UPSTREAMS` is still the mirror's pinned `noisetable.com=100.64.0.3:8080`. So the difference between the two outcomes is provably that one field. (3) `inner_door::tests::{a_single_unit_service_gets_no_inner_door, several_bundle_components_are_one_unit_and_still_get_no_door}` from phase 1, still green. THE POSITIVE: `a_two_unit_service_repoints_the_outer_door_at_its_inner_door` (outer door renders `noisetable.com=127.0.0.1:<derived>`, and the port is asserted equal to what the door itself binds — two call sites in two phases that must not be able to disagree) and `the_rendered_table_splits_the_mounts_and_carries_only_their_own_headers` (both units addressed, COOP+COEP on /app and ABSENT on the root).")
185//! @yah:gotcha("TRANSIENT BUILD FAILURE SEEN AND DISPROVEN, recorded so the next reader does not re-chase it. The first `cargo test --workspace --all-features --no-run` came back with `can't find crate for 'runner'` / `'agent_tools'` / `'camp_service'` and a linker failing on a dozen absent `.rlib`s (libgif, libzune_jpeg, libimagesize...) in crates this ticket never touched — the exact shape CLAUDE.md's orphan-gc warning describes. Followed that procedure rather than cleaning: `cargo orphan-gc log -n 300` names NONE of the missing artifacts (every entry in the window reads `deleted 0 artifacts`), so orphan-gc is NOT confirmed here. The likelier cause is plain target-dir contention: a `yah-release-check` QED pipeline was holding the same `/Users/leif/ss/yah/target` for 29 minutes alongside this build. Re-ran with nothing else on the key: CLEAN, zero errors. Not reproducible, orphan-gc log does not name it, and the artifacts were never deleted per its own record.")
186//! @yah:next("WHAT REMAINS IS THE LIVE HALF ONLY, and it is an operator call the R870 leader is holding — phase 2 deliberately landed code + tests and touched no node. The two assertions: (a) POSITIVE — a two-component service whose components deploy independently gets one inner door from one `yah cloud apply`, with https://<host>/ and https://<host>/app/ both 200 and `curl -sI` on /app/ carrying cross-origin-opener-policy: same-origin AND cross-origin-embedder-policy: require-corp while / carries neither. (b) NEGATIVE, node-side — for a single-component service (yah-marketing), NO routes file materialized under /var/lib/passway/routes and NO extra workload in kamaji's table. Note that (b) is now also asserted statically against the real tree by `xtask/tests/mirror_ingress.rs::no_service_on_disk_gets_an_inner_door`, so the node-side check is confirmation rather than discovery. BEFORE RUNNING (a): there is no service with `deploy = \"workload\"` on disk, so one has to be declared first — and the R870-B6/V8 roll-order gotcha on this ticket applies the moment anything actually SETS `WorkloadSpec::files`. Confirm the target node's kamaji link codec (kamaji-proto/src/codec.rs) and roll the kamaji+yubaba PAIR first if it is postcard.")
187//! @yah:next("ONE THING PHASE 2 DID NOT BUILD, named so it is not mistaken for done: there is still no FLEET deploy path for a `DeployTier::Workload` component. `reconcile_component` (app/yah/cli/src/cloud.rs:8264) dispatches on `component.kind`, and the only non-bundle arms are `container` — which `ContainerReconciler::up` guards on `MirrorShape::Local`, and `LocalProcessReconciler`, which is the camp/dev tier and registers as `local-process-<service>-<env>-<component>`. So on a real mirror such a component is deployed by hand today (`yah cloud workload deploy`). That is exactly why `inner_door::component_workload_ident` had to STATE the ident rather than look it up. The failure mode is loud rather than silent — a component registered under any other ident leaves its unit unresolved and `routes_file` refuses the whole table, naming the unit — but whoever wires that deploy path must make it register under `component_workload_ident(service, id)`, or change both sides together. Related and already filed: R523-F1 (a component kind that deploys a stateful binary to a fleet node) is the same missing arm seen from the other direction.")
188//! @yah:handoff("PHASE 2 COMPLETE — `yah cloud apply` produces an inner door. Everything above this entry is the detail: the five call sites, the derived-port argument, the header-ownership answer (no change — the outer passway door owns no route headers, proven at proxy.rs:694/700, and the bundle origin's overlap is convergent and strictly wider), the three pieces of plumbing built because steps 3 and 5 needed them, the two placement decisions, and the one mirror.yml provisioning defect fixed on the way through. Nothing was deployed and no node was touched, per the dispatch. Green: yah-cloud 1163/0, yah 1549/0, yubaba 952/0, xtask mirror_ingress 13/0, passway 285/0, kamaji 217/0, both --all-features --no-run sweeps clean. Git policy is `defer`, so nothing is committed — the diff is 6 files: oss/yubaba/crates/cloud/src/{inner_door.rs, reconciler/mod.rs, reconciler/ingress.rs, reconciler/service_discovery.rs}, oss/yubaba/crates/cloud/templates/mirror.yml, app/yah/cli/src/cloud.rs, plus xtask/tests/mirror_ingress.rs.")
189//! @yah:handoff("Tree anchor at handoff: 88533e01f7f578b1520b633d05846973fa47f608 — the shared tree as I left it. Diff against it (`git diff 88533e01f7f578b1520b633d05846973fa47f608..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
190//! @yah:handoff("PHASE 2 ACCEPTED BY THE RELAY LEADER (@Ashguard:hydra, session:39386823). `yah cloud apply` now produces an inner door: all five steps wired, plus three pieces of plumbing that did not exist (`component_workload_ident`, `ServiceRecordFanout::address_for_ident`, `IngressPlan::point_at_inner_door`), the derived-port decision argued from three read facts, and the open header question answered NO CHANGE with the proof at proxy.rs:694/700. Implemented by @Ashguard:blade (session:54a6be05). The detail is in the handoff entries above this one; this entry records only that it was accepted and on what evidence.")
191//! @yah:verify("WHAT IS DELIBERATELY NOT VERIFIED, and it is the operator's call rather than an oversight: the LIVE half. No node was touched, nothing was deployed, nothing committed (git policy is `defer`). Running it needs a service with `deploy = \"workload\"` declared — none exists on disk — and the moment anything actually SETS `WorkloadSpec::files`, this ticket's own V8 roll-order gotcha binds: confirm the target node's kamaji link codec (kamaji-proto/src/codec.rs) and roll the kamaji+yubaba PAIR, never one alone.")
192//! @yah:verify("INDEPENDENTLY RE-RUN BY A SECOND COURIER (@Ashguard:dove, session:60d4f41f) who did not implement it, because a courier's self-report is the inner gate and not the outer one. All six commands reproduced the claimed counts EXACTLY: yah-cloud 1163/0 (4 ignored), yubaba 952/0, passway 285/0 (205+43+37), kamaji --lib --all-features 217/0, yah --lib 1549/0 (1 ignored), workspace --all-features --no-run clean. Every content check held: `point_at_inner_door` at ingress.rs:438; `inner_door::plan` reached from the apply path via `service_inner_door` (cloud.rs:9219) through `deploy_inner_door` (cloud.rs:9274, invoked at 7100 and 11877) and the ingress repoint at 7726; the port confirmed a deterministic per-service pin (FNV-1a into 10000-19999, inner_door.rs:346) and NOT kamaji's ledger, with the native.rs:282 `pin.is_none()` justification verified at the site. The negative is asserted three times, not once. CAVEAT ON THE MEASUREMENT ITSELF: the camp skew detector flagged 4 of 6 runs SUSPECT — peers edited kamaji/src/microvm.rs, kamaji-bin/src/main.rs and cloud/reconciler/mesofact_bundle.rs mid-run — so these are shared-tree numbers, not a frozen-tree measurement.")
193//! @yah:verify("ONE CLAIM CORRECTED AND ONE DEFECT FOUND BY THAT RE-RUN, both recorded rather than smoothed over. (1) CORRECTION: `cargo test -p yah-cloud --lib` does NOT run from the repo root — yah-cloud is not a root workspace member and needs dev-dependencies; it only works from `oss/yubaba`. Anyone reproducing the 1163/0 above must cd there first. (2) DEFECT, pre-existing and NOT caused by this ticket: `embedded_template_matches_workspace_canonical` is green VACUOUSLY — it resolves the workspace root via CARGO_MANIFEST_DIR.ancestors() to oss/yubaba, whose .yah/ holds only a .gitignore, so it takes the bootstrap branch and asserts nothing, while the repo-root twin at .yah/infra/cloud-init/mirror.yml is 128 diff-lines stale and missing the whole R858-F17 turso-backup block. FILED AS R870-B25, not left here. Note `rendered_runcmd_entries_are_all_strings` is a DIFFERENT test, is genuinely green, and is the gate that really catches the colon-space footgun this ticket fixed.")
194//! @yah:gotcha("SUPERSEDED BY R870-F27, and the @yah:next that said otherwise has been removed from this block: containerd DOES materialize WorkloadSpec::files now, so an inner door is no longer pinned to native-capable nodes. The backend-check step this ticket told its reader to perform before deploying is gone. Still refusing: Docker and MicroVm. The single fact both halves read is kamaji::Backend::materializes_files (oss/kamaji/crates/kamaji/src/lib.rs), not a list in prose.")
195//!
196//! @yah:ticket(R885-T14, "No cap:bundle-serving mesh capability exists — bundle/almanac/passway workloads are placed with nothing modelling where they can run")
197//! @yah:status(review)
198//! @yah:at(2026-09-12T07:16:50Z)
199//! @yah:assignee(agent:bundle-anthropic-ashguard)
200//! @yah:parent(R885)
201//! @yah:severity(P3)
202//! @yah:next("FOUND WHILE DISPROVING R885-T13, and it is the real gap that ticket's false premise was standing next to. `cap:native-exec` exists and is honoured (config.rs:2353-2357 via wants_native_exec) for `yah.exec = native` Container specs. But the workloads actually running on the fleet — MesofactServeBundle / Almanac / TenantPassway, served by kamaji's BundleBackend and JitRuntime — have NO corresponding mesh capability at all. All four live workloads on us-east-001 are of those kinds. So placement for the entire class of workload this fleet actually runs is unmodelled: nothing declares which nodes can serve bundles, and nothing checks. It works today because there is effectively one node doing it, which is exactly the condition under which an unmodelled constraint stays invisible. THE SHAPE OF THE FIX IS ALREADY IN THE TREE: R860-T5 closed this same gap for native-exec. Follow it — a `cap:bundle-serving` capability declared in .yah/infra/machines/*.toml, a `wants_bundle_serving` predicate beside `wants_native_exec`, and the placement check wired the same way. Establish the right granularity first: whether bundle / almanac / tenant-passway want one shared capability or separate ones is a real design question, and the answer depends on whether a node can serve one kind and not another. Tier: Cleric. ITS NATURAL HOME IS R860's AXIS, NOT R885's — R885 is about workloads running unbounded once placed, this is about where they are placed at all. Filed under R885 because that is where it was found and where the evidence is; move it to R860 if that relay is still live. Not urgent: nothing is broken today, and the cost is that the first multi-node bundle placement decision will be made by something that has no model of the constraint.")
203//! @yah:handoff("GRANULARITY SETTLED BY OPENING KAMAJI, NOT BY GUESSING — and the answer is smaller than the ticket assumed. kamaji gates each backend on its own cargo feature + startup flag: BundleBackend on `bundle-serving` + `--bundle-cache-dir` + `--bundle-origin` (`attach_bundle_backend`, oss/kamaji/crates/kamaji-bin/src/main.rs:924), JitRuntime on `tenant-passway` + `--tenant-passway-dir` (:706). Independent flags means separate capabilities, not one shared tag. But only ONE of the three named in the ticket needs modelling today: (1) BUNDLE gets `cap:bundle-serving`. (2) TENANT-PASSWAY gets nothing — there is no placement decision to gate: `yubaba::tenant_passway::reconcile_once` runs inside a node's own yubaba and drives that node's own kamaji over the local UDS, so which node arms a domain is decided by `YUBABA_TENANT_PASSWAY_STATE_DIR`, not by a selector. A tag would have no reader, which is the same wrong-fact R860-T5 refused to state for `cap:microvm`. (3) ALMANAC gets nothing ever — kamaji refuses `Workload::Almanac` outright (\"almanac and static-asset live in yubaba's reconcilers\", kamaji-bin/src/server.rs:1846) and yubaba reconciles it, so it is never node-placed.")
204//! @yah:verify("MEASURED, every number an exit-visible run. `cargo test -p yah-cloud --lib` (oss/yubaba, CARGO_TARGET_DIR=/tmp/r885t14-target) = 1205 passed / 0 failed / 4 ignored. The baseline is 1200 and it is recoverable from this session's own output rather than asserted: the first post-edit run read 1195 passed / 5 FAILED, those five being pre-existing fixtures the new axis correctly broke (synthetic machines with no `mesh_tags`), so 1200 before, +5 new tests, 1205 after. `cargo test -p xtask --test main` = 69 passed / 0 failed (the real-tree suite, incl. all 11 mirror_ingress, all 4 apex_failover, all 4 fleet_build_placement). `cargo test -p xtask --doc` green. `cargo check -p yubaba --all-targets` (oss/yubaba) clean. `cargo check -p yah --all-targets` (root) clean — flagged SUSPECT by the camp build rail (peers edited app/yah/cli/src/camp.rs, mesh.rs, oss/yah-base/crates/keys/src/spec.rs, oss/yubaba/crates/yubaba/src/lib.rs mid-run); none of those is a file this ticket touched and the CLI's only contact with the change is `resolve_bundle_machines`, whose signature is unchanged.")
205//! @yah:handoff("WHAT LANDED. (1) `pub const BUNDLE_SERVING_MESH_TAG: &str = \"cap:bundle-serving\"` in oss/yubaba/crates/cloud/src/config.rs beside NATIVE_EXEC_MESH_TAG, carrying the node-side gate, why a positive capability and not a taint, and why there is no `cap:tenant-passway` beside it. (2) oss/yubaba/crates/cloud/src/reconciler/mesofact_bundle.rs: `with_bundle_capability(&RequiredSpec) -> RequiredSpec` (idempotent) and `ensure_bundle_capable(&MachineConfig, ..)`, wired into BOTH arms of `resolve_bundle_machines` — the constraint arm gets the tag appended to the derived mesh_tags, the literal `machines = [...]` pin is refused with an error naming the tag AND the .yah/infra/machines/<name>.toml to edit. (3) oss/yubaba/crates/cloud/src/reconciler/ingress.rs: `required_for_role(role, required)` applied in both `resolve_ingress_placements` and `resolve_ingress_candidates`, so the deployer and the discovery fanout stay set-for-set across the new axis — the property resolve_bundle_machines' own doc promises and which a bundle-only check would have broken. (4) .yah/infra/machines/us-east-001.toml declares the tag. (5) W338 §Placement consequences gains item 5.")
206//! @yah:handoff("THE ARCHITECTURAL POINT, because it is why this was not one line beside the native tag. `cap:native-exec` is read in `admission_spec`, which covers `Workload::Container` and NOTHING ELSE. A bundle never reaches that function — it is placed by `resolve_bundle_machines` off the mirror's `providers.bundle` declaration, a completely separate resolver that an operator writes by hand. So the gap was not \"one more tag in the same `if`\"; it was the same class of gap one resolver over. The rule the two instances share, now written into W338: a capability tag belongs wherever a SELECTOR chooses a node, not wherever a spec is admitted. Anything that picks a machine has to know what that machine's kamaji was started with, because every kamaji backend is an opt-in flag.")
207//! @yah:handoff("THREE EXTRA FIXES, all outside the ticket title, all loud. (a) xtask/src/install.rs:717 — `clear_stale_provenance`'s doc comment had an INDENTED log excerpt, which rustdoc compiles as Rust, so `cargo test -p xtask` failed its doctest on main for everyone. Fenced as ```text. Pre-existing, unrelated to this ticket, one line. (b) .yah/schema/mirror.toml.schema.json regenerated (`cargo run -p xtask -- emit-schemas`) — the schema-drift gate was RED on arrival and the drift is @Ashguard:coffee's R584-F2 `local-mailcrab` doc-comment change on `MirrorConfig::drivers`, not mine (verified by reading the one-line diff before regenerating). Regenerated per the generated-artifacts-are-not-ownable rule; they were told. Zero of my own types are schemars-derived, so this change contributes nothing to any schema. (c) Corrected three stale claims at their sites rather than leaving them: config.rs's NATIVE_EXEC_MESH_TAG doc said `Workload::MesofactServeBundle` and `Workload::Almanac` are BundleBackend-served (MesofactServeBundle is a FIELD on Workload::MesofactStatic, not a variant; Almanac is never node-placed at all), and us-east-001.toml's R885-T13 block repeated the same three wrong names for the four processes it observes — all four are bundle workloads.")
208//! @yah:verify("NEW TESTS. Five in mesofact_bundle's `mod tests`, each keeping a capable AND an incapable node in one CloudConfig so a pass provably comes from the capability rather than an empty pool: a_node_without_the_bundle_backend_cannot_be_pinned_to_serve_a_bundle (refusal names the tag and the file; the SAME fleet + same declaration shape at a capable node still resolves); a_constraint_places_past_a_matching_node_that_cannot_serve_bundles (the incapable node is declared FIRST, so a resolver ignoring the axis returns it and the test fails; then dropping the capable node makes the same declaration refuse); the_ingress_planner_applies_the_same_capability_as_the_deployer (set-for-set, plus the candidate widener must not widen past the capability); a_mirror_that_declares_the_capability_itself_is_unchanged (idempotence); a_non_bundle_slot_requires_no_capability (regression guard on the role mapping — a static slot still places on a node with no bundle backend). The shared `machine()` fixture now carries the tag by default, with `machine_without_bundle_backend()` beside it, because every placement test there presupposes an eligible pool.")
209//! @yah:verify("REAL-TREE HALF, which is where the live-safety answer is. New xtask/tests/mirror_ingress.rs::exactly_one_machine_in_the_real_fleet_can_serve_a_bundle asserts the capable set over .yah/infra/machines/ is exactly [\"us-east-001\"] and that a `replicas = 2` bundle constraint therefore refuses NAMING the tag. It is a deliberate tripwire: the day a second node declares the capability, the \"one node serves every bundle\" reasoning scattered through yah-marketing's and noisetable's mirrors stops holding and this is what says so. The pre-existing a_constraint_with_replicas_two_… had to change — its scale-2 assertions need two capable nodes and the real fleet has one — so it now grants the capability to us-south-001/us-west-001 IN MEMORY, asserts first that neither declares it on disk (so the grant cannot go silently vacuous), and says at the site that this is a hypothetical fleet, not a claim about those boxes. Every one of its original assertions (repel-by-default moving the second slot, the toleration lever recovering the pre-B7 answer, absent-replicas being exactly one, the short-count refusal, the two-backend render through collate) is unchanged and green.")
210//! @yah:gotcha("THIS FAILS CLOSED AND THE BLAST RADIUS WAS CHECKED, NOT ASSUMED. A bundle can now only be placed on a node declaring `cap:bundle-serving`, and exactly one does. Every live bundle placement in reach still resolves: yah-marketing's `required = { regions = [\"us-east\"], mesh_tags = [\"tag:cloud-runner\"] }` lands on us-east-001 (green in the real-tree suite), and the noisetable camp's two `providers.bundle.machines = [\"us-east-001\"]` pins are covered because ~/ss/noisetable/.yah/infra/machines/ is an EMPTY DIRECTORY — that camp borrows this repo's inventory read-only through its .yah/infra/sources.toml [[source]] link, so declaring the tag here fixes it there too and no cross-camp edit was needed. NOT EXECUTED, and say so rather than implying it: I did not run `yah cloud validate`/`ingress collate` from ~/ss/noisetable against a rebuilt binary. That conclusion is read off the two mirrors' pins plus the sources.toml link, not observed.")
211//! @yah:assumes("us-east-001's `cap:bundle-serving` rests on a BEHAVIOURAL reading, not on that box's ExecStart line: it is the node every bundle in the fleet is placed on today, R870-T9 records noisetable.com serving live through a bundle from it, and R885-T13 observed four kamaji-forked bundle processes there on 2026-09-11. Nothing in-repo records `--bundle-cache-dir` on its kamaji command line. The tag is therefore right about what the box demonstrably does and unverified about how it was started; if a roll ever drops the flag the tag becomes a lie in the fail-OPEN direction (a deploy that is admitted and then refused — i.e. exactly today's behaviour, not worse). Settle it with `ps -o args= -C kamaji` / the kamaji.service ExecStart, the same evidence us-west-001.toml:61-67 records for the native tag.")
212//! @yah:cleanup("us-south-001 is the obvious second bundle-capable candidate — it already shares the passway demux pair with us-east-001 — and is deliberately left undeclared because nothing in-repo reads its kamaji flags either way. One `ps -o args= -C kamaji` on that box settles it; declaring it wrongly would route a bundle to a node that refuses it, which is the failure this axis exists to remove. Until then the fleet is genuinely single-node for bundle serving and exactly_one_machine_in_the_real_fleet_can_serve_a_bundle says so out loud.")
213//!
214//! @yah:relay(R926, "Register iroh relay + headscale as first-class camp services, with descriptions and HA-aware health checks")
215//! @yah:status(review)
216//! @yah:at(2026-09-20T22:13:30Z)
217//! @yah:assignee(agent:bundle-anthropic-ashguard)
218//! @arch:see(.yah/docs/working/W122-yah-mobile.md)
219//! @yah:handoff("FILED 2026-09-17 from an operator request made during R726-S22's device run (@Ashguard:dove, session:639e7b78). THE ASK, in the operator's words: the iroh relay and headscale should be listed in the yah services with DESCRIPTIONS and HEALTH CHECKS, \"since they can move around in HA mode\". GROUNDED STATE OF THE WORLD AT FILING, read rather than assumed: (1) The service registry is .yah/services/<name>/service.toml. Exactly nine services are registered - scrabcake, yah-analytics, yah-chat, yah-cloud, yah-cloud-admin, yah-cr, yah-dashboard, yah-desktop, yah-marketing. NEITHER an iroh relay NOR headscale is among them, so the operator's observation is correct. (2) The generated schema .yah/schema/service.toml.schema.json has top-level properties [components, db, domain, health_path, name, schema_version], required [domain, name, schema_version]. So a health HOOK already exists in the shape of `health_path` - this is NOT greenfield - but there is NO `description` field at all, which is new schema surface. (3) .yah/infra/machines/*.toml are pure INVENTORY (name, [allocatable], [connect], [registration]); they are not where a service is declared, so do not add it there. (4) The Rust side of health_path lives in oss/yubaba/crates/cloud/src/config.rs and src/lib.rs (also mirrored in oss/mesofact/crates/mesofact/src/lib.rs).")
220//! @yah:next("Wire the iroh relay explicitly rather than implicitly. R726-S22 measured a phone falling back to relay because there was no shared address family with the camp; the relay is the DESIGNED path in that case, not a failure - but today nothing in the service registry says the relay exists, so its health is unobservable from the camp UI.")
221//! @yah:next("Add a `description` field to the service schema (new surface - it does not exist), then REGENERATE the artifacts: `cargo run -p xtask -- emit-schemas` and the workload-spec export. They no longer regenerate on commit and schema-drift-guard in the `check` QED pipeline will fail the build otherwise.")
222//! @yah:next("HA is the actual hard part and deserves a design decision before code: `health_path` is a single path on a single declared domain, which cannot express \"this service currently lives on whichever of N machines won the election\". Decide whether a service gains a set of candidate endpoints with a liveness winner, or whether the registry queries the mesh's own service-records endpoint (the 100.64.0.3:7443/service-records?ready=true surface referenced in us-west-001.toml) as the source of truth. Do NOT bolt a second health mechanism beside health_path - see the repo's below-v1.0.0 rule; change the one that exists.")
223//! @yah:gotcha("HEADSCALE HAS A LOUD, DOCUMENTED FAILURE HISTORY AND THIS TICKET IS PARTLY A RESPONSE TO IT - read R858 before designing the health check. .yah/infra/machines/us-west-001.toml carries R858 (\"Mesh coordination outage: cloud.mesh.yah.dev refuses :443, so no camp machine can reach any 100.64.0.0/10 address\") plus a measured 2026-09-04 gotcha: tailscale reported \"fetch control key ... connect: connection refused\", port 22 answered while 80/443 were REFUSED, and consequently every mesh address stopped answering. THE PART THAT MATTERS FOR A HEALTH CHECK: the same gotcha records that THE PUBLIC SITE STAYED GREEN THROUGHOUT (yah.dev HTTP 200 in 0.81s) because the apex serves from us-east-001's own passway and never traverses the coordination server. So a naive HTTP health check against a public domain would have reported HEALTHY during a total mesh outage. Whatever check this ticket adds for headscale MUST probe the coordination path itself (the control-key fetch, or a mesh-address dial), not a public endpoint that is up for unrelated reasons. The repo's CLAUDE.md also warns that the headscale appliance already accumulated four half-owners of \"does this node have a config.yaml\" and that the seam between two of them took the mesh down twice - so give this ONE owner.")
224//! @yah:handoff("PHASE 1 LANDED — the `description` field exists and the artifacts are regenerated. oss/yubaba/crates/cloud/src/config.rs: ServiceConfig gains `description: Option<String>` (serde default + skip_serializing_if, so the nine pre-existing services still load). 28 struct-literal construction sites updated across config.rs(13), inner_door.rs, pond.rs(5), cloudflare_worker.rs, derive_cache_prune.rs, local_process.rs, mesofact_static.rs, static_asset.rs, static_asset_prune.rs, sync_status.rs, reconciler/mod.rs, tests/whisper_derive_e2e.rs. Two new tests: a_declared_description_survives_the_loader and a_service_without_a_description_still_loads. `cargo run -p xtask -- emit-schemas` re-run; .yah/schema/service.toml.schema.json now carries the field and scripts/check-schema-drift.sh exits 0 (\"ok: .yah/schema is in sync with the Rust types\"). packages/yah/workload-spec/index.ts was NOT regenerated because it did not change — this edit touches no workload-spec source.")
225//! @yah:verify("cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib = 1255 passed / 0 failed / 4 ignored, EXIT=0 (/tmp/r926_full_e2.log). scripts/check-schema-drift.sh = exit 0.")
226//! @yah:gotcha("THE IROH-RELAY HALF OF THIS TICKET RESTS ON A THING THAT DOES NOT EXIST: yah OPERATES NO IROH RELAY. Measured 2026-09-20. (a) mshr's default is n0's PUBLIC relay — oss/mshr/crates/mshr/src/discovery.rs:82 default_relays() returns vec![N0_DNS_PKARR_RELAY_PROD], imported from iroh at discovery.rs:32. (b) The capability to self-host is BUILT AND UNUSED: `mshr::relay::Server` exists with a full ACME/TLS builder (oss/mshr/crates/mshr/src/relay.rs:258-398) and its own round-trip + pebble tests, and grep for `relay::Server|RelayServer|relay_server` across oss/ crates/ app/ returns 13 hits that are ALL the library itself or its own tests — zero production callers, and none at all in crates/yah/, app/yah/ or oss/yubaba/. (c) .yah/infra/workloads/ contains exactly ONE file, yah-cloud-admin.toml. (d) grep for a relay URL across .yah/ returns nothing. So R726-S22's phone \"falling back to relay\" fell back to n0's public infrastructure. Registering that in .yah/services/ — the registry `yah cloud apply` RECONCILES — would declare a domain yah does not own and components it does not deploy. That is a scope question, not a defensible default, which is why it is an operator call below.")
227//! @yah:gotcha("THE HA OPTION THIS TICKET'S OWN next() FAVOURED IS STRUCTURALLY BLOCKED, and the blocker is already documented in-tree. The filing said to consider \"querying the mesh's own service-records endpoint (100.64.0.3:7443/service-records?ready=true) as the source of truth\". SERVICE RECORDS ARE STRICTLY PER-NODE: app/yah/cli/src/mesh.rs:94 records the invariant from service_records.rs's module doc — \"A record's `mesh_ip` must equal the answering node's own mesh address\" (R844-B11) — and `reconcile` rebuilds the set from that node's OWN ContainerRuntime::list_workloads(). THERE IS NO CLUSTER-WIDE SERVICE VIEW. So asking any one node where headscale lives answers only \"on me\" or \"not on me\". Confirmed live in the record: on 2026-09-08 the coordinator MOVED from us-west-001 to us-south-001 (.yah/infra/machines/us-west-001.toml:36), which is exactly the event a health check must survive and exactly the one a per-node query cannot see. Corollary: the R858 gotcha's warning still governs — a check against a public domain reports HEALTHY during a total mesh outage, because yah.dev serves from us-east-001's own passway and never traverses the coordination server.")
228//! @yah:gotcha("TWO SMALL TRAPS FOR THE NEXT EDITOR OF THIS CRATE, both cost me a build. (1) `cargo check -p yah-cloud` FROM THE REPO ROOT reports `error[E0433]: cannot find module or crate serde_yaml` — this is NOT a missing dependency and NOT a vanished artifact. serde_yaml IS declared at oss/yubaba/crates/cloud/Cargo.toml:143 under [dev-dependencies]. oss/yubaba is an independent workspace excluded from the yah root (see CLAUDE.md \"Co-developed OSS repos\"), so from the root the crate is a non-member path dep via [patch.crates-io] and cargo applies no dev-dependencies to it; cargo says so plainly if you use `test` instead of `check` (\"package yah-cloud cannot be tested because it requires dev-dependencies and is not a member of the workspace\"). Correct invocation: `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib`. Also note the package is `yah-cloud`, not `cloud` — `cloud` is the [lib] name only (Cargo.toml:6 vs :31). (2) A LITERAL `@yah:` INSIDE A RUST DOC COMMENT IS STRIPPED OUT OF THE GENERATED JSON SCHEMA, from the marker to the end of that paragraph. My first draft of ServiceConfig::description's doc said \"...whichever W### doc or `@yah:` annotation last touched the thing\", and emit-schemas produced a description truncated mid-sentence at the backtick. Spell it \"board annotation\" in prose on any type that feeds .yah/schema/.")
229//! @yah:handoff("PHASE 2 LANDED — headscale is registered, described, and probed on the coordination path. NEW FILE .yah/services/headscale/service.toml: name=\"headscale\" (matching the service-record ident leader.rs already upserts), domain=\"cloud.mesh.yah.dev\", description set, health_path=\"/key?v=138\". NO components and NO mirrors/ — deliberately: the appliance is placed by the mesh leader (app/yah/cli/src/mesh.rs start_headscale), not by `yah cloud apply`, so this file makes the service observable without claiming its deployment. Verified that shape cannot break the camp's config load by READING the loader: load_services gates mirrors on `if mirrors_dir.exists()` (config.rs:2909) so a missing dir yields an empty map, and both cross_ref_validate bail-loops iterate `&svc.mirrors` and `&svc.service.components` — empty, so zero iterations. The only service-level check is health_path.starts_with('/'), which the query string satisfies. Pinned with a new test, an_observability_only_service_loads_without_components_or_mirrors, so it stays true.")
230//! @yah:handoff("THE HA QUESTION IS ANSWERED FOR HEADSCALE, AND THE ANSWER IS \"health_path IS ALREADY THE RIGHT SHAPE\" — no second mechanism, per the below-v1.0.0 rule. The filing assumed one domain cannot express \"lives on whichever of N machines won the election\". For this service it can, because the thing that moves and the thing the registry names are different things: the appliance really did migrate us-west-001 -> us-south-001 on 2026-09-08 (R858 handoff), but cloud.mesh.yah.dev terminates at a passway front door on ALL THREE doors (R858-T1, mesh.rs:141), so placement is already abstracted behind one stable address and failover is the doors' job. THE PROBE PATH IS THE PART THAT HAD TO BE RIGHT, and all three measurements were taken 2026-09-20: cloud.mesh.yah.dev/key?v=138 -> 200 in 0.23s; cloud.mesh.yah.dev/ -> 404; yah.dev/ -> 200 in 0.38s. So BOTH obvious choices are wrong and each fails in the direction that hurts — the schema default of \"/\" reports headscale BROKEN while it is healthy, and any public yah domain reports it HEALTHY during a total mesh outage (the R858 false-green, still live: measurement three). /key?v=138 is the control-key fetch, the first call every tailscaled makes and the exact request whose failure opened R858, so nothing but the coordination path can serve it.")
231//! @yah:next("The iroh relay is UNBUILT pending an operator call — see the gotcha: yah operates no relay, so there is nothing to register until someone decides between registering a dependency on n0's public relay and standing up mshr::relay::Server as a real yah service.")
232//! @yah:next("SEPARABLE FOLLOWUP, deliberately not done here: the nine pre-existing services (scrabcake, yah-analytics, yah-chat, yah-cloud, yah-cloud-admin, yah-cr, yah-dashboard, yah-desktop, yah-marketing) still carry no `description`. The field is optional so they all load, but a registry where only headscale is described is half a feature. NOT attempted in this pass because writing nine descriptions for services I had not read would be fabrication — each needs its own service.toml and mirrors read first. Worth one ticket that does all nine at once.")
233//! @yah:verify("cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib = 1256 passed / 0 failed / 4 ignored, EXIT=0 (/tmp/r926_full_h5.log), including all three new tests. Live probes 2026-09-20: cloud.mesh.yah.dev/key?v=138 = 200 (0.23s), cloud.mesh.yah.dev/ = 404, yah.dev/ = 200 (0.38s).")
234//! @yah:gotcha("CORRECTION TO THIS TICKET'S OWN RELAY GOTCHA, prompted by the operator (\"I thought we did serve a relay?\") — they were right and my framing was wrong. THERE ARE TWO DIFFERENT RELAYS and the filing conflated them. (1) The iroh/NAT-TRAVERSAL relay (`--xlb-relay` / $YAH_XLB_RELAY): we run none. oss/yubaba/crates/yubaba/src/main.rs:627 says plainly \"unset ships n0's production relays\", and mshr::relay::Server still has zero production callers. That part of the earlier gotcha stands. (2) THE PUSH RELAY: we absolutely do run one, it is ours, and it is a far better fit for this ticket than the iroh relay ever was. crates/yah/push-relay/ is a real crate with its own daemon (src/bin/yah-push-relay.rs), serving ALPN \"yah/push-relay/1\" (protocol.rs:32); yubaba takes --push-relay-node-id whose doc says \"this is normally the relay co-located on this machine\" (main.rs:544-548). It holds the FCM service-account credential and is what makes a phone buzz when a gate is raised (W122 §Push, R726-F7).")
235//! @yah:gotcha("THE PUSH RELAY IS THE CASE health_path GENUINELY CANNOT EXPRESS — and unlike headscale, here the registry really is the wrong shape. It has NO HTTP SURFACE AT ALL: src/bin/yah-push-relay.rs binds an iroh endpoint and serves the ALPN, and grep for health/healthz/axum/http across server.rs + the bin returns nothing but the `.bind()` call at :72. It is reached by hex NodeId over QUIC, so it has no domain and no path — while ServiceConfig REQUIRES `domain` and health_path is defined as \"relative to domain\". Registering it today would mean inventing a `.invalid` domain the way yah-cloud already had to (see .yah/services/yah-cloud/service.toml's note on the R546-S2 domain-requirement tension), and then having no way to probe it. THIS is the real \"a second health mechanism vs change the one that exists\" decision the filing was reaching for — it just belongs to the push relay, not to headscale. The honest shape is probably a dial-the-ALPN liveness check keyed on NodeId, which means `domain` stops being the only way to address a service. That is design work plus an operator call, not a config edit.")
236//! @yah:handoff("SCOPE DELIVERED, AND ONE HALF DELIBERATELY SPLIT OUT. Done: the `description` schema field (new surface, 28 call sites, 3 new tests, artifacts regenerated, drift guard green) and headscale registered at .yah/services/headscale/service.toml with a description and a health check that probes the coordination path itself. NOT done, and split to R926-F1: registering a relay. The filing asked for \"the iroh relay\", which yah does not run — but the operator corrected me that we DO serve a relay, and they are right: it is the PUSH relay (crates/yah/push-relay/), which is ours and is the thing that should be registered. It cannot be registered today because it has no HTTP surface and ServiceConfig requires a `domain` — the genuine schema-shape problem this ticket was reaching for, plus a live operator call on one-shared-relay vs one-per-camp. See the gotchas for both.")
237//! @yah:verify("cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib = 1256 passed / 0 failed / 4 ignored, EXIT=0 (/tmp/r926_full_h5.log). scripts/check-schema-drift.sh = exit 0, \"ok: .yah/schema is in sync with the Rust types\". Live probes 2026-09-20 proving the health check discriminates: cloud.mesh.yah.dev/key?v=138 = 200 (0.23s), cloud.mesh.yah.dev/ = 404 (so the schema default of \"/\" would have reported a false RED), yah.dev/ = 200 (the R858 false-GREEN, still live).")
238
239use anyhow::{bail, Context, Result};
240// R870-F26: an `[[ingress]]` edge embeds the renderer's own auth type rather
241// than a config-side copy of its five fields — see `IngressEdge::auth`.
242use local_driver::passway_ingress::{AuthSpelling, PasswayAuth};
243use serde::{Deserialize, Serialize};
244use std::collections::BTreeMap;
245use std::path::Path;
246use thiserror::Error;
247use workload_spec::secrets::SecretAccess;
248use workload_spec::sovereign::Membership;
249pub use workload_spec::sovereign::SovereignRole;
250use workload_spec::{validate, LifecycleArchetype, Locality, WorkloadSpec};
251
252/// Static node capacity declaration on `machine.toml` (R572-F3).
253///
254/// `memory_mb` and `cpu_millis` express the node's *total* hardware budget.
255/// F5's bin-packer subtracts the sum of committed workload requests from
256/// this floor to determine available headroom; an absent `allocatable`
257/// block means no capacity constraint is enforced (any workload fits).
258#[derive(Debug, Clone, Serialize, Deserialize)]
259#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
260pub struct NodeAllocatable {
261 /// Total physical RAM in mebibytes (e.g. 512 for a 512 MB node).
262 pub memory_mb: u32,
263 /// Total CPU in k8s millicores (1000 = 1 core, 250 = 0.25 CPU).
264 pub cpu_millis: u32,
265}
266
267/// `[registration]` — facts **observed** about a running box, written by the
268/// fleet rather than declared by an operator (R707-T1).
269///
270/// The rest of `machine.toml` is *declaration*: intent, operator-authored,
271/// reviewed and diffed like any other source. This block is the other half —
272/// what the box turned out to be once it booted and joined. Keeping the two
273/// apart is what lets the published fleet index (R707-F3) say which half it is
274/// carrying; publishing them under one schema would bake the confusion into a
275/// permanent record.
276///
277/// The split is a **provenance** boundary, not a trust or reach one:
278/// - *Declaration* answers "what did we ask for" — `name`, `region`, `arch`,
279/// `mesh_tags`, `[allocatable]`, and the declared reach in [`ConnectSpec`].
280/// - *Registration* answers "what did we observe" — the hostkey TOFU'd at
281/// attach, the mesh address headscale assigned at join.
282///
283/// It stays in the git-tracked TOML on purpose. Registration is not local
284/// scratch state: every consumer needs the mesh address to dial a node, so it
285/// has to travel with the declaration. (`.yah/infra/state/machines/<name>.json`
286/// — [`crate::state::MachineState`] — remains the *gitignored* sidecar for
287/// provider-side derivatives that nobody but this camp needs.)
288#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq, Eq)]
289#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
290pub struct MachineRegistration {
291 /// Yubaba's ed25519 `/identity` fingerprint, TOFU-recorded by
292 /// `yah cloud machine attach` on first contact (`SHA256:…`). An observed
293 /// property of a running process — not the operator's intent — which is
294 /// why it moved out of the top level here.
295 #[serde(default, skip_serializing_if = "Option::is_none")]
296 pub hostkey_fingerprint: Option<String>,
297 /// Mesh (headscale/tailnet) IPv4 assigned at join, e.g. `"100.64.0.1"`.
298 /// Bare address, not a URL: the *port* is declared reach and lives on
299 /// [`ConnectSpec::yubaba_port`]. [`MachineConfig::yubaba_url`] composes the
300 /// two. Absent until the node has joined the mesh.
301 #[serde(default, skip_serializing_if = "Option::is_none")]
302 pub mesh_ipv4: Option<String>,
303 /// RFC3339 timestamp of the mesh join that produced `mesh_ipv4`. Free-form
304 /// audit; nothing keys off it.
305 #[serde(default, skip_serializing_if = "Option::is_none")]
306 pub joined_at: Option<String>,
307}
308
309impl MachineRegistration {
310 /// True when nothing has been observed yet — used to omit the whole
311 /// `[registration]` table from a serialized machine TOML.
312 pub fn is_empty(&self) -> bool {
313 self.hostkey_fingerprint.is_none() && self.mesh_ipv4.is_none() && self.joined_at.is_none()
314 }
315}
316
317/// Per-machine TOML from `.yah/infra/machines/<name>.toml`.
318///
319/// Two halves, split by provenance (R707-T1): everything here is *declaration*
320/// — operator intent under review and blame — except [`registration`], which
321/// carries what the fleet observed. See [`MachineRegistration`] for why the
322/// boundary is drawn there and what depends on it.
323///
324/// @yah:ticket(R860-T5, "Model per-node native-exec capability as an admission axis (W338 §Placement consequences 3 / R858-T4 gap)")
325/// @yah:status(review)
326/// @yah:phase(P1)
327/// @yah:at(2026-09-05T18:29:19Z)
328/// @yah:assignee(agent:bundle-anthropic-ashguard)
329/// @yah:parent(R860)
330/// @yah:next("Cheapest defensible shape: express it on MachineConfig, which already has the two vocabularies — `mesh_tags: Vec<String>` (config.rs:246, superset match, already carries `arch:`/`os:`/`tag:build-worker`) and `taints: Vec<String>` (config.rs:337). A `native-exec` mesh tag required by any group member whose kind is native is a one-line admission axis in `admission_spec()`. Whichever is chosen, it must be declared in .yah/infra/machines/*.toml for the nodes that actually run kamaji with --native-exec-dir, and `check_inert_taints` (config.rs:703) lints unread taint keys dead — so a taint nobody reads will be flagged.")
331/// @yah:verify("cargo test -p cloud --lib config")
332/// @arch:see(.yah/docs/working/W338-workload-dependencies-and-appliance-composition.md)
333/// @yah:depends_on(R860-T4)
334/// @yah:gotcha("Verified 2026-09-04: native-exec capability is modelled NOWHERE in placement — `rg \"native\" oss/yubaba/crates/cloud/src/config.rs` returns zero hits, and the raft state machine models no member attributes, labels or taints at all (`rg \"taint|capabilit|labels|mesh_tag\"` over raft/{mod,store,network}.rs yields one unrelated comment at raft/store.rs:591). Native-exec is a node-local kamaji startup decision today: `--native-exec-dir` (oss/kamaji/crates/kamaji-bin/src/main.rs:152-156, :51-55) plus the `native-exec` cargo feature (kamaji-bin/src/server.rs:329-330). A node without it refuses the deploy at dispatch time and nothing upstream can see that in advance — which is exactly the deploy-time surprise W338 wants turned into a placement precondition.")
335/// @yah:handoff("NATIVE-EXEC IS NOW A PLACEMENT PRECONDITION, NOT A DISPATCH-TIME SURPRISE. New `pub const NATIVE_EXEC_MESH_TAG: &str = \"cap:native-exec\"` in oss/yubaba/crates/cloud/src/config.rs (declared just above `node_selector_mesh_tags`), and one axis in `admission_spec()` immediately after the R860-T4 group loop: if ANY member of `placement_group(ws, declared)` returns true from `WorkloadSpec::wants_native_exec()`, the tag is appended to the derived `RequiredSpec.mesh_tags` (deduped). No new field on `RequiredSpec`, no signature change anywhere, no wire or serde change — the mesh_tags axis is already an AND-ed superset check against `machine.mesh_tags` in `matches` and is already rendered by `describe`, so a refusal now reads `required.mesh_tags=[...,cap:native-exec]`.")
336/// @yah:handoff("ITEM 1 — HOW A NATIVE WORKLOAD IS DETECTED, settled by opening the type rather than guessing. There is no `kind` on `WorkloadSpec`: on the wire a native workload is still `Workload::Container(WorkloadSpec)`, and the ONLY difference is the annotation `yah.exec = native`, read through `WorkloadSpec::wants_native_exec()` (oss/yah-base/crates/workload-spec/src/lib.rs:2939; consts `NATIVE_EXEC_ANNOTATION` / `NATIVE_EXEC_VALUE` at :3402/:3407). That accessor is what the admission axis calls — matching kamaji, whose `deploy_container` checks the same marker first and routes to `deploy_native_exec` (oss/kamaji/crates/kamaji-bin/src/server.rs). The `yah.exec` key is a substrate selector with a second value, `microvm` (`wants_microvm`, same key, R605-F8), so per-node microVM capability is the obvious sibling axis and is NOT modelled here — see next-steps.")
337/// @yah:handoff("ITEM 2 — DECLARATIONS LANDED ON TWO NODES, FROM READINGS RECORDED IN-REPO, NOT INFERRED. `cap:native-exec` added to `mesh_tags` in .yah/infra/machines/us-west-001.toml and .yah/infra/machines/us-west-003.toml, each with a comment naming its evidence and its re-check condition. us-west-001: the R858 gotcha in its own header records a `ps` reading taken on the box 2026-09-05 — pid 515908 is `/usr/local/bin/kamaji --native-exec-dir /var/lib/yah/kamaji/native`, supervising headscale as a native child. us-west-003: its header's 'THE DEPLOYED KAMAJI PREDATES THE microVM BACKEND' note quotes the box's actual ExecStart, read over ssh 2026-09-01, carrying `--native-exec-dir /var/lib/yah/kamaji/native` (corroborated by .yah/docs/architecture/A043-yah-on-machine-daemons.md's @yah:verify for the same probe). Both comments say plainly that the capability lives in the systemd unit's ExecStart, not in the TOML, so it must be re-checked after any roll.")
338/// @yah:handoff("ITEM 2, THE NEGATIVES — TWO NODES ARE KNOWN NOT TO HAVE IT AND WERE DELIBERATELY LEFT UNSET. us-south-001: kamaji refused headscale there 2026-09-03 with 'native backend not configured — start kamaji with --native-exec-dir' (the R858 chain, quoted in .yah/infra/machines/us-west-001.toml and W267). I did NOT edit us-south-001.toml — it was already dirty in the working tree at the anchor SHA and @Ashguard:eclipse is live on R858, so I left it alone rather than race it; the mechanism fails closed there, which is the correct state. us-west-015 (the sole darwin builder): W254-darwin-build-nodes.md's own next-step records that its kamaji is built/started `--docker` only. I added a comment to us-west-015.toml explaining that the tag is deliberately absent, that this is the node where the axis changes an error message (a darwin build row is native by construction, so it is now refused at ELECTION naming cap:native-exec instead of reaching the box and being refused by kamaji), and the exact enable sequence: rebuild with `--features native-exec`, restart with `--native-exec-dir <dir>`, THEN add the tag. us-west-002/011/013/014 are unestablished from the repo and left unset. THE OPERATOR-FACING ANSWER: the file is `.yah/infra/machines/<node>.toml` and the key is `mesh_tags`; add the literal string `cap:native-exec` to that array, and only after the roll.")
339/// @yah:handoff("DECISIONS THE BRIEF LEFT OPEN, all recorded in doc comments at the site. (1) MESH TAG, NOT TAINT — as recommended, and the doc says why in the terms the brief asked for: mesh tags are positive capability with superset matching ('this node CAN'), which is the claim being made; a taint is repulsion and would have to be inverted to `no-native-exec` on every node LACKING the backend (declaration burden on the majority, and silently wrong for a node nobody has edited) AND taught to `taint_effect`, or `check_inert_taints` would correctly lint the key dead. (2) THE `cap:` NAMESPACE IS NEW. Live prefixes are `tag:` (operator-assigned role), `arch:`/`os:` (silicon and userland facts, emitted as requirements by qed::platform::build_worker_mesh_tags), and `tier:` which R763 RETIRED for architecture and reserved for the environment axis — so reusing any of them would have stated the wrong kind of fact. A capability the daemon was configured with is none of those. Nothing validates tag prefixes (only `check_retired_arch_tags` looks at one), so this costs no wiring. (3) COMPUTED OVER THE GROUP, not the requirer — that is literally W338's sentence ('supply = self specs must be placeable where their requirer lands'), and the second test proves it: an ordinary container requirer with a `local` edge to a native provider is pulled onto a capable node. (4) FAILS CLOSED, accepted deliberately: an undeclared node is simply not a candidate, so an undeclared fleet reports 'no node admits' at election rather than dispatching to a node that refuses. Nothing in `.yah/infra/workloads/` is native-marked today (only yah-cloud-admin.toml exists there), so the only live consumer is the qed darwin build row, where failing closed is strictly the better error.")
340/// @yah:handoff("BLAST RADIUS, MEASURED. `admission_spec` is private and its callers are unchanged: `admit_workload` / `admit_workload_candidates` / `admit_workload_in_group` (config.rs), reached from app/yah/cli/src/cloud.rs (deploy, rolling, topology analyzer), app/yah/cli/src/yubaba_client.rs `elect_node`, and cloud/src/migrate.rs. The headscale appliance path inside yubaba (headscale_appliance.rs) does NOT go through admission — it is node-internal — so nothing eclipse holds on R858 is touched by this. Files edited, in full: oss/yubaba/crates/cloud/src/config.rs; .yah/infra/machines/{us-west-001,us-west-003,us-west-015}.toml. Nothing in oss/yubaba/crates/yubaba/ was opened, and oss/kamaji/crates/kamaji-bin/src/server.rs was READ ONLY (to confirm the marker check), per @Ashguard:hydra's contention triage.")
341/// @yah:handoff("ONE SCOPE ADDITION, stated loudly rather than slipped in: `MachineConfig::mesh_tags` (config.rs:256) had NO doc comment at all — the operator-facing declaration key for four tag namespaces was undocumented. I gave it one enumerating `tag:` / `arch:`+`os:` / the new `cap:` / retired `tier:`, and noting that nothing validates the prefix (which is why the two lints exist). CONSEQUENCE TO KNOW: that field's doc is the source of the `mesh_tags` description in the GENERATED .yah/schema/machine.toml.schema.json, so it is schema-drift-affecting — see the gotcha.")
342/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I found it and left it. Quote this SHA rather than 'HEAD' in any revert/restore instruction; to undo a hunk, read it with `git show 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2:<path>` and put it back with Edit, never `git checkout`/`restore` (they restore whole files and would delete peers' uncommitted work).")
343/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
344/// @yah:next("MICROVM IS THE IDENTICAL UNMODELLED GAP, one line away. `yah.exec` is a substrate selector with a second value: `WorkloadSpec::wants_microvm()` (workload-spec/src/lib.rs, R605-F8), and kamaji constructs MicroVmRuntime only when started with `--microvm-dir` — A043's probe records that us-west-003's deployed kamaji has `--native-exec-dir` but NOT `--microvm-dir`, so a microvm-marked deploy is refused there by exactly the same dispatch-time surprise this ticket removed for native. The shape is `cap:microvm` alongside NATIVE_EXEC_MESH_TAG in the same `if` in `admission_spec`. Not done here because no node in the fleet can host one yet (R605-F14 must land a guest kernel + rootfs first), so declaring the tag anywhere today would be the wrong fact.")
345/// @yah:next("us-south-001 needs `cap:native-exec` DECIDED, not defaulted, and it is the R858 node. It is the one machine the repo positively records as LACKING the backend (kamaji refused headscale there 2026-09-03), so leaving the tag off is correct TODAY — but if R858's fix is 'give us-south-001 a native-capable kamaji' rather than 'stop moving headscale', then the roll and the tag must land together, in that order. I left .yah/infra/machines/us-south-001.toml untouched because it was already dirty at anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 and @Ashguard:eclipse is live on R858.")
346/// @yah:next("R860-T6 (`supply = \"self\"` provisioning) inherits this for free — `admission_spec` already requires the capability of the whole group, so a self-provisioned native member cannot be elected onto a node that cannot run it. What T6 must still not do is re-elect per member: reuse the node URL `elect_node` returned for the requirer, per R860-T4's handoff.")
347/// @yah:verify("BASELINE RECORDED BEFORE EDITING, at tree anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2: `cargo test -p yah-cloud --lib` from oss/yubaba = 1090 passed / 0 failed / 4 ignored, exit 0 — exactly the count the brief predicted. AFTER: 1093 passed / 0 failed / 4 ignored, exit 0 (+3, exactly the three tests added). `cargo check -p yah-cloud --all-targets` exit 0 and `cargo check -p yubaba --all-targets` exit 0 (yubaba consumes cloud, so it is where any signature change would surface — there is none). Every exit code echoed explicitly via an `EXIT=$?` / `${PIPESTATUS[0]}` marker and read back, never inferred from an empty grep. The four `yah-cloud` warnings are all pre-existing and in other files (object-store r2.rs, reconciler/mesofact_static.rs unused imports, app_manifest.rs, reconciler/mod.rs non_snake_case); config.rs contributes none.")
348/// @yah:verify("NEW TESTS (config.rs `mod tests`, R860-T5 section at the end, after the R860-T4 block). (1) a_node_without_the_native_exec_capability_cannot_host_a_native_workload — a `yah.exec = native` spec is refused by a bare node with an error naming `cap:native-exec`, and admitted by a node declaring it, with both nodes in the same fleet so the choice is provably the tag. (2) a_local_edge_to_a_native_provider_makes_the_requirer_need_the_capability — an ordinary container requirer (asserted `!wants_native_exec()`) with a `local` edge to a native provider lands on the capable node, while the SAME spec without the edge still lands on the plain one, so the constraint provably comes from the group. (3) a_group_with_no_native_member_does_not_require_the_capability — the regression guard: the axis is absent from `admission_spec`'s mesh_tags and a group with a local edge between two ordinary specs still admits on a node declaring nothing. Helper `native_spec()` asserts the marker reads back through `wants_native_exec()` before the test uses it, so a typo cannot make the test pass vacuously.")
349/// @yah:verify("Machine-config lints were considered and are unaffected by construction: `check_inert_taints` reads `taints` (I touched none), and `check_retired_arch_tags` flags only the `tier:` prefix. `cap:` is a new namespace and nothing validates prefixes, so no lint fires and no lint needs teaching.")
350/// @yah:gotcha("SCHEMA DRIFT IS EXPECTED FROM THIS TICKET AND WAS ALREADY RED BEFORE IT. `.yah/schema/machine.toml.schema.json` is generated from `cloud::config` by `cargo run -p xtask -- emit-schemas`, and MachineConfig's DOC COMMENT is what the generator emits as its `description` — which means (a) my new `mesh_tags` doc changes it, and (b) so does this very handoff, because R860-T5's @yah: annotation block lives inside MachineConfig's doc at config.rs:201. That file was ALSO already dirty in the working tree at anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2, before I touched anything — `scripts/check-schema-drift.sh` compares the regenerated tree against git, so it is red for any uncommitted schema edit regardless of author. Regenerate with `cargo run -p xtask -- emit-schemas` (or `scripts/check-schema-drift.sh --update`) when the root target dir is not contended; the pre-commit hook no longer does it (disabled 2026-08-15, see CLAUDE.md).")
351/// @yah:verify("SCHEMA REGENERATED IN THIS SESSION, so the drift gate is not left for the next reader: `cargo run --quiet -p xtask -- emit-schemas` exit 0, run from the repo root after the handoff was written (so it captures the annotation text too). Two files moved. `.yah/schema/machine.toml.schema.json`: MachineConfig's `description` grows by this ticket's annotation block, plus a genuinely new `mesh_tags.description` from the doc comment I added. `.yah/schema/workload.toml.schema.json`: +104 lines that are NOT mine — the `Locality` / `Requirement` / `Supply` / `WorkloadSpec.requires` types R860-T1 landed had never been emitted, so the sibling ticket's schema drift was still outstanding and my regen swept it in. Derived artifacts are not ownable (shared-tree doctrine), so this is deliberate rather than accidental; @Ashguard, whoever picks up R860-T1's review should know the schema now describes `requires`.")
352/// @yah:verify("FINAL RE-RUN AFTER THE HANDOFF ANNOTATION WAS WRITTEN INTO config.rs (the board write edits MachineConfig's doc block, so the file changed under the earlier green): `cargo test -p yah-cloud --lib` = 1093 passed / 0 failed / 4 ignored, exit 0. Unchanged. Note for anyone reading the camp build rail's skew warnings on this session: the one `SUSPECT RESULT` it emitted names `oss/yubaba/crates/cloud/src/config.rs` as modified mid-run, and that modification was MY OWN board_handoff annotation write, not a peer — the two authoritative runs (full lib test, and both cargo checks) each came back `Input closure unchanged across the whole run: no skew`.")
353/// @yah:verify("All builds were run with `CARGO_TARGET_DIR=/tmp/r860t5-target` rather than the shared oss/yubaba/target, following R860-T4's recorded gotcha — a peer (session:83093d9d) held the shared target lock for the entire session (20+ minutes of `cargo check -p yubaba --lib`). Costs one cold dep build, then every subsequent run is seconds. Worth reaching for immediately when the queue message says you are behind someone.")
354/// @yah:handoff("LEADER RE-VERIFIED (session:69b18855, independent of the courier's self-report). `cargo test -p yah-cloud --lib` from oss/yubaba: 1093 passed / 0 failed / 4 ignored, exit 0, against the 1090/0/4 baseline this relay's own T4 established — +3 = exactly its new tests. Axis confirmed by content: `NATIVE_EXEC_MESH_TAG = \"cap:native-exec\"` at config.rs:2304, appended to the derived `RequiredSpec.mesh_tags` at :2112-2114 when any `placement_group` member returns true from `WorkloadSpec::wants_native_exec()`. No new `RequiredSpec` field, no signature change, no wire change — it rides the existing AND-ed superset check, so a refusal now reads `required.mesh_tags=[...,cap:native-exec]`.")
355/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
356/// @yah:verify("MACHINE DECLARATIONS AUDITED FOR PROVENANCE, because a wrong capability declaration is worse than an absent one. Both are traceable to measurements ALREADY RECORDED IN-REPO, not inferred: us-west-001 from the `ps` reading at us-west-001.toml:21 (pid 517125, ppid 515908 = `/usr/local/bin/kamaji --native-exec-dir /var/lib/yah/kamaji/native`, cgroup `0::/yubaba.slice/kamaji.service/native`, 2026-09-05); us-west-003 from the actual ExecStart read over ssh 2026-09-01 at us-west-003.toml:141. us-west-015 was deliberately left WITHOUT the tag and carries enable instructions at :207-217 — unknown fails closed, which is the correct direction. No node was guessed at and nothing was probed live.")
357/// @yah:handoff("THIS TICKET MODELS THE EXACT DRIFT THAT CAUSED THE 25-HOUR MESH OUTAGE, which is worth stating because it turns an abstract W338 bullet into a measured one. us-west-001.toml:8 records the root-cause chain: on 2026-09-03T06:03:03Z leadership moved to us-south-001, which tried to deploy headscale and kamaji refused — \\\"workload requests native host execution (yah.exec=native) but no native backend is available (native backend not configured — start kamaji with --native-exec-dir)\\\" — then the systemd fallback failed too, both at WARN, and the mesh had no coordination server for 25 hours. us-west-001.toml:10 names it explicitly as \\\"a silent per-node capability drift that placement does not model\\\". After this ticket, placement models it: a group needing native exec can no longer be admitted onto a node that has not declared `cap:native-exec`.")
358/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
359/// @yah:verify("RE-VERIFIED AT HEAD 00ee20d1 (session:aa5e882d, 2026-09-05). `NATIVE_EXEC_MESH_TAG` present in oss/yubaba/crates/cloud/src/config.rs (declared above `node_selector_mesh_tags`, appended to the derived `RequiredSpec.mesh_tags` when any `placement_group` member wants native exec). `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` = 1141 passed / 0 failed / 4 ignored, exit 0.")
360/// @yah:cleanup("cap:microvm remains the identical unmodelled axis, one line from done in the same `if` in `admission_spec`. Deliberately NOT taken: no node in the fleet can host a microvm until R605-F14 lands a guest kernel + rootfs, so declaring the tag today would assert a false fact. Do it when R605-F14 lands, not before.")
361#[derive(Debug, Clone, Serialize, Deserialize)]
362#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
363pub struct MachineConfig {
364 pub name: String,
365 pub provider: String,
366 /// Who the hardware actually comes from (`"ovh"`, `"vultr"`, `"on-prem"`).
367 ///
368 /// Deliberately *not* [`provider`](Self::provider), which selects the
369 /// auto-provision driver: a box we rented by hand and brought up over SSH
370 /// is `provider = "static"` for its whole life, and writing the vendor
371 /// there instead would flip it driver-backed and make
372 /// [`validate`](Self::validate) demand `location` + `server_type` it has no
373 /// answer for. The two axes genuinely differ — vendor is who bills you,
374 /// `provider` is who yah can call an API against.
375 ///
376 /// Worth recording because vendor-scoped policy is invisible in every other
377 /// field and decides real work: outbound port 25, rDNS/PTR control, IP
378 /// reputation, egress billing. It survived only in TOML prose until now,
379 /// which made it ungreppable at exactly the moment you need it.
380 #[serde(default, skip_serializing_if = "Option::is_none")]
381 pub vendor: Option<String>,
382 /// Human label for the box (`"gamer"`, `"the GEEKOM"`). Free-form and never
383 /// matched on — [`name`](Self::name) stays the identity everywhere. This is
384 /// only so operators and agents can say which box they mean out loud.
385 #[serde(default, skip_serializing_if = "Option::is_none")]
386 pub nickname: Option<String>,
387 /// Provider DC code (e.g. Hetzner `"hil"`). **Provisioning-only**: required
388 /// iff the provider has an auto-provision driver ([`provider_has_machine_driver`]);
389 /// a BYO `static` node we brought up over SSH has no such code. Optional at
390 /// load time so static machine.tomls omit it; [`MachineConfig::validate`]
391 /// enforces presence at the right moment for driver-backed providers.
392 #[serde(default, skip_serializing_if = "Option::is_none")]
393 pub location: Option<String>,
394 /// Provider SKU/size (e.g. Hetzner `"ccx13"`). Provisioning-only, same
395 /// optionality contract as [`location`](Self::location).
396 #[serde(default, skip_serializing_if = "Option::is_none")]
397 pub server_type: Option<String>,
398 /// **Deprecated (R330-F16).** A machine should describe *itself* (region,
399 /// zone, provider, mesh_tags); *which* mirrors run on it is derived by the
400 /// reconciler from each mirror's `required` placement spec, not declared
401 /// here. Now optional + omitted-when-empty so new machine.tomls leave it
402 /// out. The legacy `resolve_mirror_machine` topology fallback still reads
403 /// it until yubaba's reverse-index supersedes the topology.toml path; once
404 /// that lands, this field and its readers are removed wholesale.
405 #[serde(default, skip_serializing_if = "Vec::is_empty")]
406 pub hosts_mirrors: Vec<String>,
407 /// Positive placement facts about this node, matched as a **superset**:
408 /// a workload is admitted only where every tag it requires is present, so
409 /// adding a tag can only ever make a machine match more, never fewer.
410 ///
411 /// Four namespaces are live, and they are not interchangeable:
412 /// - `tag:<role>` — a role the operator assigns (`tag:build-worker`,
413 /// `tag:qed`, `tag:cloud-runner`, `tag:mac-builder`);
414 /// - `arch:<x86|arm>` / `os:<linux|darwin>` — facts about the silicon and
415 /// userland, emitted as *requirements* by
416 /// [`qed::platform::build_worker_mesh_tags`];
417 /// - `cap:<capability>` — something the node's daemons were configured to
418 /// be able to do. Today just [`NATIVE_EXEC_MESH_TAG`] (R860-T5);
419 /// - `tier:` is **retired** for architecture (R763) and reserved for the
420 /// environment axis — [`crate::validate::check_retired_arch_tags`]
421 /// flags a machine still carrying `tier:<arch>`.
422 ///
423 /// Nothing validates the prefix, which is why the lint above exists: a tag
424 /// nobody requires is silently inert, and a *stale* one silently stops
425 /// matching and reports "no node" rather than "wrong tag".
426 pub mesh_tags: Vec<String>,
427 /// Canonical geo region label (latency axis), e.g. `"us-west"`. F16's three
428 /// topology axes are orthogonal: `region` = geo (latency), `zone` = failure
429 /// domain within a region (HA), `provider` = network/cost. `region` is
430 /// distinct from `location` (the provider's DC code, e.g. Hetzner `"hil"`):
431 /// `location` is provider-scoped, `region` is our provider-neutral label.
432 /// Optional for backward-compat; a machine without it never satisfies a
433 /// `required.regions` constraint.
434 #[serde(default, skip_serializing_if = "Option::is_none")]
435 pub region: Option<String>,
436 /// Failure-domain label within a region (HA axis), e.g. `"hil"`. For
437 /// single-DC Hetzner this typically mirrors `location`. F16 placement
438 /// matches `required.zones` against this. Optional for backward-compat.
439 #[serde(default, skip_serializing_if = "Option::is_none")]
440 pub zone: Option<String>,
441 /// Declared CPU architecture (`"x86_64"` / `"aarch64"`). A machine has
442 /// exactly one — it's a first-class property of the box, not a reach
443 /// detail and not a mesh tag. Drives the yubaba release triple. Optional
444 /// only because there's no provider API to probe it (static nodes declare
445 /// it; a driver-backed provider may leave it unset until known).
446 #[serde(default, skip_serializing_if = "Option::is_none")]
447 pub arch: Option<String>,
448 pub bucket: Option<BucketSpec>,
449 /// **Legacy location, superseded by `[registration].hostkey_fingerprint`**
450 /// (R707-T1). Still deserialized so machine TOMLs written before the split
451 /// keep parsing; never *read* directly — go through
452 /// [`MachineConfig::hostkey_fingerprint`], which prefers the registration
453 /// block. [`MachineConfig::normalize`] folds this into `registration`, and
454 /// [`MachineConfig::save`] normalizes before writing, so a load→save cycle
455 /// migrates the file rather than dropping the value.
456 #[serde(
457 rename = "hostkey_fingerprint",
458 default,
459 skip_serializing_if = "Option::is_none"
460 )]
461 pub legacy_hostkey_fingerprint: Option<String>,
462 /// Provider-side SSH-key IDs (Hetzner: from `GET /v1/ssh_keys`)
463 /// authorized for `root` at create time. Defaults to empty for
464 /// backwards-compat with existing machine declarations; an empty
465 /// list yields a Hetzner-emailed random root password (which the
466 /// driver currently discards). Populate this when you want pre-mesh
467 /// SSH access for bootstrap deploys or recovery.
468 #[serde(default, skip_serializing_if = "Vec::is_empty")]
469 pub ssh_keys: Vec<u64>,
470 /// Cloudflare Tunnel ID this machine joins (e.g. `abc123.cfargotunnel.com`).
471 /// `None` → no tunnel (mesh-only node, no public ingress).
472 /// When set, `yah cloud machine provision` reads `cloudflare-tunnel-token`
473 /// from the keys vault and injects the cloudflared install block into
474 /// cloud-init so the new machine connects to CF edge on first boot.
475 #[serde(default, skip_serializing_if = "Option::is_none")]
476 pub cloudflared: Option<String>,
477 /// Provider-issued floating/reserved IP that follows **public-ingress
478 /// ownership** onto this box — R859-F2 (W267 §Tier 1).
479 ///
480 /// The value is the provider's own identifier, opaque here and interpreted
481 /// only by the matching adapter: a Hetzner numeric floating-IP id as a
482 /// string, an OVH Additional-IP address (`"51.81.85.200"`), a Vultr
483 /// reserved-IP UUID. Same "the adapter is the boundary" convention
484 /// [`crate::envoy::floating_ip::FloatingIpAssignInput::ip_id`] documents.
485 ///
486 /// # Why it lives on the machine
487 ///
488 /// [`crate::envoy::floating_ip`] shipped the `floating_ip.*` verbs and
489 /// three provider adapters with no config anywhere saying *which* floating
490 /// IP is "the" ingress IP — the gap R594-F5 recorded and deliberately left.
491 /// This is that field, and it sits beside [`cloudflared`](Self::cloudflared)
492 /// on purpose: that is already the per-node "how the world reaches this
493 /// box" handle, and a floating IP is the sovereign-tier answer to the same
494 /// question. `[[ingress]]`'s
495 /// [`tunnel_id`](crate::config::IngressEdge::tunnel_id) is the *service*
496 /// side of ingress identity — which cohort a given service fronts through —
497 /// and a floating IP is not per-service: one IP moves between boxes, so it
498 /// cannot be partitioned by slot or hostname.
499 ///
500 /// # Absent means "no floating-IP path", never an error
501 ///
502 /// Most machines have none, and that is the normal case: mesh-only nodes,
503 /// boxes behind a Cloudflare tunnel, and every provider without a
504 /// floating-IP adapter. The effector skips such a machine cleanly rather
505 /// than refusing — see
506 /// [`plan_ingress_owner_effect`](crate::provider::floating_ip::plan_ingress_owner_effect).
507 ///
508 /// # The cohort has to agree
509 ///
510 /// Every machine that can hold the same ingress IP must declare the *same*
511 /// id: the IP is one resource that moves, so two ids inside one
512 /// [`sovereign_group`](Self::sovereign_group) means an ownership flip
513 /// silently reassigns a *different* IP than the one currently serving
514 /// traffic. `yah cloud validate` refuses that
515 /// ([`crate::validate::check_ingress_floating_ip`]) rather than leaving it
516 /// to be discovered during a failover.
517 #[serde(default, skip_serializing_if = "Option::is_none")]
518 pub ingress_floating_ip: Option<String>,
519 /// When `true`, this machine hosts operator-bridge workloads (Tailscale
520 /// operator access to mesh-internal services). `yah cloud machine provision`
521 /// will install tailscaled and run `tailscale up` during cloud-init via the
522 /// `{{OPERATOR_BRIDGE_BLOCK}}` placeholder. Defaults to `false` for
523 /// backward-compat with existing machine declarations.
524 #[serde(default)]
525 pub hosts_operator_bridge: bool,
526 /// BYO `static`-node reach descriptor. Static nodes have no provider API to
527 /// probe, so how the camp reaches them (SSH user@host + the yubaba URL,
528 /// which is loopback until the WireGuard mesh lands) is *declared* here.
529 /// `None` for driver-backed providers (Hetzner/Vultr), whose address is
530 /// resolved from the provider API / mesh at provision time.
531 #[serde(default, skip_serializing_if = "Option::is_none")]
532 pub connect: Option<ConnectSpec>,
533 /// Static node capacity (R572-F3). Declares the node's total hardware
534 /// budget; F5's scheduler subtracts committed workload requests from this
535 /// to check whether a new workload fits. Absent means unconstrained.
536 #[serde(default, skip_serializing_if = "Option::is_none")]
537 pub allocatable: Option<NodeAllocatable>,
538 /// Placement taint keys (R572-F3). A repelling key blocks placement by
539 /// default, and a placement opts back in by naming that exact key in
540 /// [`RequiredSpec::tolerates`] (R876-B7).
541 ///
542 /// The `unless` is real now. It was not between R742-T4 and R876-B7: the
543 /// spec side declared which archetypes it *was* rather than which taints it
544 /// tolerated, that field was `#[serde(skip)]`, and so every placement
545 /// declared as `required = {...}` in a mirror TOML read this list as empty
546 /// and could not be drained at all. See [`RequiredSpec::tolerates`].
547 ///
548 /// A key in this list influences placement in exactly one of two ways, and
549 /// [`taint_effect`] is the authority on which:
550 ///
551 /// - **repulsion** — `"no-server"` / `"no-appliance"` / `"no-job"` reject
552 /// any placement that does not tolerate them. The archetype in the key is
553 /// now vocabulary rather than a filter: `matches` does not compare it
554 /// against the workload's class, it checks the toleration list, and
555 /// [`admission_spec`] is what turns a workload's class into the
556 /// tolerations that reproduce the old archetype-scoped behaviour;
557 /// - **affinity** — a key in [`AFFINITY_TAINT_KEYS`] (today just
558 /// `"public-ip"`) that a workload names in
559 /// `yah.placement.requires-taint`, which then *requires* this node.
560 ///
561 /// Anything else is **inert**: it parses, it round-trips, and no scheduler
562 /// decision can ever read it. `yah cloud validate` rejects such keys
563 /// (`validate::check_inert_taints`) rather than letting them sit looking
564 /// load-bearing — which is how `no-voter` spent months asserting a
565 /// falsehood on three nodes. Facts about a node that are not placement
566 /// inputs belong in [`mesh_tags`](Self::mesh_tags) or a comment.
567 #[serde(default, skip_serializing_if = "Vec::is_empty")]
568 pub taints: Vec<String>,
569 /// Which consensus group this node belongs to — W305/R742-F1. `None` means
570 /// standalone: in no group at all, which is us-west-002 and us-west-015.
571 ///
572 /// Membership is not by itself quorum eligibility; that is
573 /// [`sovereign_role`](Self::sovereign_role), added by R605-F12 because
574 /// us-west-003 is in prod's blast radius *and* must never vote in it.
575 ///
576 /// **Not a placement input.** It is deliberately absent from
577 /// [`RequiredSpec::matches`], and adding it there would be a category
578 /// error: a sovereign group is a *blast radius*, not a filter. Nothing
579 /// about "which quorum does this box vote in" should decide where a
580 /// workload runs — that is what made the fleet express three unrelated
581 /// properties through one taint list and get all three wrong (W305).
582 ///
583 /// What it *is* for is refusal. [`judge_join`] answers "may this node join
584 /// that node's cluster", and the answer is no unless both declare the same
585 /// group. Before this field the only guard was a comment in three machine
586 /// TOMLs saying "never run a raft join against this box from a shell
587 /// pointed at prod" — habit, with no mechanism behind it, which is the
588 /// same class of guard W257 §8 admitted to.
589 ///
590 /// # Why `sovereign_group` and not `raft_group`
591 ///
592 /// Raft is today's mechanism (operator, 2026-08-10). A field named for the
593 /// mechanism goes stale the day the mechanism is swapped, and every
594 /// consumer that reads it inherits the lie. `sovereign` names what the
595 /// group *has* — its own authority, its own upgrade cadence, its own
596 /// destruction — which stays true under any consensus protocol.
597 ///
598 /// Note the word already appears in this tree as prose (W267's title, the
599 /// `IngressProvider::Passway` doc comment's "sovereign edge"). That is an
600 /// adjective meaning "self-hosted, not SaaS"; this is the first time it
601 /// carries structure.
602 #[serde(default, skip_serializing_if = "Option::is_none")]
603 pub sovereign_group: Option<String>,
604 /// Whether this node may hold a seat in its group's quorum — R605-F12.
605 /// Meaningless without [`sovereign_group`](Self::sovereign_group): a
606 /// standalone box has no quorum to be eligible for.
607 ///
608 /// **`None` is "not written", not a third role.** Read it through
609 /// [`sovereign_membership`](Self::sovereign_membership), which resolves the
610 /// absence to [`SovereignRole::Voter`] — what declaring a group has always
611 /// meant, so the six nodes stamped before this field keep their seats
612 /// without an edit. The distinction is kept only so
613 /// [`crate::validate::check_unroled_sovereign_members`] can tell an
614 /// operator who *chose* voter from one who never considered the question;
615 /// no join decision reads the `Option` directly.
616 ///
617 /// # Why this is not a taint
618 ///
619 /// It was, once: `no-voter` sat in [`taints`](Self::taints) on three nodes
620 /// for months, read by nothing, and R742-T4 removed it because the taint
621 /// list is a *placement* vocabulary and this is not a placement input (see
622 /// [`taint_effect`]). Nor is it a second group label. It is a modifier on
623 /// the membership this node already declares, which is why it lives beside
624 /// the group and is judged with it in one predicate,
625 /// [`workload_spec::sovereign::join_permitted`].
626 #[serde(default, skip_serializing_if = "Option::is_none")]
627 pub sovereign_role: Option<SovereignRole>,
628 /// `[registration]` — the observed half (R707-T1). Empty until the box has
629 /// been attached / mesh-joined. See [`MachineRegistration`].
630 #[serde(default, skip_serializing_if = "MachineRegistration::is_empty")]
631 pub registration: MachineRegistration,
632}
633
634/// True iff `provider` has an auto-provision driver (create/destroy via API).
635/// Driver-backed providers require `location` + `server_type`; BYO `static`
636/// nodes (brought up over SSH) do not. The cloud-vs-vps distinction the fleet
637/// cares about lives here — at the provider-capability layer — not as a
638/// separate machine type (W242 BYO Phase-0 decision).
639pub fn provider_has_machine_driver(provider: &str) -> bool {
640 matches!(provider, "hetzner" | "vultr" | "digitalocean")
641}
642
643/// Taint keys a workload may name in `yah.placement.requires-taint` to
644/// *require* a node (W305/R742-T4 affinity vocabulary).
645///
646/// This is a closed list on purpose. `WorkloadSpec::requires_taint` returns
647/// free text, but every producer in the tree is code — `passway_ingress.rs`
648/// and `cloudflared_ingress.rs`, both emitting
649/// [`workload_spec::PUBLIC_IP_TAINT`] — and no on-disk `workload.toml` sets the
650/// annotation at all. So the set of keys a node can usefully carry for
651/// affinity is knowable at compile time, which is what lets
652/// [`taint_effect`] call anything outside it inert instead of guessing.
653///
654/// **Adding an affinity key means adding it here**, in the same change that
655/// teaches a workload to require it. That coupling is the point: it makes the
656/// node side and the workload side impossible to land apart.
657pub const AFFINITY_TAINT_KEYS: &[&str] = &[workload_spec::PUBLIC_IP_TAINT];
658
659/// How a key in [`MachineConfig::taints`] can affect placement.
660///
661/// W305 finding 1: before R742-T4 nothing asked this question, so a key that
662/// no scheduler path could read — `"qa"`, `"no-voter"` — parsed, validated,
663/// and quietly did nothing. Both of the findings that cost real fleet state
664/// were invisible for exactly that reason.
665#[derive(Debug, Clone, Copy, PartialEq, Eq)]
666pub enum TaintEffect {
667 /// `"no-<archetype>"`: rejects placement outright unless the constraint
668 /// names this key in [`RequiredSpec::tolerates`]. Read by
669 /// [`RequiredSpec::matches`], which walks `machine.taints` and classifies
670 /// each key through [`taint_effect`] (R876-B7).
671 Repels(LifecycleArchetype),
672 /// A key in [`AFFINITY_TAINT_KEYS`]: a workload naming it in
673 /// `yah.placement.requires-taint` is restricted to nodes carrying it.
674 Attracts,
675 /// Neither. No placement decision can read this key.
676 Inert,
677}
678
679/// Classify one node taint key. See [`TaintEffect`].
680///
681/// The repulsion half is derived from [`LifecycleArchetype::ALL`] rather than
682/// a literal list, so a fourth archetype makes `no-<its key>` live without an
683/// edit here.
684pub fn taint_effect(key: &str) -> TaintEffect {
685 if let Some(arch) = LifecycleArchetype::ALL
686 .into_iter()
687 .find(|a| key == format!("no-{}", a.taint_key()))
688 {
689 return TaintEffect::Repels(arch);
690 }
691 if AFFINITY_TAINT_KEYS.contains(&key) {
692 return TaintEffect::Attracts;
693 }
694 TaintEffect::Inert
695}
696
697/// Every key the scheduler *can* act on, sorted — for error messages that
698/// tell the operator what the legal vocabulary actually is instead of only
699/// what was wrong.
700pub fn live_taint_keys() -> Vec<String> {
701 let mut keys: Vec<String> = LifecycleArchetype::ALL
702 .into_iter()
703 .map(|a| format!("no-{}", a.taint_key()))
704 .chain(AFFINITY_TAINT_KEYS.iter().map(|k| (*k).to_string()))
705 .collect();
706 keys.sort();
707 keys
708}
709
710/// What [`judge_join`] decided about one proposed cluster join.
711///
712/// Shaped like yubaba's `PromotionVerdict` / `GeographyVerdict` and for the
713/// same reason: the rule stays unit-testable without a live cluster, and a
714/// refusal carries its reason from the place that knows it.
715#[derive(Debug, Clone, PartialEq, Eq)]
716pub enum JoinVerdict {
717 /// Both nodes declare the same sovereign group and both are voters. The
718 /// join is within one blast radius and grows a quorum both sides are
719 /// eligible for.
720 Permit,
721 /// The join is refused. Carries an operator-readable reason naming both
722 /// declared values and the file to edit — a refusal that only says
723 /// "invalid" gets worked around rather than fixed.
724 Refuse(String),
725}
726
727/// May `joiner` join the cluster `target` belongs to? — W305/R742-F1.
728///
729/// **A join is permitted iff both nodes declare the same non-`None`
730/// [`sovereign_group`](MachineConfig::sovereign_group) and both are
731/// [`SovereignRole::Voter`].** One rule, no special cases, and it makes the
732/// declaration mandatory before any quorum grows.
733///
734/// The case this exists for is two *different* declared groups: joining a dev
735/// Pi into prod is refused rather than trusted, where today the only guard is
736/// a comment saying not to do it. But an undeclared node is refused too, and
737/// that is the deliberate half — `None` means "in no group", not "unknown", so
738/// growing prod with an unstamped box is exactly as much a cross-group join as
739/// the dev case is. Failing open there would leave the operator believing a
740/// guarantee that was never evaluated, which is the reasoning
741/// `QuorumGeography::judge` already applies to untagged voters.
742///
743/// No legitimate flow pays for that strictness: prod and dev are both stamped,
744/// and us-west-002/015 are deliberately in no group at all. Adding a real
745/// member means declaring it first, which is the point.
746///
747/// # The non-voting refusal (R605-F12)
748///
749/// Same group and still refused, when either side declares
750/// [`SovereignRole::NonVoter`]. This is the case a group label alone could not
751/// express. us-west-003 is a residential-uplink build box the operator counts
752/// as part of prod — same secrets, same upgrade cadence, same destruction — and
753/// which must never hold a prod raft seat, because a home-internet partition
754/// should not be able to stall the quorum. Until R605-F12 the only thing
755/// refusing it was its *absent* stamp, so recording the operator's real intent
756/// (`sovereign_group = "prod"`) would have removed the guard. Now the intent
757/// and the guard are the same two lines.
758///
759/// Note what this is not: the refusal here is about *voting*, and it says
760/// nothing about the mesh. One mesh spans the whole fleet regardless of group
761/// or role (operator, 2026-08-19); a non-voter is reachable, schedulable and
762/// rollable like any other node.
763///
764/// This is the **camp-side** rendering of the rule. The predicate itself lives
765/// in [`workload_spec::sovereign::join_permitted`] because yubaba's
766/// `POST /raft/add-learner` gate asks the same question and cannot see this
767/// crate (there is deliberately no yubaba → cloud edge). Only the prose is
768/// duplicated, and it has to be: a refusal here names
769/// `.yah/infra/machines/<name>.toml`, while the node-side one has no machine
770/// name in hand and must also name `yubaba serve --sovereign-group`.
771///
772/// The node-side gate is *narrower* on purpose, and the difference is worth
773/// knowing when reading either: a daemon started without `--sovereign-group`
774/// has declared nothing rather than declared standalone, so yubaba resolves
775/// that unknown before it judges, and its gate is in force only once the
776/// cluster being joined declares a group. See `yubaba::sovereign_group`.
777pub fn judge_join(joiner: &MachineConfig, target: &MachineConfig) -> JoinVerdict {
778 let stamp_hint = |m: &MachineConfig| {
779 format!(
780 "declare `sovereign_group = \"<group>\"` in .yah/infra/machines/{}.toml",
781 m.name
782 )
783 };
784 let role_hint = |m: &MachineConfig| {
785 format!(
786 "set `sovereign_role = \"voter\"` in .yah/infra/machines/{}.toml",
787 m.name
788 )
789 };
790 if workload_spec::sovereign::join_permitted(
791 joiner.sovereign_membership(),
792 target.sovereign_membership(),
793 ) {
794 return JoinVerdict::Permit;
795 }
796 let (j, t) = (
797 joiner.sovereign_group.as_deref(),
798 target.sovereign_group.as_deref(),
799 );
800 // Everything below is a refusal; the only permitted shape returned above.
801 //
802 // R605-F12: when both sides name the SAME group, the role is the only thing
803 // left that can have refused, and it gets its own message. Falling through
804 // to the arms below would print "cross-group join refused: 'us-west-003' is
805 // in "prod" and 'us-west-001' is in "prod"" — a message that reads as a bug
806 // in the check rather than a decision about the fleet.
807 //
808 // Deliberately not hoisted above the group comparison. A non-voting joiner
809 // whose target is standalone is refused for *both* reasons, and naming the
810 // role there would send the operator to fix a field that would not have
811 // made the join legal anyway.
812 if let (Some(a), Some(b)) = (j, t) {
813 if a == b {
814 for (m, side, other) in [
815 (joiner, "the joiner", &target.name),
816 (target, "the target", &joiner.name),
817 ] {
818 if m.sovereign_membership().role.is_voter() {
819 continue;
820 }
821 return JoinVerdict::Refuse(format!(
822 "join refused: {side} '{}' is a NON-VOTING member of sovereign group {a:?}, \
823 the same group as '{other}'. It is inside that blast radius — same secrets, \
824 same upgrade cadence, same destruction — but declares itself ineligible for \
825 the quorum, so this is refused by declaration rather than by omission. If it \
826 should genuinely vote, {}; if it should not, this refusal is the field doing \
827 its job and the join is the thing to reconsider.",
828 m.name,
829 role_hint(m),
830 ));
831 }
832 }
833 }
834 match (j, t) {
835 (Some(a), Some(b)) => JoinVerdict::Refuse(format!(
836 "cross-group join refused: '{}' is in sovereign group {a:?} and '{}' is in {b:?}. \
837 These are separate blast radii — separate quorums, separate upgrade cadences, \
838 separately destroyable — and merging them is not something a join can undo. If \
839 the move is genuinely intended, restamp '{}' to {b:?} first and treat it as \
840 leaving its old group.",
841 joiner.name,
842 target.name,
843 joiner.name,
844 )),
845 (None, Some(b)) => JoinVerdict::Refuse(format!(
846 "join refused: '{}' declares no sovereign_group, so it is standalone — in no \
847 group — while '{}' is in {b:?}. That is a cross-group join, not an unchecked \
848 one. To make '{}' a member of {b:?}, {}.",
849 joiner.name,
850 target.name,
851 joiner.name,
852 stamp_hint(joiner),
853 )),
854 (Some(a), None) => JoinVerdict::Refuse(format!(
855 "join refused: '{}' is in sovereign group {a:?} but '{}' declares none, so the \
856 target is standalone and has no group to join. Either {}, or found the group on \
857 '{}' rather than growing it.",
858 joiner.name,
859 target.name,
860 stamp_hint(target),
861 joiner.name,
862 )),
863 (None, None) => JoinVerdict::Refuse(format!(
864 "join refused: neither '{}' nor '{}' declares a sovereign_group, so this join \
865 would form a group nobody declared and nothing could later reason about. Name \
866 the group on both boxes first: {}, and the same for '{}'.",
867 joiner.name,
868 target.name,
869 stamp_hint(joiner),
870 target.name,
871 )),
872 }
873}
874
875impl MachineConfig {
876 /// This node's declared place in a sovereign group, as the shared join rule
877 /// wants it — R605-F12.
878 ///
879 /// The one place `sovereign_role`'s `None` is resolved. Absence means
880 /// [`SovereignRole::Voter`], which is what declaring a group meant before
881 /// the role existed; resolving it here rather than at each call site is what
882 /// keeps the camp-side and node-side gates from disagreeing about a node
883 /// that never wrote the field.
884 pub fn sovereign_membership(&self) -> Membership<'_> {
885 Membership {
886 group: self.sovereign_group.as_deref(),
887 role: self.sovereign_role.unwrap_or_default(),
888 }
889 }
890
891 /// Provider DC code, or `""` when omitted (static nodes). Most readers want
892 /// a `&str`; the driver-backed provision/status paths still go through
893 /// [`validate`](Self::validate) which guarantees presence for those.
894 pub fn location(&self) -> &str {
895 self.location.as_deref().unwrap_or("")
896 }
897
898 /// Provider SKU, or `""` when omitted (static nodes).
899 pub fn server_type(&self) -> &str {
900 self.server_type.as_deref().unwrap_or("")
901 }
902
903 /// Enforce the provisioning-only-field contract: a machine whose provider
904 /// has an auto-provision driver MUST declare `location` + `server_type`
905 /// (the driver can't create a server without them). Static nodes may omit
906 /// both. Call this before any provision/diff that assumes a driver.
907 pub fn validate(&self) -> Result<()> {
908 if provider_has_machine_driver(&self.provider) {
909 if self.location.is_none() {
910 anyhow::bail!(
911 "machine '{}' (provider '{}') has an auto-provision driver but no `location`",
912 self.name,
913 self.provider
914 );
915 }
916 if self.server_type.is_none() {
917 anyhow::bail!(
918 "machine '{}' (provider '{}') has an auto-provision driver but no `server_type`",
919 self.name,
920 self.provider
921 );
922 }
923 }
924 Ok(())
925 }
926
927 /// Declared taints that no placement decision can read (W305/R742-T4).
928 ///
929 /// Deliberately **not** folded into [`validate`](Self::validate): that
930 /// guard runs on the provision/diff hot path and answers a different
931 /// question (can the driver create this server). An inert taint is a lint
932 /// — it never breaks an operation in flight, it just means the file is
933 /// asserting something the scheduler will not honour. `yah cloud validate`
934 /// is where the operator asks for that judgement; see
935 /// [`crate::validate::check_inert_taints`].
936 pub fn inert_taints(&self) -> Vec<&str> {
937 self.taints
938 .iter()
939 .filter(|t| taint_effect(t) == TaintEffect::Inert)
940 .map(String::as_str)
941 .collect()
942 }
943
944 /// Yubaba's TOFU'd hostkey fingerprint, from `[registration]` and falling
945 /// back to the pre-R707-T1 top-level field. **The only read path** — a
946 /// caller that reaches for `legacy_hostkey_fingerprint` directly sees
947 /// `None` on every migrated machine.
948 pub fn hostkey_fingerprint(&self) -> Option<&str> {
949 self.registration
950 .hostkey_fingerprint
951 .as_deref()
952 .or(self.legacy_hostkey_fingerprint.as_deref())
953 }
954
955 /// Record (or clear) the observed hostkey fingerprint. Writes
956 /// `[registration]` and drops any pre-R707-T1 top-level value, so the two
957 /// locations can never disagree after a writeback.
958 pub fn set_hostkey_fingerprint(&mut self, fingerprint: Option<String>) {
959 self.registration.hostkey_fingerprint = fingerprint;
960 self.legacy_hostkey_fingerprint = None;
961 }
962
963 /// Mesh (tailnet) IPv4 for this node, or `None` pre-mesh.
964 ///
965 /// Prefers `[registration].mesh_ipv4`; falls back to the host of a legacy
966 /// `[connect].yubaba` URL when that host is in the `100.64.0.0/10` CGNAT
967 /// range the mesh uses. A loopback placeholder (`http://127.0.0.1:7443`,
968 /// meaning "pre-mesh, reachable only through an SSH tunnel") is *not* a
969 /// mesh address and yields `None`.
970 pub fn mesh_ipv4(&self) -> Option<&str> {
971 if let Some(ip) = self.registration.mesh_ipv4.as_deref() {
972 return Some(ip);
973 }
974 let url = self.connect.as_ref()?.yubaba.as_deref()?;
975 mesh_ipv4_from_url(url)
976 }
977
978 /// Base URL for this node's yubaba, or `None` when no reach resolves.
979 ///
980 /// Thin wrapper over [`reach`](Self::reach) for the many call sites that
981 /// only branch on presence. Prefer `reach` anywhere the operator sees the
982 /// outcome — a `None` here throws away a refusal that names exactly which
983 /// address is missing.
984 pub fn yubaba_url(&self) -> Option<String> {
985 self.reach().ok()
986 }
987
988 /// The **one** address automation dials for this node — mesh-only.
989 ///
990 /// `Err` is a *named refusal*, not an absence: a node with no mesh address
991 /// is unresolvable to every automated path, and R605-T10's whole complaint
992 /// is that this used to surface as a connect timeout against an address the
993 /// caller has no route to.
994 ///
995 /// Resolution order:
996 ///
997 /// 1. A declared `[connect].yubaba` on a **private** host (10/8,
998 /// 172.16/12, 192.168/16) is **not dialed** — see below.
999 /// 2. Any other declared `[connect].yubaba` wins verbatim. That includes
1000 /// the pre-mesh loopback placeholder (`http://127.0.0.1:7443`, "I have
1001 /// no mesh address; reach me through the SSH tunnel to `ssh`"), which is
1002 /// a genuine declaration and stays honoured.
1003 /// 3. Otherwise `[registration].mesh_ipv4` composed with
1004 /// `[connect].yubaba_port`.
1005 ///
1006 /// **Why a LAN literal loses (R605-T10, operator 2026-08-19).** The LAN
1007 /// address is an emergency break-glass route, never an official one, and
1008 /// automation must ALWAYS assume the caller is not on that LAN — this camp
1009 /// sits on 192.168.22.0/22 with no route to the fleet's 192.168.10.0/24 at
1010 /// all. Writing one into the field every resolver dials does not sit beside
1011 /// the mesh route, it *overrides* it: R707-T6 made a declared literal beat
1012 /// `mesh_ipv4` outright, so us-west-011 (mesh-joined, healthy) was elected
1013 /// for every aarch64 build and then dialed at an address that answers only
1014 /// from inside bldg-2506.
1015 ///
1016 /// **What R707-T6 wanted is preserved elsewhere.** Its forcing case was
1017 /// identity, not reach: the dev raft group advertises LAN addrs
1018 /// (`192.168.10.11:7443`, verified live off `/raft/status` 2026-08-27), and
1019 /// `rollout::yubaba::membership_to_nodes` has to map those back to declared
1020 /// machines. That match now runs against [`lan_endpoint`](Self::lan_endpoint),
1021 /// which is composed from the break-glass `[connect].address` metadata and
1022 /// is never dialed — so the two concerns the old precedence rule fused are
1023 /// split, and the literal can stop squatting a dialed field.
1024 ///
1025 /// The LAN address itself STAYS in the machine TOML. It is useful metadata
1026 /// and the manual `ssh` path is entitled to it; it is only disconnected
1027 /// from every automated process.
1028 pub fn reach(&self) -> Result<String, String> {
1029 let Some(connect) = self.connect.as_ref() else {
1030 return Err(format!(
1031 "machine {:?} declares no [connect] block, so nothing knows how to reach it \
1032 \u{2192} declare one, or leave it unprovisioned and out of placement",
1033 self.name
1034 ));
1035 };
1036 let mesh = || {
1037 self.registration
1038 .mesh_ipv4
1039 .as_deref()
1040 .map(|ip| format!("http://{ip}:{}", connect.yubaba_port()))
1041 };
1042 if let Some(literal) = &connect.yubaba {
1043 let Some(lan) = private_ipv4_from_url(literal) else {
1044 return Ok(literal.clone());
1045 };
1046 return mesh().ok_or_else(|| {
1047 format!(
1048 "machine {:?} is unresolvable to automation: its only declared yubaba reach \
1049 is the private literal {:?} and it has no [registration].mesh_ipv4\n\
1050 \u{2192} a LAN address is an emergency break-glass route, never an official \
1051 one (R605-T10) — every automated path assumes the caller is NOT on {}/24\n\
1052 \u{2192} mesh-join the box and record `mesh_ipv4` under [registration], then \
1053 delete `[connect].yubaba` so the port composes with it",
1054 self.name,
1055 literal,
1056 lan.rsplit_once('.').map(|(net, _)| net).unwrap_or(lan),
1057 )
1058 });
1059 }
1060 mesh().ok_or_else(|| {
1061 format!(
1062 "machine {:?} has no [registration].mesh_ipv4 and declares no \
1063 [connect].yubaba, so no automated path can reach it\n\
1064 \u{2192} mesh-join the box and record its tailnet address, or taint it out of \
1065 placement — do not point `[connect].yubaba` at a LAN address (R605-T10)",
1066 self.name
1067 )
1068 })
1069 }
1070
1071 /// The LAN `host:port` this node's yubaba answers on, composed from the
1072 /// break-glass `[connect].address` metadata plus the declared port.
1073 ///
1074 /// **Identity only — never dial this.** It exists so a raft membership
1075 /// entry that names a node by its LAN address can be mapped back to the
1076 /// declared machine (`rollout::yubaba::membership_to_nodes`) without that
1077 /// address having to live in a field a resolver reads. `None` when the
1078 /// machine is unprovisioned.
1079 pub fn lan_endpoint(&self) -> Option<String> {
1080 let connect = self.connect.as_ref()?;
1081 Some(format!("{}:{}", connect.address, connect.yubaba_port()))
1082 }
1083
1084 /// Fold the pre-R707-T1 top-level `hostkey_fingerprint` into
1085 /// `[registration]`, and lift a mesh IP out of a legacy `[connect].yubaba`
1086 /// URL. Idempotent; a machine already on the split shape is untouched.
1087 ///
1088 /// [`save`](Self::save) calls this, so writing a machine TOML migrates it
1089 /// rather than round-tripping the old shape back out.
1090 pub fn normalize(&mut self) {
1091 if let Some(fp) = self.legacy_hostkey_fingerprint.take() {
1092 self.registration.hostkey_fingerprint.get_or_insert(fp);
1093 }
1094 if self.registration.mesh_ipv4.is_none() {
1095 if let Some(ip) = self
1096 .connect
1097 .as_ref()
1098 .and_then(|c| c.yubaba.as_deref())
1099 .and_then(mesh_ipv4_from_url)
1100 .map(str::to_string)
1101 {
1102 self.registration.mesh_ipv4 = Some(ip);
1103 // The URL was pure derivation from mesh IP + port; keep only
1104 // the declared half so the two can't drift apart.
1105 if let Some(c) = self.connect.as_mut() {
1106 c.yubaba = None;
1107 }
1108 }
1109 }
1110 }
1111
1112 /// Persist to `<cloud_dir>/machines/<name>.toml`, creating the dir if needed.
1113 ///
1114 /// ⚠ Serializes the struct, so **operator comments in the target file are
1115 /// lost**. Pre-existing behaviour, not introduced here, but it is why
1116 /// registration writeback (`yah cloud machine attach`) goes through
1117 /// [`crate::state::MachineState`] and the comment-preserving path in the
1118 /// CLI rather than calling this on a hand-authored inventory file.
1119 pub fn save(&self, cloud_dir: &Path) -> Result<()> {
1120 let dir = cloud_dir.join("machines");
1121 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
1122 let path = dir.join(format!("{}.toml", self.name));
1123 let mut normalized = self.clone();
1124 normalized.normalize();
1125 let s = toml::to_string_pretty(&normalized)
1126 .with_context(|| format!("serializing machine {}", self.name))?;
1127 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
1128 }
1129}
1130
1131/// Host of an `http://host:port` URL iff it is a mesh (headscale) IPv4 in the
1132/// `100.64.0.0/10` CGNAT range. String-level rather than URL-parsed: the
1133/// inventory format is stable and this crate carries no URL dependency (same
1134/// reasoning as `fleet_metrics::extract_host` and
1135/// `hub::coordinator::is_loopback_url`).
1136fn mesh_ipv4_from_url(url: &str) -> Option<&str> {
1137 let host = ipv4_host_of(url)?;
1138 let ip: std::net::Ipv4Addr = host.parse().ok()?;
1139 let [a, b, ..] = ip.octets();
1140 // 100.64.0.0/10 ⇒ first octet 100, second octet 64..=127.
1141 (a == 100 && (64..=127).contains(&b)).then_some(host)
1142}
1143
1144/// Host of an `http://host:port` URL iff it is an **RFC1918 private** IPv4 —
1145/// `10/8`, `172.16/12`, `192.168/16`. `None` for anything else, loopback and
1146/// the `100.64/10` mesh range included: neither is a LAN literal.
1147///
1148/// The judgement R605-T10 turns on. A private literal is only ever reachable
1149/// from inside one building, so it is metadata about where the box physically
1150/// sits and never an address automation may dial — see
1151/// [`MachineConfig::reach`] and [`crate::validate::check_lan_dial_targets`].
1152pub fn private_ipv4_from_url(url: &str) -> Option<&str> {
1153 let host = ipv4_host_of(url)?;
1154 is_private_ipv4(host).then_some(host)
1155}
1156
1157/// Whether a bare host string is an RFC1918 private IPv4 literal.
1158pub fn is_private_ipv4(host: &str) -> bool {
1159 let Ok(ip) = host.parse::<std::net::Ipv4Addr>() else {
1160 return false;
1161 };
1162 ip.is_private()
1163}
1164
1165/// Bare host of a `[scheme://]host[:port][/path]` string.
1166fn ipv4_host_of(url: &str) -> Option<&str> {
1167 let after_scheme = url.split("://").nth(1).unwrap_or(url);
1168 after_scheme.split(['/', ':']).next()
1169}
1170
1171/// Declared **reach** for a BYO `static` node (no provider API). Lives under
1172/// `[connect]` in the machine TOML.
1173///
1174/// Reach only — how the camp gets to the box. *Permission* is a separate axis
1175/// that belongs to cheers' scopes (W295 §"Deliberately deferred"); the two
1176/// collapse in practice today (mesh membership grants everything) and the data
1177/// model must not fuse them, so do not add an authorization field here.
1178///
1179/// `address`, `ssh` and `identity_file` stay whole, literal, operator-authored
1180/// strings even though their values often *look* derived. They are not:
1181/// us-west-001 dials SSH over its public IP while us-west-002 was deliberately
1182/// repointed at its tailnet IP (R608-F10) precisely because the LAN address is
1183/// unreachable off-LAN. Decomposing them into user + host and recomposing
1184/// would silently undo per-machine decisions like that one. `yubaba` is the
1185/// field that *was* derived — mesh IP plus a fixed port, rewritten by
1186/// mesh-join — so that is where R707-T1 cut.
1187#[derive(Debug, Clone, Serialize, Deserialize)]
1188#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
1189pub struct ConnectSpec {
1190 /// Reachable IPv4/host for the box, e.g. `"45.32.194.254"`. Declared: which
1191 /// of a machine's several addresses the camp should use is an operator
1192 /// choice (public IP vs. LAN IP vs. tailnet IP).
1193 pub address: String,
1194 /// SSH target the camp dials for bootstrap + (pre-mesh) tunneled deploys,
1195 /// e.g. `"root@45.32.194.254"` or `"struc@100.64.0.4"`. Declared, whole —
1196 /// see the type doc. Pair with `identity_file` for a copy-pasteable
1197 /// `ssh -i <identity_file> <ssh>`.
1198 pub ssh: String,
1199 /// Private key path the camp uses to authenticate `ssh`, e.g.
1200 /// `"~/.ssh/yah"`. Every node in the fleet uses the same operator key
1201 /// today, but this is declared per-machine rather than assumed globally
1202 /// for the same reason `ssh` is whole rather than decomposed: a future
1203 /// node with a different key should not have to fight a hardcoded
1204 /// default. `~` is not shell-expanded by this crate — callers that shell
1205 /// out to `ssh`/`scp` pass it through `-i`, which expands it itself.
1206 pub identity_file: String,
1207 /// Port yubaba listens on. Declared reach; defaults to 7443 when omitted,
1208 /// which is every machine in the fleet today. Composed with the *observed*
1209 /// [`MachineRegistration::mesh_ipv4`] by [`MachineConfig::yubaba_url`].
1210 #[serde(default, skip_serializing_if = "Option::is_none")]
1211 pub yubaba_port: Option<u16>,
1212 /// Explicit yubaba base URL, overriding the composed form.
1213 ///
1214 /// Two live uses, both genuine declarations: a pre-mesh node saying
1215 /// `"http://127.0.0.1:7443"` — "I have no mesh address; reach me through
1216 /// the SSH tunnel to `ssh`" — and any node whose yubaba is not at
1217 /// `mesh_ipv4:port`. A URL here whose host *is* a mesh IP is the
1218 /// pre-R707-T1 shape; [`MachineConfig::normalize`] lifts it into
1219 /// `[registration].mesh_ipv4` and clears this field so the two cannot
1220 /// drift apart.
1221 #[serde(default, skip_serializing_if = "Option::is_none")]
1222 pub yubaba: Option<String>,
1223}
1224
1225/// Default yubaba listen port, used when `[connect].yubaba_port` is omitted.
1226pub const DEFAULT_YUBABA_PORT: u16 = 7443;
1227
1228impl ConnectSpec {
1229 /// Declared yubaba port, defaulting to [`DEFAULT_YUBABA_PORT`].
1230 pub fn yubaba_port(&self) -> u16 {
1231 self.yubaba_port.unwrap_or(DEFAULT_YUBABA_PORT)
1232 }
1233}
1234
1235#[derive(Debug, Clone, Serialize, Deserialize)]
1236#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
1237pub struct BucketSpec {
1238 pub name: String,
1239 pub public_read: bool,
1240}
1241
1242/// Per-camp mirror declaration from `.yah/cloud/mirrors/<id>/mirror.toml`
1243/// (folder form) or the legacy `.yah/cloud/mirrors/<id>.toml` (flat form).
1244///
1245/// The folder form is preferred for new mirrors so that per-mirror secrets
1246/// and override files can sit next to `mirror.toml` without polluting the
1247/// top-level `mirrors/` directory.
1248#[derive(Debug, Clone, Serialize, Deserialize)]
1249pub struct LegacyMirrorConfig {
1250 /// Logical camp name this mirror hosts, e.g. `"yah"` or `"noisetable"`.
1251 ///
1252 /// Serialised as `camp`; accepts the legacy `rig` spelling for files that
1253 /// predate the R137 rig→camp rename (one-time migration: `sed -i ''
1254 /// 's/^rig = /camp = /' ~/.yah/cloud/mirrors/*.toml`).
1255 #[serde(rename = "camp", alias = "rig")]
1256 pub camp: String,
1257 pub regions: Vec<String>,
1258 /// Workload names deployed as part of this mirror (references `workloads/<name>.toml`).
1259 /// Renamed from `services` in R092-F1; use `yah cloud config migrate-services-to-workloads`
1260 /// on repos that still have the old `services/` layout.
1261 #[serde(alias = "services")]
1262 pub workloads: Vec<String>,
1263 /// Base domain for Cloudflare-fronted services on this mirror's machines.
1264 /// Combined with the machine's `location` to build virtual-host names:
1265 /// e.g. `cloud_domain = "cloud.noisetable.example"` on machine in location
1266 /// `pdx` → Caddyfile site address `pdx.cloud.noisetable.example`.
1267 /// Optional: if unset the Caddyfile falls back to `:port` listeners.
1268 #[serde(default, skip_serializing_if = "Option::is_none")]
1269 pub cloud_domain: Option<String>,
1270}
1271
1272/// Error from loading or validating a single workload TOML file.
1273#[derive(Debug, Error)]
1274pub enum WorkloadConfigError {
1275 #[error("reading {path}: {source}")]
1276 Io {
1277 path: String,
1278 source: std::io::Error,
1279 },
1280 #[error("parsing {path}: {source}")]
1281 Toml {
1282 path: String,
1283 source: toml::de::Error,
1284 },
1285 #[error("invalid WorkloadSpec in {path}: {source}")]
1286 Shape {
1287 path: String,
1288 source: validate::ShapeError,
1289 },
1290}
1291
1292/// A workload declaration loaded from `.yah/cloud/workloads/<name>.toml`.
1293///
1294/// Each file is the human-authored TOML serialization of a [`WorkloadSpec`].
1295/// On load, the spec is validated against the shape layer; failures surface as
1296/// a [`CloudConfigError::Workload`] with the file path and field path.
1297#[derive(Debug, Clone, Serialize, Deserialize)]
1298pub struct WorkloadConfig {
1299 /// The validated spec.
1300 #[serde(flatten)]
1301 pub spec: WorkloadSpec,
1302}
1303
1304impl WorkloadConfig {
1305 /// Persist to `<cloud_dir>/workloads/<name>.toml`, creating the dir if needed.
1306 pub fn save(&self, cloud_dir: &Path) -> Result<()> {
1307 let dir = cloud_dir.join("workloads");
1308 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
1309 let path = dir.join(format!("{}.toml", self.spec.name));
1310 let s = toml::to_string_pretty(self)
1311 .with_context(|| format!("serializing workload {}", self.spec.name))?;
1312 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
1313 }
1314}
1315
1316/// Error surfaced by [`CloudConfig::load`] when a workload TOML fails validation.
1317#[derive(Debug, Error)]
1318pub enum CloudConfigError {
1319 #[error(transparent)]
1320 Anyhow(#[from] anyhow::Error),
1321 #[error("workload validation failed: {0}")]
1322 Workload(WorkloadConfigError),
1323}
1324
1325/// Mirror-to-machine assignment table from `.yah/cloud/topology.toml`.
1326///
1327/// Declares which logical mirror names are assigned to which machines.
1328/// This is the source-canonical placement until yubaba raft observes it
1329/// (per the migration tracker in the arch doc).
1330#[derive(Debug, Clone, Serialize, Deserialize, Default)]
1331pub struct TopologyConfig {
1332 /// Mirror→machine assignments.
1333 #[serde(default)]
1334 pub assignments: Vec<MirrorAssignment>,
1335 /// Declared buckets, logged by `yah cloud bucket create`.
1336 /// Source-canonical until yubaba raft observes actual placement.
1337 #[serde(default, skip_serializing_if = "Vec::is_empty")]
1338 pub buckets: Vec<BucketLogEntry>,
1339}
1340
1341impl TopologyConfig {
1342 /// Load from a `topology.toml` file, returning `Default` when absent.
1343 pub fn load(path: &Path) -> Result<Self> {
1344 if !path.exists() {
1345 return Ok(Self::default());
1346 }
1347 let s =
1348 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
1349 toml::from_str(&s).with_context(|| format!("parsing {}", path.display()))
1350 }
1351
1352 /// Persist to `topology.toml`, creating parent dirs if needed.
1353 pub fn save(&self, path: &Path) -> Result<()> {
1354 if let Some(parent) = path.parent() {
1355 std::fs::create_dir_all(parent)
1356 .with_context(|| format!("creating {}", parent.display()))?;
1357 }
1358 let s = toml::to_string_pretty(self).context("serializing topology")?;
1359 std::fs::write(path, s).with_context(|| format!("writing {}", path.display()))
1360 }
1361
1362 /// Find a declared bucket by name.
1363 pub fn bucket_by_name(&self, name: &str) -> Option<&BucketLogEntry> {
1364 self.buckets.iter().find(|b| b.name == name)
1365 }
1366
1367 /// Find a mutable declared bucket by name.
1368 pub fn bucket_by_name_mut(&mut self, name: &str) -> Option<&mut BucketLogEntry> {
1369 self.buckets.iter_mut().find(|b| b.name == name)
1370 }
1371
1372 /// Returns true if the bucket is declared as cross-machine (no owning machine).
1373 pub fn is_cross_machine_bucket(&self, name: &str) -> bool {
1374 self.buckets
1375 .iter()
1376 .any(|b| b.name == name && b.machine.is_none())
1377 }
1378}
1379
1380/// One mirror→machine placement entry in `topology.toml`.
1381#[derive(Debug, Clone, Serialize, Deserialize)]
1382pub struct MirrorAssignment {
1383 /// Logical mirror name, e.g. `"noisetable-pdx"`.
1384 pub mirror: String,
1385 /// Machine that hosts this mirror, e.g. `"noisetable-pdx-1"`.
1386 pub machine: String,
1387}
1388
1389/// A bucket declaration logged in `topology.toml` by `yah cloud bucket create`.
1390#[derive(Debug, Clone, Serialize, Deserialize)]
1391pub struct BucketLogEntry {
1392 pub name: String,
1393 /// Machine that owns this bucket. `None` marks it as cross-machine
1394 /// (no single-machine ownership; requires an explicit declaration in
1395 /// `topology.toml` before `yah cloud bucket create` will proceed without
1396 /// `--machine`).
1397 #[serde(default, skip_serializing_if = "Option::is_none")]
1398 pub machine: Option<String>,
1399 /// Logical location of the bucket, e.g. `"pdx"`.
1400 pub location: String,
1401 /// Current declared policy: `"private"` | `"public-read"` | `"signed-only"`.
1402 #[serde(default = "default_bucket_policy")]
1403 pub policy: String,
1404}
1405
1406fn default_bucket_policy() -> String {
1407 "private".to_string()
1408}
1409
1410#[derive(Debug, Clone, Serialize, Deserialize)]
1411pub struct PortMapping {
1412 pub host: u16,
1413 pub container: u16,
1414}
1415
1416/// A loaded service plus its per-environment mirrors.
1417///
1418/// Wraps the `service.toml` body and the directory of `mirrors/<env>.toml`
1419/// files that project the service onto concrete infra.
1420#[derive(Debug, Clone, Serialize, Deserialize)]
1421pub struct ServiceWithMirrors {
1422 pub service: ServiceConfig,
1423 /// Mirrors keyed by environment name (file stem of `mirrors/<env>.toml`).
1424 pub mirrors: BTreeMap<String, MirrorConfig>,
1425 /// Transform recipe names keyed by component id. Populated from each
1426 /// static-asset component's `workload.toml` at load time — not stored
1427 /// in service.toml. Only present for components that declare
1428 /// `[asset.derive.transform] recipe = "..."`.
1429 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
1430 pub component_transform_recipes: BTreeMap<String, String>,
1431 /// Nodes each mirror's passway front door is placed on, keyed by env —
1432 /// exactly what [`MirrorConfig::passway_machines`] returns, with the envs
1433 /// that declare no passway edge left out.
1434 ///
1435 /// Derived at load time like `component_transform_recipes` above: it is
1436 /// stored in no TOML file. It exists so that a consumer of this wire type —
1437 /// the desktop `service_list` command, and through it the Services tab's
1438 /// custom-domain panel — never reconciles the two `ingress` spellings
1439 /// itself. An env present here with an **empty** list is a passway edge
1440 /// whose placement is co-located rather than declared; see
1441 /// [`MirrorConfig::passway_machines`] for why that is a different answer
1442 /// from being absent.
1443 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
1444 pub passway_machines: BTreeMap<String, Vec<String>>,
1445}
1446
1447/// All cloud config loaded from a workspace root (the parent of `.yah/`).
1448///
1449/// Reads two trees:
1450/// - `.yah/infra/` — `machines/`, `providers/`
1451/// - `.yah/services/<svc>/` — `service.toml` + `mirrors/<env>.toml`
1452///
1453/// Pre-R215 fields (`legacy_mirrors`, `workloads`, `topology`) are still
1454/// populated from `.yah/cloud/` when present so pre-R215 callers (bucket
1455/// commands) keep compiling — they just see empty collections in a post-B1
1456/// workspace where the legacy data was deleted. These fields are scheduled
1457/// for removal in B3-T3. (`legacy_services` was the last of this group with
1458/// a live renderer — the compose/Caddy generator over it — and was removed
1459/// along with that renderer in R895-T2.)
1460#[derive(Debug)]
1461pub struct CloudConfig {
1462 /// Workspace root that was loaded — useful for path-resolving
1463 /// component references on a [`ServiceComponent`].
1464 pub workspace_root: std::path::PathBuf,
1465
1466 // ─── R215+ tree ────────────────────────────────────────────────────────
1467 /// `.yah/infra/machines/<name>.toml`
1468 pub machines: Vec<MachineConfig>,
1469 /// `.yah/infra/providers/<id>.toml`
1470 pub providers: Vec<ProviderConfig>,
1471 /// Provenance for every entry in `machines` that came from a linked
1472 /// `.yah/infra/sources.toml` source rather than this camp's own
1473 /// `.yah/infra/machines/` (R615-F2 / W274). Keyed by
1474 /// [`MachineConfig::name`]; a name absent here is camp-local. Empty from
1475 /// [`CloudConfig::load_from_config_dir`] — see its doc for why sources
1476 /// don't apply to multi-root sibling trees.
1477 pub machine_origins: BTreeMap<String, InfraOrigin>,
1478 /// Same as [`machine_origins`](Self::machine_origins), keyed by
1479 /// [`ProviderConfig::id`].
1480 pub provider_origins: BTreeMap<String, InfraOrigin>,
1481 /// `.yah/services/<svc>/` — service.toml plus mirrors/<env>.toml.
1482 pub services: BTreeMap<String, ServiceWithMirrors>,
1483 /// Measured restore times replayed from `.yah/cloud/recovery.jsonl`
1484 /// (R850-T2), keyed by workload name and summed across that workload's
1485 /// subjects. Empty in every camp that has never timed a restore — a missing
1486 /// journal is not an error, exactly like an unsynced infra source.
1487 ///
1488 /// This field is what lets [`crate::topology::analyze`] report a measured
1489 /// recovery instead of an extrapolation **while staying pure**: the
1490 /// measurement is read off the local tree here, at load, alongside every
1491 /// other declaration, so the analyzer gains no I/O and no `Path` argument.
1492 /// See [`crate::recovery_journal`].
1493 pub recovery_measurements: BTreeMap<String, crate::recovery_journal::WorkloadRecovery>,
1494 /// `.yah/domains/<name>.toml` — public-facing routing manifests
1495 /// (R347). Single file per domain; no nested per-env tree because
1496 /// domains themselves aren't projected onto infra — they describe
1497 /// how a Worker bundle ingresses requests onto services.
1498 pub domains: BTreeMap<String, DomainConfig>,
1499
1500 // ─── Pre-R215 legacy (slated for removal in B3-T3) ────────────────────
1501 /// Legacy mirrors from `.yah/cloud/mirrors/`.
1502 pub legacy_mirrors: Vec<LegacyMirrorConfig>,
1503 /// Workloads from `.yah/cloud/workloads/*.toml` (R092-F1 schema).
1504 pub workloads: Vec<WorkloadConfig>,
1505 /// Topology from `.yah/cloud/topology.toml` (mirror→machine assignments).
1506 pub topology: TopologyConfig,
1507}
1508
1509impl CloudConfig {
1510 /// Load all cloud config rooted at `workspace_root` (the parent of `.yah/`).
1511 ///
1512 /// Reads the R215+ tree (`.yah/infra/`, `.yah/services/<svc>/`) eagerly
1513 /// and the pre-R215 `.yah/cloud/` tree opportunistically. Returns `Err`
1514 /// immediately if any TOML fails to parse or a workload TOML fails
1515 /// shape validation; the error includes the file path and field path.
1516 ///
1517 /// Cross-ref validation runs after both trees finish loading: every
1518 /// `mirror.providers.X.use = "<id>"` must resolve to a real provider
1519 /// declared under `.yah/infra/providers/`.
1520 ///
1521 /// R844-B7 — **a missing `.yah/` is a wrong-root error, not an empty
1522 /// fleet.** Every sub-loader below tolerates a missing directory by
1523 /// returning empty, so before this check a call against the wrong
1524 /// directory produced a perfectly valid `CloudConfig` with zero machines,
1525 /// zero services and zero providers. Nothing downstream can tell that
1526 /// apart from a camp that genuinely declares nothing, so the failure
1527 /// surfaces as an operation that silently does nothing to nothing: a
1528 /// collate that renders no backends, a fanout that asks no nodes, a
1529 /// rollout that plans against an empty fleet. It was found the hard way —
1530 /// a live-fleet test in `app/yah/cli` called this with `"."`, which under
1531 /// `cargo test` is the *package* root, and passed while measuring nothing.
1532 ///
1533 /// The line is drawn at `.yah/` and only there: a workspace whose
1534 /// `.yah/infra/machines/` is absent or empty is a real, if unusual, camp
1535 /// with an empty fleet and still loads. `unknown` is not `answered with
1536 /// none`.
1537 pub fn load(workspace_root: &Path) -> Result<Self> {
1538 let yah_dir = crate::paths::yah_dir(workspace_root);
1539 if !yah_dir.is_dir() {
1540 anyhow::bail!(
1541 "not a yah workspace: no {} — expected the camp root (the parent \
1542 of `.yah/`), got {}. This is a wrong-root error, not an empty \
1543 fleet; a camp with no machines declared still has a `.yah/`.",
1544 yah_dir.display(),
1545 workspace_root.display(),
1546 );
1547 }
1548
1549 let mut providers = load_providers(&crate::paths::providers_dir(workspace_root))?;
1550 let services = load_services(&crate::paths::services_dir(workspace_root), workspace_root)?;
1551 let domains = load_domains(&crate::paths::domains_dir(workspace_root))?;
1552
1553 Self::cross_ref_validate(&providers, &services, &domains)?;
1554
1555 // Legacy `.yah/cloud/` reads — empty in post-B1 workspaces. Wrapped in
1556 // a helper so a missing tree is silent (no error, no warning).
1557 let cloud_dir = crate::paths::legacy_cloud_dir(workspace_root);
1558 let (legacy_mirrors, legacy_workloads, topology) = if cloud_dir.exists() {
1559 (
1560 load_mirrors(cloud_dir.join("mirrors"))?,
1561 load_workloads(cloud_dir.join("workloads"))?,
1562 load_topology(cloud_dir.join("topology.toml"))?,
1563 )
1564 } else {
1565 Default::default()
1566 };
1567
1568 // Workloads come from `.yah/infra/workloads/` (R215+). R568-T7: before
1569 // that path was read here, this field was populated *only* from the
1570 // legacy tree above — which R222-B1 emptied — so `cfg.workload(name)`
1571 // resolved nothing in every post-R215 camp and `yah cloud workload
1572 // deploy` could not find any declaration at all. The bug survived
1573 // because the only workloads ever deployed were forge/QED runs, which
1574 // build their spec in memory and never come through here. Same
1575 // dedupe-by-name shape as machines below: R215+ wins.
1576 let mut workloads = load_workloads(crate::paths::workloads_dir(workspace_root))?;
1577 let workload_names: std::collections::HashSet<String> =
1578 workloads.iter().map(|w| w.spec.name.clone()).collect();
1579 for w in legacy_workloads {
1580 if !workload_names.contains(&w.spec.name) {
1581 workloads.push(w);
1582 }
1583 }
1584
1585 // R870-B13: machines are resolved by [`resolve_fleet_inventory`] —
1586 // camp-local, the pre-R215 legacy tree, and every machine borrowed
1587 // through `.yah/infra/sources.toml`, in that precedence. This used to
1588 // be spelled out inline here, which made `CloudConfig::load` the only
1589 // reader that saw borrowed machines at all; the two *resolution*
1590 // callers in `validate`/`reconciler::domain` read a camp-local-only
1591 // loader and could not see a borrowing camp's fleet. There is now one
1592 // implementation and three callers.
1593 let fleet = resolve_fleet_inventory(workspace_root)?;
1594
1595 // Providers overlay here rather than inside `resolve_fleet_inventory`:
1596 // that function answers "which machines does this camp have", which is
1597 // the question with three readers. Providers have exactly one reader —
1598 // this load — so hoisting them would build a seam nothing crosses.
1599 let mut provider_origins = BTreeMap::new();
1600 overlay_source_providers(
1601 workspace_root,
1602 &fleet.sources,
1603 &mut providers,
1604 &mut provider_origins,
1605 );
1606
1607 Ok(Self {
1608 workspace_root: workspace_root.to_path_buf(),
1609 machines: fleet.machines,
1610 providers,
1611 machine_origins: fleet.origins,
1612 provider_origins,
1613 // Local, offline, and absent in most camps — the journal replays to
1614 // an empty map when the file isn't there (R850-T2).
1615 recovery_measurements: crate::recovery_journal::RecoveryJournal::at_workspace(
1616 workspace_root,
1617 )
1618 .replay(),
1619 services,
1620 domains,
1621 legacy_mirrors,
1622 workloads,
1623 topology,
1624 })
1625 }
1626
1627 /// Load the R215+ tree (`infra/`, `services/`, `domains/`) rooted at an
1628 /// arbitrary config directory instead of the hardcoded `.yah/`. This is the
1629 /// building block for multi-root deployments (W206 config layout (b), sibling
1630 /// `.noisetable/` trees) — see [`crate::multi_root`]. Part of R558-F4.
1631 ///
1632 /// `config_dir` is the `.X/` directory itself (e.g. `<parent>/.noisetable`);
1633 /// `workspace_root` remains the camp dir (the config dir's parent) so a
1634 /// component's `path` reference resolves against the same tree the classic
1635 /// [`CloudConfig::load`] uses. The legacy `.yah/cloud/` reads are skipped —
1636 /// multi-root deployments are post-R215 by construction — so `legacy_*`,
1637 /// `workloads`, and `topology` come back empty. Machines are read from
1638 /// `config_dir/infra/machines` directly (sibling trees declare their own
1639 /// inventory or none).
1640 ///
1641 /// R615-F2 decision, explicit rather than silent: **sources.toml overlay
1642 /// does NOT apply here.** This function
1643 /// exists specifically because a multi-root sibling tree (W206 layout
1644 /// (b), e.g. `.noisetable/`) is a *second config root inside the same
1645 /// camp*, not a second camp — `config_dir` is already wherever the
1646 /// caller decided this tree's infra lives, and `.yah/infra/sources.toml`
1647 /// (singular, tied to `paths::infra_dir(workspace_root)`) has no
1648 /// well-defined meaning for an arbitrary `config_dir` that isn't that
1649 /// path. A sibling tree that wants borrowed infra declares its own
1650 /// `sources.toml` under whichever root actually calls
1651 /// [`CloudConfig::load`] for it; `machine_origins`/`provider_origins`
1652 /// come back empty here, not wrong — there is nothing to overlay.
1653 pub fn load_from_config_dir(config_dir: &Path, workspace_root: &Path) -> Result<Self> {
1654 let providers = load_providers(&config_dir.join("infra").join("providers"))?;
1655 let services = load_services(&config_dir.join("services"), workspace_root)?;
1656 let domains = load_domains(&config_dir.join("domains"))?;
1657
1658 Self::cross_ref_validate(&providers, &services, &domains)?;
1659
1660 let machines = load_dir::<MachineConfig>(config_dir.join("infra").join("machines"))?;
1661
1662 Ok(Self {
1663 workspace_root: workspace_root.to_path_buf(),
1664 machines,
1665 providers,
1666 machine_origins: BTreeMap::new(),
1667 provider_origins: BTreeMap::new(),
1668 // Same reasoning as the sources overlay above: the recovery journal
1669 // is tied to `paths::recovery_journal(workspace_root)`, which has no
1670 // meaning for an arbitrary sibling config dir — and this loader
1671 // returns no `workloads` for a measurement to attach to anyway.
1672 recovery_measurements: BTreeMap::new(),
1673 services,
1674 domains,
1675 legacy_mirrors: vec![],
1676 workloads: vec![],
1677 topology: TopologyConfig::default(),
1678 })
1679 }
1680
1681 /// Cross-reference validation shared by [`CloudConfig::load`] and
1682 /// [`CloudConfig::load_from_config_dir`]: every mirror `providers.X.use =
1683 /// "<id>"` must resolve to a declared provider, and every domain route's
1684 /// `component = "<service>/<component-id>"` must resolve to a real component.
1685 fn cross_ref_validate(
1686 providers: &[ProviderConfig],
1687 services: &BTreeMap<String, ServiceWithMirrors>,
1688 domains: &BTreeMap<String, DomainConfig>,
1689 ) -> Result<()> {
1690 // Mirror `use = "<id>"` slots must resolve to a declared provider.
1691 let provider_ids: std::collections::HashSet<&str> =
1692 providers.iter().map(|p| p.id.as_str()).collect();
1693 for (svc_name, svc) in services {
1694 // Addressing is checked here rather than at probe time: a
1695 // truncated NodeId or a relative health path otherwise fails
1696 // with an error naming neither the file nor the field.
1697 svc.service.address.validate(svc_name)?;
1698 for (env, mirror) in &svc.mirrors {
1699 for (slot, body) in &mirror.providers {
1700 if let Some(id) = body.provider_id() {
1701 if !provider_ids.contains(id) {
1702 anyhow::bail!(
1703 "services/{svc_name}/mirrors/{env}.toml: \
1704 providers.{slot}.use = \"{id}\" — no such provider; \
1705 declare it at infra/providers/{id}.toml"
1706 );
1707 }
1708 }
1709 }
1710 // An `[[ingress]]` edge's own `use` is the same kind of
1711 // reference (R845) and gets the same check: a typo there is
1712 // otherwise invisible until `yah cloud apply` reaches the
1713 // Cloudflare arm and fails on a missing provider file.
1714 for (idx, edge) in mirror.ingress_edge_slice().iter().enumerate() {
1715 if let Some(id) = edge.provider_id.as_deref() {
1716 if !provider_ids.contains(id) {
1717 anyhow::bail!(
1718 "services/{svc_name}/mirrors/{env}.toml: \
1719 ingress[{idx}].use = \"{id}\" — no such provider; \
1720 declare it at infra/providers/{id}.toml"
1721 );
1722 }
1723 }
1724 }
1725 // R905. A `[build.<id>]` override keyed by a component this
1726 // service does not declare is always a typo, and it is the
1727 // silent kind: nothing reads the table for a component that
1728 // isn't there, so the environment goes on building with the
1729 // command the operator believed they had replaced — which is
1730 // exactly the defect the override exists to fix, wearing a
1731 // config that looks like the fix.
1732 for key in mirror.build.keys() {
1733 if !svc.service.components.iter().any(|c| &c.id == key) {
1734 let declared: Vec<&str> = svc
1735 .service
1736 .components
1737 .iter()
1738 .map(|c| c.id.as_str())
1739 .collect();
1740 anyhow::bail!(
1741 "services/{svc_name}/mirrors/{env}.toml: \
1742 [build.{key}] — no component {key:?} in \
1743 services/{svc_name}/service.toml (declared: {declared:?})"
1744 );
1745 }
1746 }
1747 }
1748 }
1749
1750 // R870-B11. Two bundle-tier components sharing a mount would stage
1751 // into the same `app/dist/<mount>/` prefix inside the service's one
1752 // assembled bundle and silently clobber each other on disk — the
1753 // exact failure class this ticket exists to fix, one level down
1754 // (there it was two components silently overwriting the same
1755 // *workload*; here it would be two components silently overwriting
1756 // the same *path inside* the workload). A mount is owned by exactly
1757 // one component; refuse the config before the clobber happens.
1758 //
1759 // R870-F23 widens the same loop to the workload tier rather than
1760 // adding a parallel one. A mount is owned by exactly one component
1761 // whichever tier serves it: two workload-tier components at one mount
1762 // would hand the inner door two upstream sets for one prefix, and two
1763 // components in *different* tiers at one mount is the same clobber
1764 // read from the routing side — the request reaches whichever of the
1765 // bundle and the workload the mount table happened to name. So the
1766 // rule is now "one component per mount, service-wide", and only the
1767 // explanation branches on tier.
1768 for (svc_name, svc) in services {
1769 let mut owner_by_mount: BTreeMap<String, (&str, DeployTier)> = BTreeMap::new();
1770 for component in &svc.service.components {
1771 let bundle_tier =
1772 component.kind == "mesofact-static" || component.kind == "mesofact-spa";
1773 if !bundle_tier && component.deploy != DeployTier::Workload {
1774 continue;
1775 }
1776 let mount = component
1777 .mount
1778 .as_deref()
1779 .map(normalize_mount)
1780 .unwrap_or_default();
1781 if let Some((existing, existing_tier)) =
1782 owner_by_mount.insert(mount.clone(), (&component.id, component.deploy))
1783 {
1784 let where_ = if mount.is_empty() {
1785 "the service root (no `mount`)".to_string()
1786 } else {
1787 format!("mount = \"/{mount}\"")
1788 };
1789 let why = if existing_tier == component.deploy {
1790 match component.deploy {
1791 DeployTier::Bundle => {
1792 "a bundle-tier component's mount is a storage prefix inside the \
1793 service's single assembled bundle (app/dist/<mount>/), so two \
1794 components at the same mount would stage into the same path and \
1795 silently overwrite each other"
1796 }
1797 DeployTier::Workload => {
1798 "a workload-tier component's mount is its prefix in the service's \
1799 inner-door route table, so two components at the same mount would \
1800 claim one prefix and requests would reach whichever the table \
1801 named"
1802 }
1803 }
1804 } else {
1805 "one is staged into the service bundle and the other deploys as its own \
1806 workload, so the mount names two different things that serve one prefix \
1807 — the inner door can only route it to one of them"
1808 };
1809 anyhow::bail!(
1810 "services/{svc_name}/service.toml: components \"{existing}\" and \
1811 \"{}\" both declare {where_} — {why}. Give one of them a distinct \
1812 `mount`.",
1813 component.id,
1814 );
1815 }
1816 }
1817 }
1818
1819 // Every domain route's `component = "<service>/<component-id>"` must
1820 // resolve to a real component.
1821 for (dom_name, dom) in domains {
1822 for (idx, route) in dom.routes.iter().enumerate() {
1823 let Some(component_ref) = route.mode.component() else {
1824 continue; // redirects don't reference components
1825 };
1826 let Some((svc_name, comp_id)) = split_component_ref(component_ref) else {
1827 anyhow::bail!(
1828 "domains/{dom_name}.toml: routes[{idx}].component = \
1829 \"{component_ref}\" — expected \"<service>/<component-id>\""
1830 );
1831 };
1832 let Some(svc) = services.get(svc_name) else {
1833 anyhow::bail!(
1834 "domains/{dom_name}.toml: routes[{idx}].component = \
1835 \"{component_ref}\" — no such service \"{svc_name}\" \
1836 under services/"
1837 );
1838 };
1839 let Some(component) = svc.service.components.iter().find(|c| c.id == comp_id)
1840 else {
1841 anyhow::bail!(
1842 "domains/{dom_name}.toml: routes[{idx}].component = \
1843 \"{component_ref}\" — service \"{svc_name}\" has no \
1844 component with id \"{comp_id}\""
1845 );
1846 };
1847
1848 // R746: a mounted component must be routed where it publishes.
1849 // The publisher writes its bundle under the mount and the front
1850 // door looks a request up by its own path, so a route path and
1851 // a mount that disagree produce a 404 with its cause two files
1852 // away. Checked in both directions, since either one alone is
1853 // the same silent miss.
1854 //
1855 // Static routes only: `mount` is a *storage* prefix, and a
1856 // backend route proxies to an origin that owns its own paths.
1857 if !matches!(route.mode, RouteMode::Static { .. }) {
1858 continue;
1859 }
1860 let mount = component.mount.as_deref().map(normalize_mount);
1861 let route_prefix = route_path_prefix(&route.path);
1862 if let Some(mount) = mount {
1863 if mount != route_prefix {
1864 anyhow::bail!(
1865 "domains/{dom_name}.toml: routes[{idx}].path = \
1866 \"{path}\" serves \"{component_ref}\", which \
1867 declares mount = \"/{mount}\" — a mounted \
1868 component publishes under its mount, so the route \
1869 must be \"/{mount}\" or \"/{mount}/*\" (or drop \
1870 the mount to serve from the service root)",
1871 path = route.path,
1872 );
1873 }
1874 } else if !route_prefix.is_empty() {
1875 anyhow::bail!(
1876 "domains/{dom_name}.toml: routes[{idx}].path = \
1877 \"{path}\" serves \"{component_ref}\", which declares \
1878 no `mount` — its bundle publishes at the service root, \
1879 so nothing is stored under \"/{route_prefix}\". Set \
1880 mount = \"/{route_prefix}\" on the component, or route \
1881 it at \"/*\"",
1882 path = route.path,
1883 );
1884 }
1885 }
1886 }
1887 Ok(())
1888 }
1889
1890 /// Look up a domain manifest by name (file stem under `.yah/domains/`).
1891 pub fn domain(&self, name: &str) -> Option<&DomainConfig> {
1892 self.domains.get(name)
1893 }
1894
1895 pub fn machine(&self, name: &str) -> Option<&MachineConfig> {
1896 self.machines.iter().find(|m| m.name == name)
1897 }
1898
1899 /// Look up a provider by id (matches `provider.id`, not the file stem).
1900 pub fn provider(&self, id: &str) -> Option<&ProviderConfig> {
1901 self.providers.iter().find(|p| p.id == id)
1902 }
1903
1904 /// Look up a service by name (matches `service.toml`'s `name` field).
1905 pub fn service(&self, name: &str) -> Option<&ServiceWithMirrors> {
1906 self.services.get(name)
1907 }
1908
1909 /// Look up a legacy mirror by camp name (pre-R215 .yah/cloud/mirrors/).
1910 pub fn legacy_mirror(&self, camp: &str) -> Option<&LegacyMirrorConfig> {
1911 self.legacy_mirrors.iter().find(|m| m.camp == camp)
1912 }
1913
1914 pub fn workload(&self, name: &str) -> Option<&WorkloadConfig> {
1915 self.workloads.iter().find(|w| w.spec.name == name)
1916 }
1917
1918 /// Every machine declaring `sovereign_group == group`, in declaration order.
1919 ///
1920 /// W305/R742-F3. A sovereign group has no file of its own — it exists only
1921 /// as the set of machines that name the same string — so "which boxes are
1922 /// the dev cluster" has to be *derived*, and before this it was not derived
1923 /// anywhere: `yah cloud rollout plan` still takes a hand-listed
1924 /// `--voter us-west-011 --voter us-west-013 …` for a fact the machine TOMLs
1925 /// already state (W314 gap 1).
1926 ///
1927 /// **This is not placement.** Resolving a group to its members is a
1928 /// *lookup*, and it stays outside [`RequiredSpec`] on purpose — see
1929 /// [`MachineConfig::sovereign_group`]. `migrate` calls this to pick the
1930 /// candidate set it then admits a workload against; nothing here filters
1931 /// scheduling, and adding `sovereign_group` to `matches` would still be the
1932 /// category error that doc warns about.
1933 ///
1934 /// An empty result means no machine declares `group`, which is
1935 /// indistinguishable from a typo — callers should say so with
1936 /// [`Self::declared_sovereign_groups`] rather than reporting "no
1937 /// candidates".
1938 pub fn machines_in_group(&self, group: &str) -> Vec<&MachineConfig> {
1939 self.machines
1940 .iter()
1941 .filter(|m| m.sovereign_group.as_deref() == Some(group))
1942 .collect()
1943 }
1944
1945 /// Every distinct `sovereign_group` declared by any machine, sorted.
1946 ///
1947 /// Exists so a bad `--to` names the real vocabulary instead of complaining
1948 /// abstractly — the same fail-loud shape [`taint_effect`]'s legal-key list
1949 /// gives `check_inert_taints`. Standalone machines (`None`) contribute
1950 /// nothing: "in no group" is not a group you can migrate *to*.
1951 pub fn declared_sovereign_groups(&self) -> Vec<&str> {
1952 let mut groups: Vec<&str> = self
1953 .machines
1954 .iter()
1955 .filter_map(|m| m.sovereign_group.as_deref())
1956 .collect();
1957 groups.sort_unstable();
1958 groups.dedup();
1959 groups
1960 }
1961
1962 /// F16 placement v1: the first machine satisfying every hard axis of `req`
1963 /// (region/zone/provider membership + mesh_tags superset). Declaration order
1964 /// in `.yah/infra/machines/` decides ties — deterministic-greedy, no
1965 /// backtracking. A fully-unconstrained `req` matches the first machine.
1966 ///
1967 /// Fails loud with the constraint summary and the candidate machine names
1968 /// when nothing matches, so `yah cloud apply` surfaces *why* placement
1969 /// failed instead of a silent empty set.
1970 pub fn resolve_machine(&self, req: &RequiredSpec) -> Result<&MachineConfig> {
1971 resolve_machine_among(&self.machines, req)
1972 }
1973
1974 /// F16 placement at horizontal scale: the first
1975 /// [`RequiredSpec::replica_count`] machines satisfying every hard axis of
1976 /// `req`, in declaration order (R844-F8).
1977 ///
1978 /// The N-valued form of [`Self::resolve_machine`], which is the N=1 case of
1979 /// this and not a different selector — both land in [`select_matching`].
1980 /// That shared bottom is what makes the deploy resolver
1981 /// (`reconciler::mesofact_bundle::resolve_bundle_machines`, which calls
1982 /// this) and the ingress planner's
1983 /// (`reconciler::ingress::resolve_ingress_placements`, which calls
1984 /// [`resolve_machines_among`] over the same `machines` slice) agree on the
1985 /// same N machines **by construction**. They must agree set-for-set, not
1986 /// merely in count: a front door aimed at nodes the workload was never
1987 /// deployed to renders a *subset* of the backends, which is the failure that
1988 /// looks like it worked.
1989 pub fn resolve_machines(&self, req: &RequiredSpec) -> Result<Vec<&MachineConfig>> {
1990 resolve_machines_among(&self.machines, req)
1991 }
1992
1993 /// F16 placement: first machine whose `mesh_tags` is a superset of
1994 /// `required`. Declaration order in `.yah/infra/machines/` decides ties.
1995 /// Empty `required` matches the first machine; callers should treat
1996 /// empty-required as "no constraint" and skip this lookup.
1997 ///
1998 /// Back-compat thin wrapper over [`CloudConfig::resolve_machine`] for the
1999 /// mesh-tags-only call sites that predate the topology axes.
2000 pub fn resolve_machine_by_mesh_tags(&self, required: &[String]) -> Option<&MachineConfig> {
2001 let req = RequiredSpec {
2002 mesh_tags: required.to_vec(),
2003 ..Default::default()
2004 };
2005 self.resolve_machine(&req).ok()
2006 }
2007
2008 /// Admission: resolve the target machine for a remote [`WorkloadSpec`],
2009 /// honoring the R594 mesh-tag node-selector annotation
2010 /// (`velveteen_exec::remote::NODE_SELECTOR_MESH_TAGS_ANNOTATION` =
2011 /// `yah.node-selector.mesh-tags`, comma-joined).
2012 ///
2013 /// The producer side (`velveteen_exec::remote::build_workload_spec`, R594) writes
2014 /// `TaskLocation::RemoteAny.mesh_tags` — e.g. `[tag:build-worker, arch:x86]`
2015 /// from [`qed::platform::build_worker_mesh_tags`] — into the workload's
2016 /// annotations. This is the consumer: candidates are restricted to machines
2017 /// whose `mesh_tags` are a **superset** of the requested set, so an amd64
2018 /// build lands on the `arch:x86` build-worker (us-west-002) and an arm64
2019 /// build on a `arch:arm` Pi5. Declaration order in `.yah/infra/machines/`
2020 /// breaks ties.
2021 ///
2022 /// An absent or empty annotation means "no mesh-tag constraint" — pre-R594
2023 /// behavior (any node), matching [`RequiredSpec::is_unconstrained`].
2024 ///
2025 /// This is the single admission seam: R572-F5 extends it with the capacity
2026 /// floor (workload request fits node allocatable−committed) and taint
2027 /// repulsion/affinity by enriching [`RequiredSpec::matches`] /
2028 /// [`Self::resolve_machine`]. Do not fork a second selector.
2029 pub fn admit_workload(&self, ws: &WorkloadSpec) -> Result<&MachineConfig> {
2030 self.resolve_machine(&admission_spec(ws, &self.workloads)?)
2031 }
2032
2033 /// Every machine that admits `ws`, in declaration order — the *pool*
2034 /// [`Self::admit_workload`] returns the head of (R605-T14).
2035 ///
2036 /// # Why a pool and not just the winner
2037 ///
2038 /// `tag:build-worker` is a statement that the tagged boxes are
2039 /// **interchangeable**: a build is booked against the tag, not against
2040 /// `us-west-002`. Returning one machine forced every caller to act as if it
2041 /// were booked against a name, and admission has no liveness input — so a
2042 /// tagged box that is asleep won the file-name tie-break and its builds
2043 /// failed rather than landing on the identical box next to it. That is
2044 /// exactly what happened on 2026-09-03 when `us-west-002` regained the tag.
2045 ///
2046 /// The fix is **not** to teach this function about liveness. It stays a pure
2047 /// function of the declared inventory (see `xtask/tests/fleet_build_placement.rs`
2048 /// on why a placement pin that needs the network is a flake). It hands the
2049 /// dispatcher the whole interchangeable set instead, and the dispatcher —
2050 /// which has the network — probes and fails over within it:
2051 /// `app/yah/cli/src/yubaba_client.rs`'s `MeshYubabaClient::deploy`.
2052 ///
2053 /// Order is the declaration order `admit_workload` already used, and callers
2054 /// should preserve it as their preference order rather than load-balancing
2055 /// across it: a retried build wants the node still holding its warm
2056 /// `target/`, which is the same reason [`first_match`] is deliberately
2057 /// first-fit.
2058 ///
2059 /// `Err` — never `Ok(vec![])` — when nothing admits `ws`, carrying the same
2060 /// message [`Self::admit_workload`] would have produced. "No node admits
2061 /// this" and "the pool is empty" are the same failure and must read the same.
2062 pub fn admit_workload_candidates(&self, ws: &WorkloadSpec) -> Result<Vec<&MachineConfig>> {
2063 let req = admission_spec(ws, &self.workloads)?;
2064 let all: Vec<&MachineConfig> = self.machines.iter().collect();
2065 let matched = matching(&all, &req);
2066 if matched.is_empty() {
2067 // Delegate the wording so the two paths cannot drift apart.
2068 return Err(first_match(&all, &req, DECLARED_POOL, EMPTY_DECLARED_POOL)
2069 .expect_err("matching() found nothing, so first_match cannot succeed"));
2070 }
2071 Ok(matched)
2072 }
2073
2074 /// [`Self::admit_workload`] restricted to the machines of one sovereign
2075 /// group (W305/R742-F3, `yah cloud migrate --to <group>`).
2076 ///
2077 /// Same [`RequiredSpec`], same [`RequiredSpec::matches`], same
2078 /// declaration-order tie-break — only the candidate *set* differs. That is
2079 /// the whole reason this is a narrowing of the admission seam rather than a
2080 /// second selector: a workload that cannot be scheduled onto a group's
2081 /// boxes must fail here for exactly the reason it would fail anywhere else,
2082 /// and `no-appliance` on the dev Pis (W305 finding 2) is precisely the case
2083 /// that must not be silently routed around by a migration verb.
2084 ///
2085 /// `Err` when the group has no members *or* when no member admits `ws`; the
2086 /// two are different mistakes, so callers wanting to tell them apart should
2087 /// check [`Self::machines_in_group`] first.
2088 pub fn admit_workload_in_group(
2089 &self,
2090 ws: &WorkloadSpec,
2091 group: &str,
2092 ) -> Result<&MachineConfig> {
2093 let members = self.machines_in_group(group);
2094 let empty_pool = format!(
2095 "(no machine declares sovereign_group = \"{group}\" — declared groups: {})",
2096 match self.declared_sovereign_groups().as_slice() {
2097 [] => "(none)".to_string(),
2098 gs => gs.join(", "),
2099 }
2100 );
2101 first_match(
2102 &members,
2103 &admission_spec(ws, &self.workloads)?,
2104 &format!("machines in sovereign group '{group}'"),
2105 &empty_pool,
2106 )
2107 }
2108}
2109
2110/// **The** placement selector: the first candidate satisfying every axis of
2111/// `req`, declaration order breaking ties, deterministic-greedy with no
2112/// backtracking.
2113///
2114/// Every path that picks a machine goes through here, and the only thing any
2115/// of them varies is *which machines are candidates* — never the predicate.
2116/// [`CloudConfig::resolve_machine`] passes the whole fleet;
2117/// [`CloudConfig::admit_workload_in_group`] passes one sovereign group's
2118/// members. That split is the point: a candidate-set narrowing composes with
2119/// the [`RequiredSpec`] axes for free, whereas expressing the same narrowing
2120/// *as* an axis would put facts like blast radius into a filter they must
2121/// never be in (see [`MachineConfig::sovereign_group`]).
2122///
2123/// So a new placement scope is a new candidate set plus a `pool` label, and a
2124/// new placement *constraint* is a field on [`RequiredSpec`] — those are the
2125/// two extension points, and neither is a second selector. `pool` and
2126/// `empty_pool` exist only so the failure names the set it actually searched;
2127/// a refusal that says "no candidates" without saying *among what* is one the
2128/// operator has to reconstruct by hand.
2129/// F16 placement v1 resolution over an explicit machine list — the
2130/// `.machines`-only half of [`CloudConfig::resolve_machine`], for callers that
2131/// have loaded just the machines tree rather than the whole cross-ref-validated
2132/// config.
2133///
2134/// R772: `resolve_ingress_placements` (`reconciler::ingress`) is the reason
2135/// this is `pub(crate)` rather than staying folded into
2136/// `CloudConfig::resolve_machine` — ingress collation walks every mirror in
2137/// the workspace and has no business hard-failing over an unrelated mirror's
2138/// `providers.X.use = "<id>"` typo, which is what going through
2139/// `CloudConfig::load`'s cross-ref validation would do. "Do not fork a second
2140/// selector" (see the module doc above) still holds: this is the *same*
2141/// [`first_match`], just handed a narrower candidate set than `self.machines`.
2142pub(crate) fn resolve_machine_among<'a>(
2143 machines: &'a [MachineConfig],
2144 req: &RequiredSpec,
2145) -> Result<&'a MachineConfig> {
2146 let all: Vec<&MachineConfig> = machines.iter().collect();
2147 first_match(&all, req, DECLARED_POOL, EMPTY_DECLARED_POOL)
2148}
2149
2150/// R844-F8: [`resolve_machine_among`] widened to the constraint's own replica
2151/// count — the first [`RequiredSpec::replica_count`] matching machines, in the
2152/// same declaration order, from the same candidate slice.
2153///
2154/// **The one entry point both resolvers share.**
2155/// `reconciler::ingress::resolve_ingress_placements` calls this directly and
2156/// `reconciler::mesofact_bundle::resolve_bundle_machines` reaches it through
2157/// [`CloudConfig::resolve_machines`], both over `cfg.machines` — so the ingress
2158/// planner and the deployer cannot pick different subsets. That is a structural
2159/// guarantee, not a tested coincidence, and it has to be: discovery aimed at a
2160/// node the bundle was never placed on publishes a hostname with a dead
2161/// backend behind it, and at scale > 1 the front door still answers from the
2162/// nodes that *did* get it.
2163///
2164/// Determinism is therefore part of correctness here. `machines` arrives in
2165/// file-name order (`load_dir`, pinned by
2166/// `machines_load_in_file_name_order_not_read_dir_order`), and selection is a
2167/// stable prefix of that order — so "the first two matching" is the same two
2168/// on both sides of the same tree.
2169pub(crate) fn resolve_machines_among<'a>(
2170 machines: &'a [MachineConfig],
2171 req: &RequiredSpec,
2172) -> Result<Vec<&'a MachineConfig>> {
2173 let all: Vec<&MachineConfig> = machines.iter().collect();
2174 select_matching(
2175 &all,
2176 req,
2177 req.replica_count(),
2178 DECLARED_POOL,
2179 EMPTY_DECLARED_POOL,
2180 )
2181}
2182
2183const DECLARED_POOL: &str = "declared machines";
2184const EMPTY_DECLARED_POOL: &str = "(no machines declared under .yah/infra/machines/)";
2185
2186fn first_match<'a>(
2187 candidates: &[&'a MachineConfig],
2188 req: &RequiredSpec,
2189 pool: &str,
2190 empty_pool: &str,
2191) -> Result<&'a MachineConfig> {
2192 Ok(select_matching(candidates, req, 1, pool, empty_pool)?
2193 .into_iter()
2194 .next()
2195 .expect("select_matching errors rather than returning short"))
2196}
2197
2198/// The N-selecting core of the placement selector: the first `want` candidates
2199/// satisfying `req`, in candidate order (R844-F8).
2200///
2201/// [`first_match`] is this with `want = 1`, which is why widening a caller to a
2202/// replica count cannot introduce a second selector — the predicate, the
2203/// ordering and the failure vocabulary are all one implementation.
2204///
2205/// **A shortfall is an error.** Matching one machine when two were asked for
2206/// returns `Err` naming both numbers and the pool searched, never a one-element
2207/// vec: a half-placed workload that reports success is worse than a failed
2208/// apply, because the front door then publishes a hostname whose backend set is
2209/// quietly smaller than declared. `want = 0` is the same mistake spelled
2210/// differently and is refused for the same reason.
2211fn select_matching<'a>(
2212 candidates: &[&'a MachineConfig],
2213 req: &RequiredSpec,
2214 want: usize,
2215 pool: &str,
2216 empty_pool: &str,
2217) -> Result<Vec<&'a MachineConfig>> {
2218 let names = || {
2219 if candidates.is_empty() {
2220 empty_pool.to_string()
2221 } else {
2222 candidates
2223 .iter()
2224 .map(|m| m.name.as_str())
2225 .collect::<Vec<_>>()
2226 .join(", ")
2227 }
2228 };
2229
2230 if want == 0 {
2231 anyhow::bail!(
2232 "replicas = 0 places {} on nothing — a placement that deploys to no machine is \
2233 a typo, not a scale-down; remove the slot instead",
2234 req.describe()
2235 );
2236 }
2237
2238 let mut matched = matching(candidates, req);
2239 if matched.len() >= want {
2240 matched.truncate(want);
2241 return Ok(matched);
2242 }
2243
2244 if want == 1 {
2245 anyhow::bail!(
2246 "no candidates matching {} — {pool}: {}",
2247 req.describe(),
2248 names()
2249 );
2250 }
2251 anyhow::bail!(
2252 "only {} of {want} machines match {} — placing fewer than the declared \
2253 `replicas = {want}` would publish a smaller backend set than the mirror asks for; \
2254 {pool}: {}",
2255 matched.len(),
2256 req.describe(),
2257 names()
2258 )
2259}
2260
2261/// The predicate itself, applied to every candidate in order — the one place
2262/// `req.matches` is called on a set.
2263///
2264/// [`select_matching`] takes a prefix of this; [`CloudConfig::admit_workload_candidates`]
2265/// takes all of it. Keeping both on this function is what makes "the pool the
2266/// dispatcher failed over within" and "the machine admission picked" the same
2267/// answer by construction rather than by two filters that happen to agree.
2268fn matching<'a>(candidates: &[&'a MachineConfig], req: &RequiredSpec) -> Vec<&'a MachineConfig> {
2269 candidates
2270 .iter()
2271 .copied()
2272 .filter(|m| req.matches(m))
2273 .collect()
2274}
2275
2276/// The [`RequiredSpec`] a workload is admitted against — the single place the
2277/// axes are derived from a [`WorkloadSpec`].
2278///
2279/// Extracted from [`CloudConfig::admit_workload`] so that
2280/// [`CloudConfig::admit_workload_in_group`] narrows the candidate set without
2281/// restating the axes. Forking that derivation is how the two paths would
2282/// silently disagree about whether a workload fits a node.
2283///
2284/// # It admits a group, not a workload (R860-T4 / W338)
2285///
2286/// The axes come from [`placement_group`] — `ws` plus the transitive closure of
2287/// its `local` requirement edges — because those members are placed together or
2288/// not at all. Capacity is their **sum**, archetype repulsion their **union**,
2289/// and mesh tags their union too. `prefer-local` and `anywhere` edges bind
2290/// nothing: a spec with neither `requires` nor `depends_on` local edges has a
2291/// group of exactly itself and resolves byte-identically to the pre-R860 axes.
2292///
2293/// This is the **only** gate. Node election is CLI-side
2294/// (`MeshYubabaClient::elect_node`, which picks a live member of the pool this
2295/// produces); the yubaba node process accepts whatever it is handed and never
2296/// re-checks placement, so a wrong group here is not caught downstream.
2297///
2298/// @yah:ticket(R860-T4, "Admission: place the transitive closure of `local` edges as one group, not one workload")
2299/// @yah:status(review)
2300/// @yah:phase(P1)
2301/// @yah:at(2026-09-05T18:29:13Z)
2302/// @yah:assignee(agent:bundle-anthropic-ashguard)
2303/// @yah:parent(R860)
2304/// @yah:next("W338 §Placement consequences 1 and 2. `admission_spec()` (config.rs:1974-1998) derives its axes from ONE spec; it must derive them from the group — the transitive closure of `local` requirement edges over `effective_requirements()`. `prefer-local` and `anywhere` edges do NOT bind the group. Three consequences: memory/cpu floor becomes the SUM of the group's requests, not the requirer's alone; `repel_archetype` becomes the union over members (so a group containing an Appliance is repelled by `no-appliance` even if the requirer is a Server); and the group is non-drainable if ANY member is an Appliance, which today is a per-workload check at yubaba/src/lib.rs:3117-3128 and now has to be computed over a set.")
2305/// @yah:verify("cargo test -p cloud --lib config")
2306/// @yah:gotcha("Node election is CLI-side, not cluster-side: `MeshYubabaClient::elect_node` (app/yah/cli/src/yubaba_client.rs:235-268) calls `admit_workload_candidates` (config.rs:1753), picks one node, and POSTs the deploy there. The yubaba node process never decides placement — it accepts whatever it is handed. So group admission has to be right in `config.rs` because there is no second gate downstream to catch it.")
2307/// @arch:see(.yah/docs/working/W338-workload-dependencies-and-appliance-composition.md)
2308/// @yah:depends_on(R860-T1)
2309/// @yah:handoff("ADMISSION NOW PLACES A GROUP, NOT A WORKLOAD. `admission_spec` (oss/yubaba/crates/cloud/src/config.rs:2009) takes `(ws, declared: &[WorkloadConfig])` and derives every axis from `placement_group(ws, declared)` (:2108) — the transitive closure of `local` requirement edges over `effective_requirements()`, traversing `Requirement::provides` where present and resolving by ident against `cfg.workloads` (.yah/infra/workloads/) otherwise, mesh-identity first and workload name second. Capacity is the SUM of the members' `memory_request_mb()` / `resources.cpu_millis` (saturating). Only `local` binds: `prefer-local` and `anywhere` (which every legacy `depends_on` folds into) are skipped, so a spec without local edges has a group of exactly itself and its axes are bit-identical to the pre-R860 derivation.")
2310/// @yah:handoff("REPEL BECAME A SET. `RequiredSpec::repel_archetype: Option<LifecycleArchetype>` is now `repel_archetypes: Vec<LifecycleArchetype>` (config.rs:3878), the union over group members; `matches` (:3971) rejects a node carrying `no-<taint_key()>` for ANY of them, `describe` emits one `not-tainted(...)` part per archetype, `is_unconstrained` tests `is_empty()`. The field is `#[serde(skip)]`, so no wire or schema drift, and grep over app/ crates/ oss/ xtask/ finds no other referent of the old name and no `RequiredSpec { .. }` literal outside config.rs — the rename is contained. `admit_workload` / `admit_workload_candidates` / `admit_workload_in_group` signatures are unchanged; all three now pass `&self.workloads`.")
2311/// @yah:handoff("CYCLE GUARD, AND THE BUG IT TOOK TO GET RIGHT. The walker keeps TWO visited lists: `in_group` (member mesh identities) and `expanded` (requirement idents already resolved). The first version used one list and was silently wrong in the common case — a requirement's ident IS its provider's mesh identity, so marking the ident before resolving made every provider look already-present and `placement_group` returned a group of one. Five of the new tests caught it. If you refactor this, keep the two questions separate.")
2312/// @yah:handoff("ELECT_NODE NEEDS NO CHANGE FOR THIS TICKET — read it (app/yah/cli/src/yubaba_client.rs:235-268). It calls `admit_workload_candidates`, so it now receives a pool already filtered to nodes that can host the WHOLE group, then probes for liveness within it. That is correct for T4 because only the requirer is deployed today. It becomes load-bearing at R860-T6: `supply = \"self\"` provisioning MUST reuse the node URL `elect_node` returned for the requirer and must not re-elect per member — the probe is liveness-sensitive, so a second election can legally return a different member of the same pool and split the group across two nodes.")
2313/// @yah:handoff("DECISIONS THE BRIEF DID NOT COVER, all recorded in doc comments at the site. (1) `mesh_tags` are UNIONED over the group — the axis is already a superset/AND check, so a node that cannot host one member cannot host the group; zero regression risk since nothing in the tree declares `requires` yet. (2) `nodes` (the R833-F8 operator pin) stays REQUIRER-ONLY: it is a membership list, so intersecting two members' pins can yield an empty vec, which the axis reads as no-constraint — the exact inverse of the conflict. (3) `requires_taint` is a single Option: the requirer's wins, else the first member declaring one. Two members demanding DIFFERENT taints is not representable and would be an unplaceable group; widening that axis to a set is a follow-up if a real case appears. (4) An unresolvable `local` ident is SKIPPED, not an error — admission is a pure function of the declared inventory and must not start refusing deploys over a provider a later ticket declares; the cost is that its request does not count toward the floor, which is the exposure `depends_on` has always had.")
2314/// @yah:handoff("DRAINABILITY: placement half landed, node half deliberately NOT touched. `group_is_drainable(members)` (config.rs:2160) is the set-valued predicate W338 §Placement consequences 2 asks for — false as soon as any member is an Appliance — and the `no-appliance` repulsion that follows from it is enforced through `repel_archetypes`. The node-side loop `drain_workloads` (oss/yubaba/crates/yubaba/src/lib.rs, the R572-F4 archetype_registry skip) still decides per workload and knows nothing about requirement edges, so a Server bound to an Appliance by a `local` edge would still be drained alone. Not fixed here for two reasons: that file has three sessions live in it (the brief named them), and the fix needs group edges plumbed to the node process, which is R860-T6's rail rather than a local edit. `yubaba` already depends on `cloud`, so the predicate is directly callable from there when that plumbing exists.")
2315/// @yah:verify("BASELINE recorded before editing, tree anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2: `cargo test -p yah-cloud --lib` (from oss/yubaba) = 1081 passed, 0 failed, 4 ignored, exit 0. AFTER: 1090 passed, 0 failed, 4 ignored, exit 0 — +9, exactly the nine tests added. `cargo check -p yah-cloud --all-targets` exit 0, and `cargo check -p yubaba --all-targets` exit 0 as well (yubaba consumes `cloud`, so it is where the `repel_archetypes` rename would have surfaced). Every exit code echoed explicitly, never inferred from an empty grep.")
2316/// @yah:verify("NEW TESTS (config.rs `mod tests`, R860-T4 section at the end): a_local_edge_binds_the_provider_into_the_placement_group; prefer_local_and_anywhere_edges_do_not_bind_the_group (covers a legacy `depends_on` too); the_group_is_the_transitive_closure_and_traverses_inline_provides; an_ident_cycle_closes_the_group_instead_of_looping_forever; an_unresolvable_local_ident_is_skipped_rather_than_refused; the_capacity_floor_is_the_sum_of_the_group_not_the_requirer_alone (a 300 MiB node refuses two 256 MiB members and the error names memory_mb>=512; a 512 MiB node admits); a_server_requiring_an_appliance_locally_is_repelled_by_no_appliance (same requirer alone still lands on the tainted Pi, so the repulsion provably comes from the edge); a_group_containing_an_appliance_is_not_drainable; a_spec_with_no_local_edges_admits_exactly_as_it_did_before.")
2317/// @yah:gotcha("The camp's `yah build run` rail killed three consecutive verification runs against the shared oss/yubaba/target dir: each ended with only `Blocking waiting for file lock on build directory` in the log and no exit code, after 121s / 720s. The green result above was obtained with `CARGO_TARGET_DIR=/tmp/r860t4-target`, which sidesteps the contended lock at the cost of one cold dep build. Worth reaching for directly when the yubaba target dir is busy rather than burning three cycles discovering it.")
2318/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
2319/// @yah:next("R860-T6 (supply = \"self\"): deploy the group's non-requirer members onto the node `elect_node` already returned for the requirer — do NOT re-elect per member, or a liveness probe can split the group across two nodes. `placement_group` (config.rs:2108) hands you the member specs in traversal order, requirer first.")
2320/// @yah:next("Node-side drain is still per-workload: teach `drain_workloads` (oss/yubaba/crates/yubaba/src/lib.rs) to consult `cloud::config::group_is_drainable` over the requesting workload's placement group once R860-T6 plumbs group membership to the node. Left untouched here on purpose — three sessions were live in that file.")
2321/// @yah:handoff("LEADER RE-VERIFIED (session:69b18855, independent of the courier's self-report). `cargo test -p yah-cloud --lib` from oss/yubaba: 1090 passed / 0 failed / 4 ignored, exit 0, against the courier's recorded 1081/0/4 baseline — +9 = exactly its new tests. Confirmed by content in config.rs: `placement_group` :2120 with the `req.locality != Locality::Local` guard at :2136 (so `prefer-local` and `anywhere` correctly do NOT bind), `group_is_drainable` :2172, and `RequiredSpec::repel_archetype: Option<_>` widened to `repel_archetypes: Vec<_>` at :3890 with the union built at :2039-2050 and enforced at :4004/:4044. The repel rename is `#[serde(skip)]`, so no wire or schema drift.")
2322/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
2323/// @yah:verify("cargo test -p yah-cloud --lib (from oss/yubaba): 1090 passed / 0 failed / 4 ignored, exit 0, vs a 1081/0/4 baseline. Exit codes echoed explicitly throughout rather than inferred from an empty grep — the trap that cost R860-T1 three misses.")
2324/// @yah:gotcha("CORRECTION FROM R860-T6, and the leader propagated the error so it is worth naming: this ticket's handoff asserted \\\"`yubaba` already depends on `cloud`, so the predicate is directly callable from there\\\". THAT IS WRONG. `cloud` is a DEV-dependency of yubaba only — oss/yubaba/crates/yubaba/Cargo.toml:150-152, under the comment \\\"Integration test harness\\\" — and cloud's own Cargo.toml records that the runtime yubaba→cloud edge was DELIBERATELY avoided from R374-F3 onward. The leader repeated the claim verbatim in R860-T6's dispatch brief; T6's courier checked it against the manifest instead of trusting it, which is the only reason it did not become a runtime dependency inversion. Resolution: `group_is_drainable`'s body moved down to `workload_spec::group_is_drainable` (workload-spec/src/lib.rs:2365), the shared home both crates already depend on, and `cloud::config::group_is_drainable` (config.rs:2240) now delegates to it keeping its signature. Verified after the move: yah-cloud still 1093/0/4, yah-workload-spec 171+98/0.")
2325/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
2326/// @yah:verify("RE-VERIFIED AT HEAD 00ee20d1 (session:aa5e882d, 2026-09-05). `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` = 1141 passed / 0 failed / 4 ignored, exit 0 (was 1093 at the first leader's check, 1110 at the second; the deltas are peers' tests). Group placement confirmed by content in oss/yubaba/crates/cloud/src/config.rs: `placement_group` derivation at :2073/:2108, `repel_archetypes: Vec<LifecycleArchetype>` at :3878. NOTE FOR ANYONE RE-RUNNING THIS: `cargo test -p yah-cloud --lib` from the repo root FAILS with \"package `yah-cloud` cannot be tested because it requires dev-dependencies and is not a member of the workspace\" — yah-cloud lives in the oss/yubaba workspace, so the invocation needs `--manifest-path oss/yubaba/Cargo.toml`.")
2327fn admission_spec(ws: &WorkloadSpec, declared: &[WorkloadConfig]) -> Result<RequiredSpec> {
2328 let group = placement_group(ws, declared);
2329
2330 // R894-F1: trust is a declared axis, and the substrate floor it implies is
2331 // a REFUSAL here — not a tag, not a downgrade, not a warning.
2332 //
2333 // Every other axis in this function narrows the candidate set: "which nodes
2334 // can host this". Trust is not that question. A `yah.trust = untrusted`
2335 // spec asking for `yah.exec = native` is not unplaceable — it is
2336 // *incoherent*, and there is no fleet on which it becomes coherent. Making
2337 // it a mesh tag would render it as "no node admits this", which reads like
2338 // a capacity problem and sends the operator to look at machines.
2339 //
2340 // It mirrors `wants_microvm`'s no-silent-downgrade semantics from the other
2341 // direction: kamaji refuses to run a microVM-marked spec as a container
2342 // because that delivers less isolation than was asked for; admission
2343 // refuses to place an untrusted spec on a weaker substrate for exactly the
2344 // same reason, one layer earlier, where the operator can still read why.
2345 //
2346 // Checked per member over the whole `placement_group`, not just `ws`.
2347 // R860-T4 made a `local` edge co-place its provider, so an untrusted member
2348 // reaching a node is reaching it whether or not the requirer is the
2349 // untrusted one — and each member carries its own pair, so the check is
2350 // per-member rather than over a group-wide maximum.
2351 check_trust_substrate(&group)?;
2352
2353 // Capacity is the group's demand, not the requirer's (W338 §Placement
2354 // consequences 1). Saturating rather than wrapping: an absurd declared
2355 // request must read as "nothing is big enough", never as a small number.
2356 //
2357 // `memory_request_mb()` and NOT `resources.memory_mb`: the latter is a
2358 // cgroup ceiling, and reading a ceiling as a floor made `for_forge`'s
2359 // deliberately-roomy 32 GiB limit mean "only place me on a 32 GiB node".
2360 // That excluded every build-worker in the fleet but one. The accessor falls
2361 // back to `resources.memory_mb` when no request is declared, so specs that
2362 // never set one are admitted exactly as before.
2363 let mut memory_mb: u32 = 0;
2364 let mut cpu_millis: u32 = 0;
2365 // R572-F5 taint repulsion, unioned over the group (W338 §Placement
2366 // consequences 2): a group is non-drainable — and `no-appliance`-repelled —
2367 // if *any* member is an Appliance, even when the requirer is a Server.
2368 //
2369 // R876-B7 inverted the sense. Repulsion is now unconditional in `matches`,
2370 // so what this loop collects is still the group's archetype union, but it is
2371 // converted below into the complementary TOLERATION set. Same predicate,
2372 // stated from the other side.
2373 let mut group_archetypes: Vec<LifecycleArchetype> = Vec::new();
2374 // Mesh tags are already AND-ed (a machine must be a superset), so unioning
2375 // them over the group is the same predicate applied to every member: a node
2376 // that cannot host one member cannot host the group.
2377 let mut mesh_tags = node_selector_mesh_tags(ws);
2378
2379 for member in &group {
2380 memory_mb = memory_mb.saturating_add(member.memory_request_mb());
2381 cpu_millis = cpu_millis.saturating_add(member.resources.cpu_millis);
2382 let arch = member.effective_archetype();
2383 if !group_archetypes.contains(&arch) {
2384 group_archetypes.push(arch);
2385 }
2386 for tag in node_selector_mesh_tags(member) {
2387 if !mesh_tags.contains(&tag) {
2388 mesh_tags.push(tag);
2389 }
2390 }
2391 }
2392
2393 // R860-T5 / W338 §Placement consequences 3: per-node native-exec
2394 // capability. Computed over the group for the same reason every other axis
2395 // is — a `local` edge to a native provider makes the *requirer* unplaceable
2396 // on a node without the backend, even when the requirer is an ordinary
2397 // container workload. This is the `supply = "self"` precondition W338 names:
2398 // a self-supplied native provider has to be placeable where its requirer
2399 // lands, and until now nothing upstream could see whether it was.
2400 //
2401 // Appended to `mesh_tags` rather than given its own field: the axis is
2402 // already an AND-ed superset check against `machine.mesh_tags`, `describe`
2403 // already renders it, and `RequiredSpec` needs no new shape. See
2404 // [`NATIVE_EXEC_MESH_TAG`] for why a tag and not a taint.
2405 if group.iter().any(WorkloadSpec::wants_native_exec)
2406 && !mesh_tags.iter().any(|t| t == NATIVE_EXEC_MESH_TAG)
2407 {
2408 mesh_tags.push(NATIVE_EXEC_MESH_TAG.to_string());
2409 }
2410
2411 // R894-F1 / R860-T5's deferred second half: the identical axis for the
2412 // microVM backend. Same `if`, same group-wide reasoning, same reason it is a
2413 // tag and not a taint — see [`MICROVM_MESH_TAG`] for why the precondition
2414 // R860-T5 was waiting on (a node that can actually boot a guest) is now met.
2415 //
2416 // This is what makes the trust floor above land somewhere real: without it,
2417 // an untrusted workload passes the coherence check and is then placed on a
2418 // node whose kamaji has no microVM backend, which refuses at dispatch.
2419 if group.iter().any(WorkloadSpec::wants_microvm)
2420 && !mesh_tags.iter().any(|t| t == MICROVM_MESH_TAG)
2421 {
2422 mesh_tags.push(MICROVM_MESH_TAG.to_string());
2423 }
2424
2425 Ok(RequiredSpec {
2426 mesh_tags,
2427 // R833-F8: imperative node pin. Derived here alongside the inferred
2428 // mesh tags rather than short-circuiting the resolver, so a pinned
2429 // workload is still checked against capacity and taints.
2430 //
2431 // Requirer-only on purpose: the pin is what the operator typed on
2432 // *this* deploy, and `nodes` is a membership list, so intersecting two
2433 // members' pins could yield an empty vec — which this axis reads as "no
2434 // constraint", i.e. the exact opposite of the conflict it represents.
2435 nodes: node_selector_node(ws).into_iter().collect(),
2436 memory_mb,
2437 cpu_millis,
2438 // R876-B7: the archetype union, restated as tolerations — every
2439 // repelling key that is NOT this group's own class. A Server group
2440 // tolerates `no-appliance` and `no-job` and is still blocked by
2441 // `no-server`, which is precisely what the pre-B7 `repel_archetypes`
2442 // axis computed. That equivalence is the migration: the `admit_workload`
2443 // path's placement answers are unchanged for every fleet machine, while
2444 // the mirror-declared path — which could never populate an archetype set
2445 // and so read no taints at all — becomes repel-by-default.
2446 //
2447 // Derived from `LifecycleArchetype::ALL` rather than a literal list, so
2448 // a fourth archetype is tolerated by unrelated groups automatically,
2449 // exactly as `taint_effect` already derives the repulsion half.
2450 tolerates: LifecycleArchetype::ALL
2451 .into_iter()
2452 .filter(|a| !group_archetypes.contains(a))
2453 .map(|a| format!("no-{}", a.taint_key()))
2454 .collect(),
2455 // R572-F5: taint affinity from the requires-taint annotation. The
2456 // requirer's wins; otherwise the first member that declares one, since
2457 // the group shares a node and this axis holds a single key. Two members
2458 // demanding *different* taints is not representable here and would be
2459 // an unplaceable group anyway — see the R860-T4 handoff.
2460 requires_taint: group
2461 .iter()
2462 .find_map(|m| m.requires_taint().map(str::to_owned)),
2463 ..Default::default()
2464 })
2465}
2466
2467/// Refuse any group member whose declared trust exceeds what its requested
2468/// [`workload_spec::ExecSubstrate`] provides (R894-F1).
2469///
2470/// The rule in one line: **a caller may request a stricter substrate than its
2471/// trust level requires, never a looser one.** `Trusted` floors at
2472/// [`workload_spec::ExecSubstrate::Native`] — the bottom of the ordering, i.e. no constraint,
2473/// which is what every workload in the fleet has today — and `Untrusted` floors
2474/// at [`workload_spec::ExecSubstrate::MicroVm`], which is the operator's 2026-09-11 call that
2475/// untrusted code never shares a kernel with the fleet.
2476///
2477/// A malformed [`workload_spec::TRUST_ANNOTATION`] is refused too, rather than
2478/// resolving to either side. See [`workload_spec::TrustDeclError`] for why
2479/// guessing in either direction is worse than a named refusal.
2480///
2481/// # Why here and not in `RequiredSpec::matches`
2482///
2483/// `matches` answers "can this node host this group". Trust coherence is a
2484/// property of the *spec alone* — no node makes an untrusted-and-native spec
2485/// legal — so it belongs on the path in, where it can fail with its own
2486/// sentence. Putting it in the predicate would spend a fleet scan to conclude
2487/// "no candidates", which names the wrong thing.
2488///
2489/// It is sited inside [`admission_spec`] and not in each `admit_*` method for
2490/// the reason that function's own docs give: `admission_spec` is the one place
2491/// the three admission entry points share, so a fourth entry point cannot be
2492/// added that skips this. Returning `Result` from it is what makes that
2493/// structural rather than a convention.
2494fn check_trust_substrate(group: &[WorkloadSpec]) -> Result<()> {
2495 for member in group {
2496 let trust = member.trust().map_err(|e| {
2497 anyhow::anyhow!(
2498 "workload '{}' has an unreadable trust declaration: {e}",
2499 member.name
2500 )
2501 })?;
2502 let floor = trust.minimum_substrate();
2503 let requested = member.exec_substrate();
2504 if requested < floor {
2505 bail!(
2506 "workload '{}' declares {}={} but requests the {} substrate, which is weaker than \
2507 the {} minimum that trust level requires — untrusted code does not share a kernel \
2508 with the fleet, so set {}={} on this spec (a stricter substrate is always allowed, \
2509 a weaker one never is)",
2510 member.name,
2511 workload_spec::TRUST_ANNOTATION,
2512 trust.as_str(),
2513 requested.as_str(),
2514 floor.as_str(),
2515 workload_spec::NATIVE_EXEC_ANNOTATION,
2516 floor.annotation_value().unwrap_or("<container>"),
2517 );
2518 }
2519 }
2520 Ok(())
2521}
2522
2523/// The workloads that must be placed together with `ws`: the transitive closure
2524/// of `local` requirement edges over [`WorkloadSpec::effective_requirements`],
2525/// starting at the requirer (R860-T4 / W338 §"Each member keeps its own mesh
2526/// identity").
2527///
2528/// **Only `local` binds.** `prefer-local` explicitly "never blocks placement"
2529/// (W338's locality table) and `anywhere` is an ordinary service dependency —
2530/// treating either as a co-scheduling constraint would turn every `depends_on`
2531/// in the tree into one, since the legacy field folds in as `anywhere` + `wait`.
2532///
2533/// A group is **not** a new addressable object: every member keeps its own mesh
2534/// identity, spec and healthcheck (W338). This function returns the members'
2535/// specs so admission can take the sum / union over them, and nothing here
2536/// deploys, provisions or tears anything down — `supply = "self"` provisioning
2537/// is R860-T6 and per-node native-exec capability is R860-T5.
2538///
2539/// Two ways a member is reached, in this order:
2540/// - [`Requirement::provides`], the inline spec a `supply = "self"` requirement
2541/// carries;
2542/// - otherwise an ident lookup against `declared` (`.yah/infra/workloads/`),
2543/// matched on mesh identity first and on workload name second, because those
2544/// coincide for every spec in the tree today but the requirement is written in
2545/// the mesh-identity currency.
2546///
2547/// An ident that resolves to neither is **skipped**, not an error: admission is
2548/// a pure function of the declared inventory and must not start failing deploys
2549/// over a provider that a not-yet-written ticket will declare. The cost is that
2550/// its request does not count toward the floor, which is the same exposure
2551/// `depends_on` has always had.
2552///
2553/// **Cycle-guarded.** `validate::check_requires` bounds `provides` *nesting* to
2554/// depth 1 but nothing stops two separately-declared specs from requiring each
2555/// other, and this closure would otherwise not terminate. Each requirement ident
2556/// is resolved at most once and each member joins the group at most once, so a
2557/// cycle simply closes the group.
2558pub fn placement_group(ws: &WorkloadSpec, declared: &[WorkloadConfig]) -> Vec<WorkloadSpec> {
2559 let mut members = vec![ws.clone()];
2560 // Two separate visited sets, because the two questions differ: `in_group`
2561 // stops a workload being added twice, `expanded` stops an ident being
2562 // resolved twice. Folding them into one list makes the ident of a member
2563 // already in the group indistinguishable from the member itself — and since
2564 // a requirement's ident *is* its provider's mesh identity, that reads every
2565 // provider as already-present and silently returns a group of one.
2566 let mut in_group: Vec<String> = vec![group_key(ws)];
2567 let mut expanded: Vec<String> = Vec::new();
2568 let mut next = 0;
2569
2570 while next < members.len() {
2571 let requirements = members[next].effective_requirements();
2572 next += 1;
2573 for req in requirements {
2574 if req.locality != Locality::Local {
2575 continue;
2576 }
2577 if expanded.contains(&req.ident.0) {
2578 continue;
2579 }
2580 expanded.push(req.ident.0.clone());
2581
2582 let provider = match req.provides.as_deref() {
2583 Some(spec) => spec.clone(),
2584 None => match resolve_requirement_ident(&req.ident, declared) {
2585 Some(spec) => spec,
2586 None => continue,
2587 },
2588 };
2589 let key = group_key(&provider);
2590 if in_group.contains(&key) {
2591 continue;
2592 }
2593 in_group.push(key);
2594 members.push(provider);
2595 }
2596 }
2597
2598 members
2599}
2600
2601/// Whether a placement group may be drained off its node (W338 §Placement
2602/// consequences 2): false as soon as **any** member is an Appliance.
2603///
2604/// The set-valued form of the per-workload check the node itself makes in
2605/// `drain_workloads` (`oss/yubaba/crates/yubaba/src/lib.rs`), which skips an
2606/// Appliance by its own archetype and knows nothing about requirement edges. A
2607/// `Server` bound to an Appliance by a `local` edge has to move with it or not
2608/// at all, so draining it alone breaks the group the same way placing it alone
2609/// would.
2610///
2611/// R860-T6 moved the body to [`workload_spec::group_is_drainable`] and left this
2612/// signature untouched. The node's `drain_workloads` needs the identical
2613/// predicate, and yubaba has no runtime dependency on this crate by design
2614/// (R374-F3) — so the one implementation now lives in the crate both sides
2615/// already depend on, rather than being copied into the second caller.
2616pub fn group_is_drainable(members: &[WorkloadSpec]) -> bool {
2617 workload_spec::group_is_drainable(members)
2618}
2619
2620/// Identity a placement-group member is deduplicated by — its mesh identity,
2621/// which is the currency [`Requirement::ident`] is written in.
2622fn group_key(ws: &WorkloadSpec) -> String {
2623 ws.expose.mesh.identity.0.clone()
2624}
2625
2626/// Resolve a requirement's ident to a separately-declared spec: mesh identity
2627/// first, workload file name second.
2628fn resolve_requirement_ident(
2629 ident: &workload_spec::MeshIdent,
2630 declared: &[WorkloadConfig],
2631) -> Option<WorkloadSpec> {
2632 declared
2633 .iter()
2634 .find(|w| w.spec.expose.mesh.identity == *ident)
2635 .or_else(|| declared.iter().find(|w| w.spec.name == ident.0))
2636 .map(|w| w.spec.clone())
2637}
2638
2639/// The mesh tag a node declares to advertise that its kamaji can run **native**
2640/// (fork+exec) workloads — R860-T5 / W338 §"Placement consequences" 3.
2641///
2642/// A workload marked `yah.exec = native` ([`WorkloadSpec::wants_native_exec`])
2643/// is not containerized: kamaji fork+execs it on the node's own userland. That
2644/// backend only exists when the node's kamaji was **built** with the
2645/// `native-exec` cargo feature and **started** with `--native-exec-dir`
2646/// (`oss/kamaji/crates/kamaji-bin/src/main.rs`). Both are node-local startup
2647/// decisions, invisible to everything upstream — so before this tag, placement
2648/// happily elected a node whose kamaji then refused the deploy with
2649/// `BackendRefused: ... no native backend is available (native backend not
2650/// configured — start kamaji with --native-exec-dir)`. That is exactly how the
2651/// mesh lost its coordination server for 25 hours on 2026-09-03 (R858: raft
2652/// leadership moved headscale, a native workload, to `us-south-001`, which has
2653/// no such kamaji). [`admission_spec`] now requires this tag whenever any
2654/// placement-group member is native, which turns that dispatch-time surprise
2655/// into a placement precondition.
2656///
2657/// # Why a mesh tag and not a taint
2658///
2659/// The two vocabularies on [`MachineConfig`] mean opposite things. `mesh_tags`
2660/// are **positive capability** matched as a superset — "this node CAN" — which
2661/// is precisely the claim being made, and an extra tag on a machine can only
2662/// ever make it match *more* requirement sets, so declaring it is regression-
2663/// free. `taints` are **repulsion** — "keep this class off" — and would have to
2664/// be inverted (`no-native-exec` on every node lacking the backend, i.e. the
2665/// declaration burden falls on the majority) *and* taught to
2666/// [`taint_effect`], or [`crate::validate::check_inert_taints`] would correctly
2667/// lint the key dead.
2668///
2669/// # The `cap:` namespace
2670///
2671/// New here. The live prefixes are `tag:` (role — `tag:build-worker`,
2672/// `tag:qed`, `tag:cloud-runner`), `arch:` and `os:` (facts about the silicon
2673/// and userland), and `tier:` is reserved for the environment axis (R763, see
2674/// [`crate::validate::check_retired_arch_tags`]). A *capability the daemon was
2675/// configured with* is none of those: it is not a role an operator assigns and
2676/// not a property of the hardware, it is a fact about how kamaji was started,
2677/// and it changes when the node is rolled. Nothing validates tag prefixes, so
2678/// this costs no wiring.
2679///
2680/// # Fails closed
2681///
2682/// A node that does not declare it is not a candidate. An undeclared fleet
2683/// therefore reports "no node admits" at election time rather than dispatching
2684/// to a node that will refuse — the refusal moves earlier and names the
2685/// constraint, which is the whole point. Declared today (from readings recorded
2686/// in-repo, not inferred) on `us-west-001` and `us-west-003`; see those
2687/// machines' TOMLs for the evidence and the date.
2688///
2689/// # What it does NOT cover — the reading that looks like an inversion
2690///
2691/// This tag gates exactly one backend: [`WorkloadSpec::wants_native_exec`] on a
2692/// `Workload::Container` spec, whose only two in-tree producers are
2693/// `yubaba::headscale_appliance::appliance_spec` and
2694/// `velveteen_exec::remote::mark_native_exec` (forge build steps).
2695///
2696/// kamaji has **other** ways to fork a process onto the host userland, and none
2697/// of them are `yah.exec = native`: a `Workload::MesofactStatic` carrying a
2698/// `serve_bundle` is served by `BundleBackend` (cargo feature `bundle-serving`
2699/// + `--bundle-cache-dir` + `--bundle-origin`), and `Workload::TenantPassway`
2700/// by `kamaji::jit::JitRuntime` (cargo feature `tenant-passway` +
2701/// `--tenant-passway-dir`). Those are separate node-local startup decisions.
2702/// The bundle one is now modelled — see [`BUNDLE_SERVING_MESH_TAG`], R885-T14.
2703/// The JIT one deliberately is **not**; that const's docs say why.
2704///
2705/// The practical consequence, because it has already misled one reader: a node
2706/// can run several kamaji-forked host processes and correctly carry no
2707/// `cap:native-exec`. On 2026-09-11 `us-east-001` ran four (two bundle servers,
2708/// a revalidate receiver, an almanac feed) with no such tag while `us-west-001`
2709/// carried the tag and ran none, which reads as an inversion and is not one:
2710/// none of those four is a native-exec workload, and the tag never claimed
2711/// them.
2712pub const NATIVE_EXEC_MESH_TAG: &str = "cap:native-exec";
2713
2714/// The mesh tag a node declares to advertise that its kamaji can **serve W272
2715/// bundles** — R885-T14, the same axis [`NATIVE_EXEC_MESH_TAG`] models for the
2716/// native-exec backend.
2717///
2718/// A `Workload::MesofactStatic` carrying a `serve_bundle` is materialized and
2719/// supervised by `kamaji_bin::BundleBackend`, which only exists when the node's
2720/// kamaji was **built** with the `bundle-serving` cargo feature and **started**
2721/// with `--bundle-cache-dir` *and* `--bundle-origin` (or `$KAMAJI_BUNDLE_ORIGIN`)
2722/// — `attach_bundle_backend` in `oss/kamaji/crates/kamaji-bin/src/main.rs`
2723/// returns the context untouched without the cache dir, logging
2724/// "Deploy { MesofactStatic + serve_bundle } will refuse with BackendRefused".
2725/// All three are node-local startup decisions, invisible to everything upstream.
2726///
2727/// # The gap this closes
2728///
2729/// Bundle placement does not go through [`admission_spec`] at all — a bundle is
2730/// placed by `reconciler::mesofact_bundle::resolve_bundle_machines`, off the
2731/// mirror's `providers.bundle` declaration (`machines = [...]` or `required =
2732/// { … }`), which the operator writes and which knows nothing about backends.
2733/// So before this tag, a mirror whose constraint matched a node without the
2734/// bundle backend deployed there and was refused at dispatch, exactly the way
2735/// R858 lost headscale for 25 hours on the native axis. It stayed invisible
2736/// because one node (`us-east-001`) serves every bundle in the fleet — the
2737/// condition under which an unmodelled constraint costs nothing right up until
2738/// the second node appears.
2739///
2740/// Both arms of `resolve_bundle_machines` enforce it, including the literal
2741/// `machines = [...]` pin: an operator naming a node by hand is making exactly
2742/// the claim this tag exists to check, and a pin is where the mistake is most
2743/// likely, not least.
2744///
2745/// # Fails closed, like the native tag
2746///
2747/// A node that does not declare it cannot serve a bundle. Declared today on
2748/// `us-east-001` only; see that machine's TOML for the evidence and the date.
2749///
2750/// # Why there is no `cap:tenant-passway` beside this
2751///
2752/// Checked rather than assumed, and it is a real asymmetry.
2753/// `Workload::TenantPassway` has **no placement decision to gate**:
2754/// `yubaba::tenant_passway::reconcile_once` runs *inside the node's own yubaba*
2755/// and drives *that node's own* kamaji over the local UDS. Which node arms a
2756/// domain is decided by which node was started with
2757/// `YUBABA_TENANT_PASSWAY_STATE_DIR`, not by any selector — so a capability tag
2758/// would have no consumer, and R852-B4 records that the tier is enabled on no
2759/// fleet machine today. Declaring it now would be the same wrong fact
2760/// [`NATIVE_EXEC_MESH_TAG`]'s notes refuse to state for `cap:microvm`. Model it
2761/// when a *selector* exists to read it.
2762///
2763/// `Workload::Almanac` needs no tag either, and the reason is stronger: kamaji
2764/// refuses it outright (`server.rs`, "almanac and static-asset live in yubaba's
2765/// reconcilers"), so it is never node-placed. An "almanac feed" observed
2766/// forked on `us-east-001` is a bundle-staged feed fetcher running under
2767/// `BundleBackend`, covered by *this* tag — not a `Workload::Almanac`.
2768pub const BUNDLE_SERVING_MESH_TAG: &str = "cap:bundle-serving";
2769
2770/// The mesh tag a node declares to advertise that its kamaji can **boot a
2771/// Firecracker microVM** — the third instance of the axis
2772/// [`NATIVE_EXEC_MESH_TAG`] and [`BUNDLE_SERVING_MESH_TAG`] model, and the one
2773/// R860-T5 deliberately left open.
2774///
2775/// # Why it was deferred, and why it is no longer
2776///
2777/// R860-T5's `@yah:cleanup` says this tag is "one line from done in the same
2778/// `if` in [`admission_spec`]" and refuses to take it, because **no node in the
2779/// fleet could host a microVM**: the guest kernel and rootfs were gated on
2780/// R605-F14, and declaring a capability nothing has is asserting a false fact.
2781/// That precondition is met. `us-west-003` has had the backend attached since
2782/// 2026-09-10T23:27Z (its TOML records the drop-in, the journal line and the
2783/// staged guest material), and on 2026-09-11 R605-T24's
2784/// `.yah/qed/microvm-dispatch-smoke.toml` drove a forge through the entire
2785/// chain — qed → velveteen-exec → yubaba admission → kamaji → `MicroVmRuntime`
2786/// — asserting on `/proc/cmdline` tokens a container cannot produce.
2787///
2788/// R894-F1 is what made taking it *necessary* rather than merely available: an
2789/// untrusted workload now has [`workload_spec::ExecSubstrate::MicroVm`] as a hard floor, so
2790/// without this axis every untrusted workload would be admitted onto whichever
2791/// node won the tie-break and refused at dispatch — the exact 25-hour-outage
2792/// shape R858 paid for on the native axis.
2793///
2794/// # Fails closed, and the node-side fact is in a drop-in, not ExecStart
2795///
2796/// A node that does not declare it cannot host a microVM workload. Declared
2797/// today on `us-west-003` only.
2798///
2799/// Reading `ExecStart` **cannot** answer whether a node has this backend, and
2800/// that trap is written into `us-west-003`'s own TOML: the backend is enabled by
2801/// `Environment=KAMAJI_MICROVM_DIR=…` in
2802/// `/etc/systemd/system/kamaji.service.d/10-microvm.conf`, so the unit's
2803/// `ExecStart` carries no `--microvm-dir` while `MicroVmRuntime` constructs
2804/// anyway. The same TOML records that rolling the node to a published version
2805/// reverts the staged guest material — so this tag, like the other two, must be
2806/// re-checked after any roll.
2807pub const MICROVM_MESH_TAG: &str = "cap:microvm";
2808
2809/// Parse the R594 mesh-tag node-selector off a workload's annotations into the
2810/// requested tag set. Absent annotation or empty value ⇒ empty vec ("no
2811/// constraint"). Whitespace around each comma-separated tag is trimmed and
2812/// empty segments are dropped, so `"tag:build-worker, arch:x86"` and
2813/// `"tag:build-worker,arch:x86"` parse identically.
2814pub fn node_selector_mesh_tags(ws: &WorkloadSpec) -> Vec<String> {
2815 ws.annotations
2816 .get(velveteen_exec::remote::NODE_SELECTOR_MESH_TAGS_ANNOTATION)
2817 .map(|v| {
2818 v.split(',')
2819 .map(str::trim)
2820 .filter(|s| !s.is_empty())
2821 .map(String::from)
2822 .collect()
2823 })
2824 .unwrap_or_default()
2825}
2826
2827/// Parse the R833-F8 imperative node-selector off a workload's annotations —
2828/// the single machine `name` the operator pinned the run to
2829/// (`--where=node:us-west-003`). Absent or blank ⇒ `None` ("no constraint"),
2830/// which is every workload built before this axis existed.
2831///
2832/// One node, not a list: the annotation exists to express "run it *there*", and
2833/// a comma-joined set would be a worse spelling of the mesh-tag selector that
2834/// already handles "any of these".
2835pub fn node_selector_node(ws: &WorkloadSpec) -> Option<String> {
2836 ws.annotations
2837 .get(velveteen_exec::remote::NODE_SELECTOR_NODE_ANNOTATION)
2838 .map(|v| v.trim())
2839 .filter(|v| !v.is_empty())
2840 .map(String::from)
2841}
2842
2843/// Load every `.yah/infra/providers/*.toml` into a [`ProviderConfig`] list.
2844/// Missing directory → empty list.
2845fn load_providers(dir: &Path) -> Result<Vec<ProviderConfig>> {
2846 if !dir.exists() {
2847 return Ok(vec![]);
2848 }
2849 let mut items = vec![];
2850 let mut entries: Vec<_> = std::fs::read_dir(dir)
2851 .with_context(|| format!("reading {}", dir.display()))?
2852 .filter_map(|e| e.ok())
2853 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
2854 .collect();
2855 entries.sort_by_key(|e| e.file_name());
2856 for entry in entries {
2857 items.push(ProviderConfig::load(&entry.path())?);
2858 }
2859 Ok(items)
2860}
2861
2862/// Map legacy mirror file stems to their canonical tier names.
2863///
2864/// Canonical tiers: `dev` / `pond` / `prod` / `ha`.
2865/// Legacy stems pre-R362: `local` (dev tier), `local-sim` / `sim` (pond tier).
2866/// Legacy stem `cloud` (prod tier) — renamed 2026-09-14: "cloud" named the
2867/// deployment mechanism, not the environment, and every operator-facing
2868/// surface already said "prod" (`yah cloud apply --env prod`, `mirror up ...
2869/// for prod`, the mirror files themselves are `prod.toml`) while only this
2870/// loader's internal key disagreed.
2871/// Both forms are accepted; canonical names are preferred for new files.
2872pub fn canonical_tier(stem: &str) -> &str {
2873 match stem {
2874 "local" => "dev",
2875 "local-sim" | "sim" => "pond",
2876 "cloud" => "prod",
2877 other => other,
2878 }
2879}
2880
2881/// Walk `.yah/services/<svc>/` for every service and its mirrors.
2882/// Missing directory → empty map. Mirror file stems are normalized to canonical
2883/// tier names via [`canonical_tier`] so callers always see `dev/pond/prod/ha`.
2884fn load_services(
2885 dir: &Path,
2886 workspace_root: &Path,
2887) -> Result<BTreeMap<String, ServiceWithMirrors>> {
2888 if !dir.exists() {
2889 return Ok(BTreeMap::new());
2890 }
2891 let mut out = BTreeMap::new();
2892 let mut entries: Vec<_> = std::fs::read_dir(dir)
2893 .with_context(|| format!("reading {}", dir.display()))?
2894 .filter_map(|e| e.ok())
2895 .filter(|e| e.path().is_dir())
2896 .collect();
2897 entries.sort_by_key(|e| e.file_name());
2898
2899 for entry in entries {
2900 let svc_dir = entry.path();
2901 let service_toml = svc_dir.join("service.toml");
2902 if !service_toml.exists() {
2903 // Skip directories without a service.toml — leaves room for
2904 // future siblings (e.g. `secrets/`, `README.md`) without
2905 // triggering false-positive parse errors.
2906 continue;
2907 }
2908 let service = ServiceConfig::load(&service_toml)?;
2909 let mut mirrors = BTreeMap::new();
2910 let mirrors_dir = svc_dir.join("mirrors");
2911 if mirrors_dir.exists() {
2912 let mut menv: Vec<_> = std::fs::read_dir(&mirrors_dir)
2913 .with_context(|| format!("reading {}", mirrors_dir.display()))?
2914 .filter_map(|e| e.ok())
2915 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
2916 .collect();
2917 menv.sort_by_key(|e| e.file_name());
2918 for m in menv {
2919 let path = m.path();
2920 let stem = path
2921 .file_stem()
2922 .and_then(|s| s.to_str())
2923 .unwrap_or("")
2924 .to_string();
2925 let tier = canonical_tier(&stem).to_string();
2926 // Last-write wins if both legacy and canonical forms coexist
2927 // (e.g. local-sim.toml + pond.toml). Sort order ensures the
2928 // canonical file (pond.toml) wins because 'p' > 'l'.
2929 mirrors.insert(tier, MirrorConfig::load(&path)?);
2930 }
2931 }
2932 let mut component_transform_recipes = BTreeMap::new();
2933 for component in &service.components {
2934 if component.kind == "static-asset" {
2935 if let Some(recipe) =
2936 read_component_transform_recipe(workspace_root, &component.path)
2937 {
2938 component_transform_recipes.insert(component.id.clone(), recipe);
2939 }
2940 }
2941 }
2942 let passway_machines = mirrors
2943 .iter()
2944 .filter_map(|(env, m)| m.passway_machines().map(|ms| (env.clone(), ms)))
2945 .collect();
2946 out.insert(
2947 service.name.clone(),
2948 ServiceWithMirrors {
2949 service,
2950 mirrors,
2951 component_transform_recipes,
2952 passway_machines,
2953 },
2954 );
2955 }
2956 Ok(out)
2957}
2958
2959/// Read the first transform recipe name from a component's `workload.toml`.
2960/// Returns `None` when the file is absent or has no `[asset.derive.transform]`
2961/// section. Best-effort — parse failures are silently ignored so a malformed
2962/// workload.toml doesn't abort the entire service catalog load.
2963fn read_component_transform_recipe(workspace_root: &Path, component_path: &str) -> Option<String> {
2964 let workload_path = workspace_root.join(component_path).join("workload.toml");
2965 let text = std::fs::read_to_string(&workload_path).ok()?;
2966 let value: toml::Value = toml::from_str(&text).ok()?;
2967 let assets = value.get("asset")?.as_array()?;
2968 for asset in assets {
2969 if let Some(recipe) = asset
2970 .get("derive")
2971 .and_then(|d| d.get("transform"))
2972 .and_then(|t| t.get("recipe"))
2973 .and_then(|r| r.as_str())
2974 {
2975 return Some(recipe.to_string());
2976 }
2977 }
2978 None
2979}
2980
2981/// Load every `.yah/domains/*.toml` into a [`DomainConfig`] map keyed by
2982/// file stem. Missing directory → empty map.
2983pub(crate) fn load_domains(dir: &Path) -> Result<BTreeMap<String, DomainConfig>> {
2984 if !dir.exists() {
2985 return Ok(BTreeMap::new());
2986 }
2987 let mut out = BTreeMap::new();
2988 let mut entries: Vec<_> = std::fs::read_dir(dir)
2989 .with_context(|| format!("reading {}", dir.display()))?
2990 .filter_map(|e| e.ok())
2991 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
2992 .collect();
2993 entries.sort_by_key(|e| e.file_name());
2994 for entry in entries {
2995 let path = entry.path();
2996 let stem = path
2997 .file_stem()
2998 .and_then(|s| s.to_str())
2999 .unwrap_or("")
3000 .to_string();
3001 let dom = DomainConfig::load(&path)?;
3002 if dom.name != stem {
3003 anyhow::bail!(
3004 "domains/{}.toml: name = \"{}\" must match the file stem",
3005 stem,
3006 dom.name
3007 );
3008 }
3009 out.insert(dom.name.clone(), dom);
3010 }
3011 Ok(out)
3012}
3013
3014/// Load and shape-validate all `*.toml` files in `dir` as [`WorkloadConfig`].
3015fn load_workloads(dir: std::path::PathBuf) -> Result<Vec<WorkloadConfig>> {
3016 if !dir.exists() {
3017 return Ok(vec![]);
3018 }
3019 let mut items = vec![];
3020 let mut entries: Vec<_> = std::fs::read_dir(&dir)
3021 .with_context(|| format!("reading {}", dir.display()))?
3022 .filter_map(|e| e.ok())
3023 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
3024 .collect();
3025 entries.sort_by_key(|e| e.file_name());
3026
3027 for entry in entries {
3028 let path = entry.path();
3029 let path_str = path.display().to_string();
3030 let src =
3031 std::fs::read_to_string(&path).with_context(|| format!("reading {}", path_str))?;
3032 let spec: WorkloadSpec =
3033 toml::from_str(&src).with_context(|| format!("parsing {}", path_str))?;
3034
3035 // R892-B1: refuse a file whose keys the parser silently threw away.
3036 refuse_dropped_keys(&src, &spec, &path_str)?;
3037
3038 // Shape-validate before accepting into the loaded config.
3039 validate::shape(&spec)
3040 .map_err(|e| anyhow::anyhow!("workload {} failed shape validation: {e}", path_str))?;
3041
3042 items.push(WorkloadConfig { spec });
3043 }
3044 Ok(items)
3045}
3046
3047/// Refuse a workload file that declares keys the parser did not keep (R892-B1).
3048///
3049/// `WorkloadSpec` deliberately does **not** carry `deny_unknown_fields`, and
3050/// must not: the JSON leg of the deploy wire relies on an un-rolled node
3051/// ignoring a field it has never heard of, which is what lets a fleet cross a
3052/// schema change one node at a time. That forgiveness is right on the wire and
3053/// wrong in a hand-authored file — there, an ignored key is an operator's
3054/// declared intent evaporating between parse and serialise, with no diagnostic.
3055///
3056/// On 2026-09-11 that cost a production outage: `[resources]
3057/// ephemeral_storage_mb = 256` in noisetable's `noisetable-account.toml` had
3058/// been deleted from `ResourceLimits` by R885-T6, so the CLI read the file, drop
3059/// the value, and sent a spec the (older) node could not parse — after it had
3060/// destroyed the incumbent. The file said the right thing the whole time.
3061///
3062/// The check is a round trip rather than a key whitelist, so it needs no list to
3063/// maintain and catches every renamed, removed or misspelled key at once: parse
3064/// the file, re-serialise the parsed spec, and report any key the source
3065/// declared that the re-serialisation does not carry. Value *representation* may
3066/// legitimately change across that trip (an enum canonicalising its spelling),
3067/// so only missing KEYS are reported, never differing values.
3068fn refuse_dropped_keys(src: &str, spec: &WorkloadSpec, path_str: &str) -> Result<()> {
3069 let declared: toml::Value = match toml::from_str(src) {
3070 Ok(v) => v,
3071 // Unreachable: the caller just parsed this same text into a typed spec.
3072 Err(_) => return Ok(()),
3073 };
3074 let kept = match toml::Value::try_from(spec) {
3075 Ok(v) => v,
3076 // A spec that cannot be re-serialised is a bug in the schema, not in the
3077 // operator's file — say so rather than blaming their config, and let the
3078 // load proceed exactly as it did before this check existed.
3079 Err(e) => {
3080 eprintln!(
3081 "warning: {path_str} could not be checked for silently-dropped keys \
3082 (re-serialising the parsed spec failed: {e})"
3083 );
3084 return Ok(());
3085 }
3086 };
3087
3088 let mut dropped = Vec::new();
3089 collect_dropped_keys(&declared, &kept, "", &mut dropped);
3090
3091 // R896-B5: a key whose field was deleted as inert is ignored, not refused —
3092 // otherwise every field deletion forces a same-day sweep of every camp's
3093 // committed TOML. Say so, so the dead key still gets cleaned up eventually.
3094 dropped.retain(|path| {
3095 let Some(retired) = workload_spec::RETIRED_KEYS.iter().find(|r| r.path == path) else {
3096 return true;
3097 };
3098 eprintln!(
3099 "warning: {path_str} declares `{path}`, retired by {} and ignored; delete it",
3100 retired.retired_by
3101 );
3102 false
3103 });
3104 if dropped.is_empty() {
3105 return Ok(());
3106 }
3107
3108 anyhow::bail!(
3109 "{path_str} declares {} the workload schema does not have, and their values were being \
3110 discarded in silence:\n\
3111 \x20 {}\n\
3112 \n\
3113 Delete them, or correct the spelling. A key that is present in the file and absent \
3114 from the deployed spec is exactly the failure that destroyed a live workload on \
3115 2026-09-11 (R892): the declaration reads as honoured and is not.",
3116 if dropped.len() == 1 { "a key" } else { "keys" },
3117 dropped.join("\n ")
3118 );
3119}
3120
3121/// Recursive half of [`refuse_dropped_keys`] — keys in `declared` with no
3122/// counterpart in `kept`, reported as dotted paths.
3123fn collect_dropped_keys(
3124 declared: &toml::Value,
3125 kept: &toml::Value,
3126 prefix: &str,
3127 out: &mut Vec<String>,
3128) {
3129 match (declared, kept) {
3130 (toml::Value::Table(d), toml::Value::Table(k)) => {
3131 for (key, value) in d {
3132 match k.get(key) {
3133 Some(kept_value) => {
3134 collect_dropped_keys(value, kept_value, &format!("{prefix}{key}."), out)
3135 }
3136 None => out.push(format!("{prefix}{key}")),
3137 }
3138 }
3139 }
3140 // Element-wise only when nothing was added or removed. A length change
3141 // means the serialiser reshaped the list (materialisation appends mounts,
3142 // for one), and pairing across that would report nonsense.
3143 (toml::Value::Array(d), toml::Value::Array(k)) if d.len() == k.len() => {
3144 for (i, (dv, kv)) in d.iter().zip(k).enumerate() {
3145 collect_dropped_keys(dv, kv, &format!("{prefix}{i}."), out);
3146 }
3147 }
3148 _ => {}
3149 }
3150}
3151
3152/// Load all mirror configs from the `mirrors/` directory.
3153///
3154/// Handles two layouts that may coexist:
3155/// - **Folder**: `mirrors/<id>/mirror.toml` — preferred; allows secrets and
3156/// per-mirror overrides to live next to the config file.
3157/// - **Flat**: `mirrors/<id>.toml` — legacy; still supported.
3158///
3159/// Each file is parsed as [`LegacyMirrorConfig`]. A malformed file returns an error
3160/// that includes the file path and the TOML field path + line/column, so the
3161/// caller can surface it to the user directly.
3162fn load_mirrors(dir: std::path::PathBuf) -> Result<Vec<LegacyMirrorConfig>> {
3163 if !dir.exists() {
3164 return Ok(vec![]);
3165 }
3166 let mut mirrors = vec![];
3167 let mut entries: Vec<_> = std::fs::read_dir(&dir)
3168 .with_context(|| format!("reading {}", dir.display()))?
3169 .filter_map(|e| e.ok())
3170 .collect();
3171 entries.sort_by_key(|e| e.file_name());
3172
3173 for entry in entries {
3174 let path = entry.path();
3175 if path.is_dir() {
3176 // Folder layout: mirrors/<id>/mirror.toml
3177 let mirror_toml = path.join("mirror.toml");
3178 if mirror_toml.exists() {
3179 let src = std::fs::read_to_string(&mirror_toml)
3180 .with_context(|| format!("reading {}", mirror_toml.display()))?;
3181 let cfg: LegacyMirrorConfig = toml::from_str(&src)
3182 .with_context(|| format!("parsing {}", mirror_toml.display()))?;
3183 mirrors.push(cfg);
3184 }
3185 } else if path.extension().map_or(false, |e| e == "toml") {
3186 // Flat layout: mirrors/<id>.toml
3187 let src = std::fs::read_to_string(&path)
3188 .with_context(|| format!("reading {}", path.display()))?;
3189 let cfg: LegacyMirrorConfig =
3190 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
3191 mirrors.push(cfg);
3192 }
3193 }
3194 Ok(mirrors)
3195}
3196
3197/// Load `topology.toml` if it exists; return a default (empty) topology otherwise.
3198fn load_topology(path: std::path::PathBuf) -> Result<TopologyConfig> {
3199 if !path.exists() {
3200 return Ok(TopologyConfig::default());
3201 }
3202 let src =
3203 std::fs::read_to_string(&path).with_context(|| format!("reading {}", path.display()))?;
3204 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
3205}
3206
3207/// R555-S1: entries are sorted by file name before parsing, so "declaration
3208/// order in `.yah/infra/machines/` breaks ties" — the contract
3209/// [`CloudConfig::admit_workload`] documents — is actually true. `read_dir`
3210/// yields filesystem order, which is unspecified and differs between APFS and
3211/// a hashed-dir ext4; without the sort, *which* of two equally-matching nodes a
3212/// workload admits to could change when an unrelated file is added to the
3213/// directory. That was latent while each tag set had one match and became
3214/// observable the day us-west-003 joined us-west-002 on
3215/// `[tag:build-worker, arch:x86, os:linux]`. Same sort `load_providers` has
3216/// always done.
3217fn load_dir<T: for<'de> Deserialize<'de>>(dir: std::path::PathBuf) -> Result<Vec<T>> {
3218 if !dir.exists() {
3219 return Ok(vec![]);
3220 }
3221 let mut entries: Vec<_> = std::fs::read_dir(&dir)
3222 .with_context(|| format!("reading {}", dir.display()))?
3223 .collect::<std::io::Result<Vec<_>>>()
3224 .with_context(|| format!("reading {}", dir.display()))?;
3225 entries.sort_by_key(|e| e.file_name());
3226
3227 let mut items = vec![];
3228 for entry in entries {
3229 let path = entry.path();
3230 if path.extension().map_or(false, |e| e == "toml") {
3231 let src = std::fs::read_to_string(&path)
3232 .with_context(|| format!("reading {}", path.display()))?;
3233 let item: T =
3234 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
3235 items.push(item);
3236 }
3237 }
3238 Ok(items)
3239}
3240
3241// ─── New manifest shapes (R222 B2) ───────────────────────────────────────────
3242//
3243// The post-R215 layout splits substrate from service declarations:
3244//
3245// .yah/infra/providers/<id>.toml → ProviderConfig
3246// .yah/services/<svc>/service.toml → ServiceConfig
3247// .yah/services/<svc>/mirrors/<env>.toml → MirrorConfig
3248//
3249// CloudConfig::load still reads the legacy layout — B3 swaps in these types
3250// and removes the Legacy* shapes plus TopologyConfig.
3251
3252/// Tag for the infrastructure provider kind. Drives which fields are valid in
3253/// a [`ProviderConfig`] body or a [`MirrorProviderSlot::Inline`] block.
3254///
3255/// Two flavors:
3256/// - **Account/runtime providers** (`cloudflare`, `hetzner`, `local-container`)
3257/// live as files under `.yah/infra/providers/<id>.toml` and are referenced
3258/// from a mirror via `use = "<id>"`.
3259/// - **Inline-only providers** (`miniflare-native`, `miniflare-container`,
3260/// `minio-container`) declare an operator-local stand-in directly inside a
3261/// mirror via `kind = "..."`. They carry no credentials and have no provider
3262/// file. The container-backed kinds ride on top of whichever
3263/// `local-container` runtime is declared in infra (orbstack/colima/docker);
3264/// the reconciler resolves the runtime at up-time.
3265#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
3266#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3267#[serde(rename_all = "kebab-case")]
3268pub enum Provider {
3269 /// Cloudflare account: R2 buckets, DNS, Workers, Tunnels.
3270 Cloudflare,
3271 /// Hetzner Cloud + Object Storage account.
3272 Hetzner,
3273 /// Vultr cloud VPS — auto-provisioned via the `cloud.vps.*` Envoy
3274 /// (`VultrEnvoy`), the burst/scaling counterpart to Hetzner. Driver-backed.
3275 Vultr,
3276 /// BYO bare/static node (OVH, on-prem, anything we did NOT provision via a
3277 /// cloud API). Brought up over SSH (`stand-up-yubaba.sh` / `yah cloud
3278 /// machine bootstrap`); reach is declared in the machine's `[connect]`
3279 /// block. No create/destroy driver — placement-only.
3280 Static,
3281 /// Dev-tier static surface: miniflare (workerd) running **natively** — no
3282 /// container — in front of whatever the mirror binds to the `s3`
3283 /// capability (W265, R584-F4).
3284 ///
3285 /// The door at every tier is the same compiled Worker bundle
3286 /// (`worker/router.bundle.js`); only the object store underneath differs,
3287 /// and that difference is now declared rather than branched on:
3288 ///
3289 /// ```toml
3290 /// [providers.static]
3291 /// kind = "miniflare-native"
3292 /// port = 4321
3293 ///
3294 /// [drivers.s3]
3295 /// kind = "local-s3-fs"
3296 /// ```
3297 ///
3298 /// This replaces `local-static`, which served a workload's `dist/` off
3299 /// disk and so gave the dev tier a storage interface no other tier had —
3300 /// the fork W265 exists to delete. Inline-only; it carries no credentials
3301 /// (the store is loopback, the Worker runs on this machine).
3302 MiniflareNative,
3303 /// Local container runtime (orbstack/colima/docker). Configured by a
3304 /// provider file under `.yah/infra/providers/` so the discovery hints +
3305 /// runtime override sit in one place.
3306 LocalContainer,
3307 /// Dev-tier compute: the component runs as a kamaji-supervised host
3308 /// process against the operator's real workspace, no container and no
3309 /// build step per edit. Inline-only — it carries no credentials, and
3310 /// "the machine you are sitting at" is not an account to point at.
3311 /// See `reconciler::local_process`.
3312 LocalProcess,
3313 /// Containerized miniflare (workerd subprocess) fronting MinIO — the
3314 /// pond-tier stand-in for a CF Worker + R2 static surface. Inline-only;
3315 /// the reconciler spawns miniflare via the JS runtime and starts a MinIO
3316 /// container on the local-container runtime.
3317 MiniflareContainer,
3318 /// Containerized MinIO providing an S3-compatible API — the pond-tier
3319 /// stand-in for Cloudflare R2. Inline-only; the reconciler spins up the
3320 /// container on the local-container runtime and auto-creates the declared
3321 /// bucket on first up.
3322 MinioContainer,
3323 /// Dev-tier PostgreSQL — a real server speaking real pgwire on loopback,
3324 /// supervised by kamaji as the `yah-pg-dev` workload (W265, R584-F1). No
3325 /// docker daemon: the driver fetches a per-arch PostgreSQL tarball on first
3326 /// run and `initdb`s a cluster under `.yah/infra/state/dev/pg/`.
3327 ///
3328 /// Inline-only — it carries no credentials worth a provider file (the
3329 /// cluster is loopback-bound with a fixed dev password). Declared under
3330 /// [`MirrorConfig::drivers`], not `providers`:
3331 ///
3332 /// ```toml
3333 /// [drivers.pg]
3334 /// kind = "local-pg-dev"
3335 /// ```
3336 LocalPgDev,
3337 /// Dev/pond-tier SMTP — [mailcrab] supervised by kamaji as the
3338 /// `yah-smtp-dev` workload (W265, R584-F2). A real SMTP listener that
3339 /// accepts every message and delivers none of them, plus a web inbox to
3340 /// read what was sent. No docker daemon: the driver fetches the per-arch
3341 /// mailcrab release binary on first run and caches it under
3342 /// `.yah/cache/mailcrab/`.
3343 ///
3344 /// Inline-only — it carries no credentials at all (the listener is
3345 /// loopback-bound and unauthenticated, which is the point: a catcher that
3346 /// refused unauthenticated mail would not catch the mail your app sends).
3347 /// Declared under [`MirrorConfig::drivers`], not `providers`:
3348 ///
3349 /// ```toml
3350 /// [drivers.smtp]
3351 /// kind = "local-mailcrab"
3352 /// ```
3353 ///
3354 /// [mailcrab]: https://github.com/tweedegolf/mailcrab
3355 LocalMailcrab,
3356 /// Dev-tier S3 — a filesystem-backed, path-style S3 surface supervised by
3357 /// kamaji as the `yah-s3-fs` workload (W265, R584-F3). No docker daemon
3358 /// and no download: the driver *is* the server, and objects live under
3359 /// `.yah/infra/state/dev/s3/data/<bucket>/`.
3360 ///
3361 /// Inline-only — the credentials are fixed dev strings on a loopback
3362 /// listener, which is not an account to point a provider file at.
3363 /// Declared under [`MirrorConfig::drivers`], not `providers`:
3364 ///
3365 /// ```toml
3366 /// [drivers.s3]
3367 /// kind = "local-s3-fs"
3368 /// ```
3369 ///
3370 /// Unlike its two siblings the binding is **optional**: the camp brings
3371 /// this driver up for any camp with a dev mirror whether or not a stanza
3372 /// says so, because every static-asset component needs object storage and
3373 /// [`crate::capability::Capability::for_component_kind`] already says as
3374 /// much. Declaring it is documentation, not activation — see
3375 /// `crate::reconciler::s3_driver::camp_needs_s3_driver`.
3376 LocalS3Fs,
3377}
3378
3379/// A provider account/runtime binding from `.yah/infra/providers/<id>.toml`.
3380///
3381/// The `kind` discriminator picks the schema for the remaining fields. Strict
3382/// on `kind` (unknown values are a parse error); permissive on per-kind fields
3383/// (carried as a free-form map so this loader stays stable as new fields land).
3384/// B3/B4 will tighten by introducing typed variants alongside JSON Schema.
3385#[derive(Debug, Clone, Serialize, Deserialize)]
3386#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3387pub struct ProviderConfig {
3388 pub schema_version: u32,
3389 pub id: String,
3390 pub kind: Provider,
3391 /// Reference into the OS keystore for live credentials (e.g.
3392 /// `"keystore://cloudflare/yah"`). `None` for providers that don't need
3393 /// creds (miniflare-native, optionally local-container).
3394 #[serde(default, skip_serializing_if = "Option::is_none")]
3395 pub credentials: Option<String>,
3396 /// Kind-specific fields. Examples:
3397 /// - cloudflare: `default_zone`
3398 /// - hetzner: `default_location`, `default_server_type`, `ssh_keys`
3399 /// - local-container: `runtime`, `discovery`
3400 #[serde(flatten)]
3401 #[cfg_attr(
3402 feature = "json-schema",
3403 schemars(with = "std::collections::BTreeMap<String, serde_json::Value>")
3404 )]
3405 pub fields: BTreeMap<String, toml::Value>,
3406}
3407
3408impl ProviderConfig {
3409 /// Parse a single `providers/<id>.toml` file.
3410 pub fn load(path: &Path) -> Result<Self> {
3411 let src =
3412 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
3413 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
3414 }
3415}
3416
3417/// Where a service answers, and therefore what a liveness probe asks
3418/// (R926-F1).
3419///
3420/// # Why this replaced a bare `domain: String`
3421///
3422/// `domain` used to be required, and two of this camp's services have no
3423/// honest value for it. `yah-cloud` is publish-only and carries
3424/// `unset.yah-cloud.invalid` — RFC 2606's reserved TLD, chosen precisely
3425/// because the field demanded a string and there was none (the R546-S2
3426/// decision, recorded in that file's own header). The push relay is worse
3427/// than awkward: it has no HTTP surface at all. It binds an iroh endpoint,
3428/// serves ALPN `yah/push-relay/1`, and is addressed by hex `NodeId` over
3429/// QUIC — so a `.invalid` domain there would not merely look wrong, it
3430/// would leave the service permanently unprobeable, which is the exact
3431/// hole R926 exists to close.
3432///
3433/// The alternative was a second health mechanism beside `health_path`.
3434/// This repo is below v1.0.0 and its standing rule is to change the one
3435/// mechanism rather than grow a parallel one, so addressing became a sum
3436/// type: a service declares exactly one way to be reached, and the prober
3437/// switches on it. A service cannot accidentally declare both, and the
3438/// enum is what makes that unrepresentable rather than merely validated.
3439///
3440/// # TOML
3441///
3442/// ```toml
3443/// [address]
3444/// kind = "front-door"
3445/// domain = "cloud.mesh.yah.dev"
3446/// health_path = "/key?v=138"
3447/// ```
3448///
3449/// ```toml
3450/// [address]
3451/// kind = "node"
3452/// node_id = "8f2c…" # 64 hex chars
3453/// alpn = "yah/push-relay/1"
3454/// ```
3455///
3456/// Tagged rather than untagged: an operator edits this file by hand, and
3457/// serde's untagged errors name none of the variants it tried.
3458#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3459#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3460#[serde(tag = "kind", rename_all = "kebab-case")]
3461pub enum ServiceAddress {
3462 /// An HTTPS front door on a public or mesh domain. What all ten
3463 /// services registered in this camp today are.
3464 FrontDoor {
3465 domain: String,
3466 /// Path the probe requests, relative to `domain`. `None` means
3467 /// `/`.
3468 ///
3469 /// `/` is the right question for a service whose root serves a
3470 /// site, and the wrong one for an API. An account/RPC origin that
3471 /// versions its surface answers only under its prefix and 404s
3472 /// everything else *on purpose* — so probing the root reported a
3473 /// broken cell for a service that was behaving exactly as
3474 /// designed, which is the failure mode the probe exists to remove
3475 /// rather than add to. Naming the path here makes the probe ask a
3476 /// question the service has agreed to answer.
3477 ///
3478 /// Service-scoped rather than per-component or per-mirror: one
3479 /// domain has one front door and therefore one canonical liveness
3480 /// URL, and that URL is a property of the service's own router —
3481 /// it does not vary by environment, so repeating it per mirror
3482 /// would only let the copies drift.
3483 ///
3484 /// Must be absolute (leading `/`); the loader refuses anything
3485 /// else rather than silently joining it onto the origin.
3486 #[serde(default, skip_serializing_if = "Option::is_none")]
3487 health_path: Option<String>,
3488 },
3489 /// An iroh endpoint: a stable `NodeId` speaking one ALPN. Probed by
3490 /// dialling that ALPN and expecting an answer.
3491 ///
3492 /// There is no path and no port because there is no URL — QUIC
3493 /// multiplexes on the ALPN, and the `NodeId` is the address. A
3494 /// service reached this way is unreachable from a browser tab, so the
3495 /// Services grid links nothing and renders the node id instead.
3496 Node {
3497 /// Hex-encoded `NodeId`, 64 characters. Not parsed here — this
3498 /// crate does not depend on mshr — but the length and alphabet
3499 /// are checked at load, because a truncated paste otherwise fails
3500 /// at first dial with an error naming neither the file nor the
3501 /// field.
3502 node_id: String,
3503 /// The ALPN to negotiate, e.g. `yah/push-relay/1`. Declared
3504 /// rather than inferred from the service name: the protocol is
3505 /// the thing being probed and it is versioned independently of
3506 /// whatever the service is called.
3507 alpn: String,
3508 },
3509}
3510
3511impl ServiceAddress {
3512 /// Shorthand for the common case.
3513 pub fn front_door(domain: impl Into<String>) -> Self {
3514 Self::FrontDoor {
3515 domain: domain.into(),
3516 health_path: None,
3517 }
3518 }
3519
3520 /// A front door with a declared probe path.
3521 pub fn front_door_at(domain: impl Into<String>, health_path: impl Into<String>) -> Self {
3522 Self::FrontDoor {
3523 domain: domain.into(),
3524 health_path: Some(health_path.into()),
3525 }
3526 }
3527
3528 pub fn node(node_id: impl Into<String>, alpn: impl Into<String>) -> Self {
3529 Self::Node {
3530 node_id: node_id.into(),
3531 alpn: alpn.into(),
3532 }
3533 }
3534
3535 /// The domain, for the callers that genuinely need one (DNS, zone
3536 /// selection, a public URL). `None` for a node-addressed service —
3537 /// and those callers must say so rather than substitute a
3538 /// placeholder, which is how `unset.yah-cloud.invalid` happened.
3539 pub fn domain(&self) -> Option<&str> {
3540 match self {
3541 Self::FrontDoor { domain, .. } => Some(domain),
3542 Self::Node { .. } => None,
3543 }
3544 }
3545
3546 pub fn health_path(&self) -> Option<&str> {
3547 match self {
3548 Self::FrontDoor { health_path, .. } => health_path.as_deref(),
3549 Self::Node { .. } => None,
3550 }
3551 }
3552
3553 /// One line for a status table or a log: the domain, or `node:<8 hex
3554 /// prefix>/<alpn>`. Never a placeholder.
3555 pub fn label(&self) -> String {
3556 match self {
3557 Self::FrontDoor { domain, .. } => domain.clone(),
3558 Self::Node { node_id, alpn } => {
3559 let short: String = node_id.chars().take(8).collect();
3560 format!("node:{short}/{alpn}")
3561 }
3562 }
3563 }
3564
3565 /// Reject what would otherwise fail at probe time with an error
3566 /// naming neither the file nor the field.
3567 fn validate(&self, svc_name: &str) -> Result<()> {
3568 match self {
3569 Self::FrontDoor { domain, health_path } => {
3570 if domain.trim().is_empty() {
3571 anyhow::bail!(
3572 "services/{svc_name}/service.toml: address.domain must not be empty"
3573 );
3574 }
3575 // A relative health path would be joined onto the origin as
3576 // if it were absolute by one URL builder and dropped by the
3577 // next, so the probe would silently ask a different question
3578 // than the file reads. Refuse it at load instead.
3579 if let Some(p) = health_path {
3580 if !p.starts_with('/') {
3581 anyhow::bail!(
3582 "services/{svc_name}/service.toml: health_path = \"{p}\" \
3583 must be absolute — write \"/{p}\""
3584 );
3585 }
3586 }
3587 }
3588 Self::Node { node_id, alpn } => {
3589 let id = node_id.trim();
3590 if id.len() != 64 || !id.chars().all(|c| c.is_ascii_hexdigit()) {
3591 anyhow::bail!(
3592 "services/{svc_name}/service.toml: address.node_id = \"{node_id}\" \
3593 is not a NodeId — want 64 hex characters, got {}",
3594 id.len()
3595 );
3596 }
3597 if alpn.trim().is_empty() {
3598 anyhow::bail!(
3599 "services/{svc_name}/service.toml: address.alpn must not be empty — \
3600 a NodeId with no ALPN names a process, not a service"
3601 );
3602 }
3603 }
3604 }
3605 Ok(())
3606 }
3607}
3608
3609/// An operator-facing service declaration from
3610/// `.yah/services/<svc>/service.toml`.
3611///
3612/// A service groups one or more components (a static surface, a containerized
3613/// API, an almanac…) under a single [`address`](ServiceConfig::address).
3614/// Mirrors project the service onto concrete infra; see [`MirrorConfig`].
3615#[derive(Debug, Clone, Serialize, Deserialize)]
3616#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3617pub struct ServiceConfig {
3618 pub schema_version: u32,
3619 pub name: String,
3620 /// One-line operator-facing answer to "what is this and why does the
3621 /// camp run it?", rendered beside the service wherever it is listed.
3622 ///
3623 /// New surface in R926, and it exists because the registry had no
3624 /// place to say what a service IS. A name and a domain identify a
3625 /// service to someone who already knows the fleet; they tell a reader
3626 /// who does not nothing at all, so the knowledge lived in whichever
3627 /// working doc or board annotation last touched the thing. That is
3628 /// exactly the knowledge a health check makes actionable — a red cell
3629 /// is only useful next to what went red.
3630 ///
3631 /// Deliberately not a free-form `[meta]` table: one field with one
3632 /// meaning cannot accumulate a second half-owner, which is the
3633 /// failure the headscale appliance already demonstrated four times
3634 /// over (see this repo's CLAUDE.md on giving a thing ONE owner).
3635 ///
3636 /// Optional, because nine services predate it and a required field
3637 /// would make every one of them fail to load. `None` renders as no
3638 /// description rather than as an empty one.
3639 #[serde(default, skip_serializing_if = "Option::is_none")]
3640 pub description: Option<String>,
3641 /// How this service is reached, and therefore how its liveness is
3642 /// probed. See [`ServiceAddress`].
3643 pub address: ServiceAddress,
3644 #[serde(default, skip_serializing_if = "Vec::is_empty")]
3645 pub components: Vec<ServiceComponent>,
3646 /// Databases this service exposes, grouped by environment (W241). Every
3647 /// entry becomes a data-workbench / `sql_*` catalog id of the shape
3648 /// `<env>:<service>:<name>` (e.g. `pond:scrabcake:main`). Optional and
3649 /// default-empty — services without databases omit the `[db]` table
3650 /// entirely.
3651 #[serde(default, skip_serializing_if = "DbCatalog::is_empty")]
3652 pub db: DbCatalog,
3653}
3654
3655impl ServiceConfig {
3656 /// The service's domain, or `None` when it is node-addressed
3657 /// (R926-F1). Callers that cannot proceed without one must say which
3658 /// service and why — see [`ServiceAddress::domain`].
3659 pub fn domain(&self) -> Option<&str> {
3660 self.address.domain()
3661 }
3662
3663 /// The domain, or an error naming this service and what the caller
3664 /// wanted it for. The shape every `yah cloud apply` reader wants: a
3665 /// node-addressed service is not deployed by the reconcilers at all,
3666 /// so reaching one of them with a `Node` address is a config bug and
3667 /// deserves a sentence, not an `unwrap`.
3668 pub fn require_domain(&self, wanted_for: &str) -> Result<&str> {
3669 self.address.domain().ok_or_else(|| {
3670 anyhow::anyhow!(
3671 "service {} is node-addressed ({}) and has no domain, but {wanted_for} needs one",
3672 self.name,
3673 self.address.label()
3674 )
3675 })
3676 }
3677
3678 pub fn health_path(&self) -> Option<&str> {
3679 self.address.health_path()
3680 }
3681
3682 /// Parse a single `services/<svc>/service.toml` file.
3683 pub fn load(path: &Path) -> Result<Self> {
3684 let src =
3685 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
3686 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
3687 }
3688
3689 /// Persist to `.yah/services/<name>/service.toml`, creating the service
3690 /// directory if needed. Create-or-overwrite — the canonical replacement
3691 /// for the legacy `sites.json` write path. `workspace_root` is the camp
3692 /// dir (the parent of `.yah/`).
3693 pub fn save(&self, workspace_root: &Path) -> Result<()> {
3694 let dir = crate::paths::service_dir(workspace_root, &self.name);
3695 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
3696 let path = crate::paths::service_toml(workspace_root, &self.name);
3697 let s = toml::to_string_pretty(self)
3698 .with_context(|| format!("serializing service {}", self.name))?;
3699 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
3700 }
3701
3702 /// Remove `.yah/services/<name>/` and everything under it (service.toml
3703 /// plus its `mirrors/`). Returns `false` when the directory was already
3704 /// absent, so callers can distinguish "deleted" from "no-op".
3705 pub fn delete(workspace_root: &Path, name: &str) -> Result<bool> {
3706 let dir = crate::paths::service_dir(workspace_root, name);
3707 if !dir.exists() {
3708 return Ok(false);
3709 }
3710 std::fs::remove_dir_all(&dir).with_context(|| format!("removing {}", dir.display()))?;
3711 Ok(true)
3712 }
3713}
3714
3715/// A git source for a component (R561-F1, "BYO git").
3716///
3717/// When a [`ServiceComponent`] sets `git`, the component's code is NOT in this
3718/// workspace — it lives in an external repo that the reconciler shallow-clones
3719/// into a source cache before build (approach A: clone-at-reconcile, so config
3720/// load + validation stay offline). The component's `path` is then interpreted
3721/// relative to `<checkout>/<subdir>` instead of the workspace root.
3722#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3723#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3724pub struct GitSource {
3725 /// Clone URL (https or ssh) of the tenant repo.
3726 pub repo: String,
3727 /// Branch, tag, or commit SHA to check out. Defaults to `"main"`.
3728 #[serde(default = "default_git_ref")]
3729 pub r#ref: String,
3730 /// Optional sub-directory within the repo that the workspace is rooted at
3731 /// (e.g. a monorepo's `site/`). `path` is resolved relative to this.
3732 #[serde(default, skip_serializing_if = "Option::is_none")]
3733 pub subdir: Option<String>,
3734}
3735
3736fn default_git_ref() -> String {
3737 "main".to_string()
3738}
3739
3740/// How to reach an external infra root (R615-F1 / W274, "linked infra
3741/// sources"): a filesystem link to a sibling camp's live tree, or a git
3742/// checkout of an extracted infra repo.
3743#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3744#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3745#[serde(tag = "kind", rename_all = "kebab-case")]
3746pub enum InfraSourceKind {
3747 /// Filesystem link — reads the owner's live tree. The dev-loop shortcut,
3748 /// and the whole story until W274's "infra as its own repo" end-state.
3749 /// `path` is relative to *this* camp's root; infra is read from
3750 /// `<path>/.yah/infra/`.
3751 Path {
3752 path: String,
3753 },
3754 /// Git link — reused verbatim from [`GitSource`] (R561, "BYO git"),
3755 /// lifted here from "a component's code" to "a camp's infra registry."
3756 /// Loading stays offline (W274 §3): `yah infra sync` (R615-T3) is what
3757 /// clones/pulls this into `.yah/cache/infra/<owner>/`; `CloudConfig::load`
3758 /// only ever reads that cache, never the network.
3759 Git(GitSource),
3760}
3761
3762/// Write-gate for a linked [`InfraSource`] (R615-F1 / W274).
3763///
3764/// An enum, not a bool: the two states today are "borrower renders/plans but
3765/// cannot reconcile" and "this camp genuinely co-administers the shared
3766/// root," and a future read-write-with-approval tier is a third variant, not
3767/// a renamed boolean.
3768#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq, Eq)]
3769#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3770#[serde(rename_all = "kebab-case")]
3771pub enum SourceMode {
3772 /// Borrower can render and plan against the linked entries but cannot
3773 /// reconcile/mutate them — the owner remains the single manager. Default:
3774 /// a borrower is opt-in to write access, never opt-out of the safe state.
3775 #[default]
3776 ReadOnly,
3777 /// Escape hatch for a camp that genuinely co-administers a shared root.
3778 Manage,
3779}
3780
3781/// One `[[source]]` entry in `.yah/infra/sources.toml` (R615-F1 / W274) — an
3782/// external infra root this camp borrows machines/providers from.
3783#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3784#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3785pub struct InfraSource {
3786 /// Logical owner name, badged in the Infra tab (e.g. `"yah"`). Distinct
3787 /// from any camp/repo name the `kind` resolves through — this is what an
3788 /// operator sees on a borrowed row, not a path.
3789 pub owner: String,
3790 #[serde(flatten)]
3791 pub kind: InfraSourceKind,
3792 #[serde(default)]
3793 pub mode: SourceMode,
3794 /// Optional filter — name globs or mesh-tag selectors — to borrow a
3795 /// subset of the source root rather than everything it declares. Empty
3796 /// (the default) borrows everything.
3797 #[serde(default)]
3798 pub select: Vec<String>,
3799}
3800
3801fn default_sources_schema_version() -> u32 {
3802 1
3803}
3804
3805/// `.yah/infra/sources.toml` — the ordered list of external infra roots this
3806/// camp borrows from (R615-F1 / W274).
3807#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3808#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3809pub struct SourcesConfig {
3810 #[serde(default = "default_sources_schema_version")]
3811 pub schema_version: u32,
3812 /// `[[source]]` entries, in declaration order — overlay order matters
3813 /// when two linked sources both name the same machine (R615-F2).
3814 #[serde(default, rename = "source")]
3815 pub source: Vec<InfraSource>,
3816}
3817
3818impl Default for SourcesConfig {
3819 fn default() -> Self {
3820 Self {
3821 schema_version: default_sources_schema_version(),
3822 source: Vec::new(),
3823 }
3824 }
3825}
3826
3827impl SourcesConfig {
3828 /// Load `<infra_dir>/sources.toml`. A missing file is not an error —
3829 /// every camp without linked infra has none, which today is every camp —
3830 /// and yields an empty source list rather than `Err`.
3831 pub fn load(infra_dir: &Path) -> Result<Self> {
3832 let path = infra_dir.join("sources.toml");
3833 if !path.exists() {
3834 return Ok(Self::default());
3835 }
3836 let src =
3837 std::fs::read_to_string(&path).with_context(|| format!("reading {}", path.display()))?;
3838 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
3839 }
3840}
3841
3842impl InfraSource {
3843 /// Human-readable descriptor of *which* source this is, for
3844 /// [`InfraOrigin::source`] — distinguishes two linked sources from the
3845 /// same owner. Never includes credentials: `GitSource.repo` is a clone
3846 /// URL (https/ssh), the same thing R561 already treats as safe to log,
3847 /// with any real secret resolved separately via `keystore://` (W274's
3848 /// own precedent).
3849 fn describe(&self) -> String {
3850 match &self.kind {
3851 InfraSourceKind::Path { path } => format!("path:{path}"),
3852 InfraSourceKind::Git(g) => format!("git:{}@{}", g.repo, g.r#ref),
3853 }
3854 }
3855
3856 /// Resolve this source to an infra root directory (R615-F2 / W274 §3).
3857 /// Does no I/O and touches no network: `path` sources read the owner's
3858 /// live tree directly; `git` sources read wherever `yah infra sync`
3859 /// (R615-T3) last synced to, which may not exist yet (an unsynced git
3860 /// source overlays nothing, not an error — see [`load_dir_tolerant`]).
3861 ///
3862 /// `git.subdir` (reused verbatim from [`GitSource`]/R561) is honoured
3863 /// exactly like the component case: the checkout root when unset, or
3864 /// `<checkout>/<subdir>` when set — e.g. `subdir = "infra"` for a
3865 /// monorepo whose infra registry lives under `infra/` rather than at the
3866 /// clone's root. `yah infra sync` (R615-T3) clones into the *checkout*
3867 /// root ([`crate::paths::infra_source_cache_dir`]), never into a
3868 /// subdir-suffixed path, so this is the one place that appends `subdir`.
3869 fn infra_root(&self, workspace_root: &Path) -> std::path::PathBuf {
3870 match &self.kind {
3871 InfraSourceKind::Path { path } => workspace_root.join(path).join(".yah").join("infra"),
3872 InfraSourceKind::Git(g) => {
3873 let checkout = crate::paths::infra_source_cache_dir(workspace_root, &self.owner);
3874 match g.subdir.as_deref() {
3875 Some(subdir) => checkout.join(subdir),
3876 None => checkout,
3877 }
3878 }
3879 }
3880 }
3881}
3882
3883/// Provenance for a [`MachineConfig`] or [`ProviderConfig`] pulled in from a
3884/// linked `.yah/infra/sources.toml` entry, rather than declared in this
3885/// camp's own `.yah/infra/` (R615-F2 / W274).
3886///
3887/// Lives in [`CloudConfig::machine_origins`] / `provider_origins`, keyed by
3888/// name/id, rather than as a field on `MachineConfig`/`ProviderConfig`
3889/// themselves: those two types are constructed by struct literal in test
3890/// helpers across several crates (including ones this ticket has no reason to
3891/// touch), so widening either shape would ripple out past this crate for no
3892/// semantic gain — origin is a property of *this load*, not an inherent
3893/// property of the machine/provider. A name absent from the map is
3894/// camp-local; present means borrowed, and the Infra tab (R615-F4) / reconcile
3895/// gating (`InfraSource::mode`, copied onto `mode` below) read it from here.
3896#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3897#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3898pub struct InfraOrigin {
3899 /// The [`InfraSource::owner`] that supplied this entry, e.g. `"yah"`.
3900 pub owner: String,
3901 /// Which source, rendered — see [`InfraSource::describe`].
3902 pub source: String,
3903 /// The write-gate that applied when this entry was overlaid — copied
3904 /// from [`InfraSource::mode`] so a caller holding just the machine/
3905 /// provider doesn't need the source list in hand to know it's borrowed
3906 /// read-only.
3907 pub mode: SourceMode,
3908}
3909
3910/// Like [`load_dir`], but tolerant **per file**: a foreign infra root (an
3911/// owner's live tree, or a synced git checkout) can carry entries this
3912/// binary's `T` predates — noisetable's pre-migration machines used an older
3913/// schema than yah's, and the reverse will happen too as each side evolves
3914/// independently. One unparseable file on a source this camp doesn't own must
3915/// never sink every other entry in the same directory, let alone this camp's
3916/// own load (R615-F2 gotcha). Contrast [`load_dir`], which stays strict for
3917/// camp-local files, where a malformed TOML genuinely should be a hard error.
3918///
3919/// Returns the entries that parsed, plus `(path, error)` for every file that
3920/// didn't — the caller logs those, it doesn't drop them silently. A missing
3921/// or unreadable directory yields `(vec![], vec![])`, same "no entries" as
3922/// `load_dir`'s `!dir.exists()` case (an unsynced git source, or a source
3923/// root with no `providers/` at all, are both normal, not warnings).
3924fn load_dir_tolerant<T: for<'de> Deserialize<'de>>(
3925 dir: &Path,
3926) -> (Vec<T>, Vec<(std::path::PathBuf, anyhow::Error)>) {
3927 let Ok(read_dir) = std::fs::read_dir(dir) else {
3928 return (Vec::new(), Vec::new());
3929 };
3930 let mut entries: Vec<_> = read_dir.filter_map(|e| e.ok()).collect();
3931 entries.sort_by_key(|e| e.file_name());
3932
3933 let mut items = Vec::new();
3934 let mut skipped = Vec::new();
3935 for entry in entries {
3936 let path = entry.path();
3937 if path.extension().map_or(true, |e| e != "toml") {
3938 continue;
3939 }
3940 let parsed = std::fs::read_to_string(&path)
3941 .with_context(|| format!("reading {}", path.display()))
3942 .and_then(|src| {
3943 toml::from_str::<T>(&src).with_context(|| format!("parsing {}", path.display()))
3944 });
3945 match parsed {
3946 Ok(item) => items.push(item),
3947 Err(e) => skipped.push((path, e)),
3948 }
3949 }
3950 (items, skipped)
3951}
3952
3953/// Whether a borrowed machine passes an [`InfraSource::select`] filter
3954/// (R615-F2 / W274). Empty `select` borrows everything. A non-empty `select`
3955/// entry matches either the machine's exact `name` or literal membership in
3956/// its `mesh_tags` — the one shape W274's own example uses
3957/// (`select = ["tag:cloud-runner"]`). Not a glob engine: mesh tags are
3958/// already flat strings compared for exact equality everywhere else in this
3959/// crate (see `resolve_machine_by_mesh_tags`), so a select entry is that same
3960/// comparison, not a new pattern language.
3961fn machine_matches_select(machine: &MachineConfig, select: &[String]) -> bool {
3962 select.is_empty()
3963 || select
3964 .iter()
3965 .any(|s| *s == machine.name || machine.mesh_tags.contains(s))
3966}
3967
3968/// What one `[[source]]` in `.yah/infra/sources.toml` actually contributed to
3969/// [`FleetInventory`] on this load (R870-B13).
3970///
3971/// Recorded because a link that resolves to *nothing* is indistinguishable, at
3972/// every downstream use site, from a camp that declared no link at all — and
3973/// that is precisely the failure this ticket exists to fix. A source that
3974/// contributes zero machines is not an error here (an unsynced `kind = "git"`
3975/// source is legitimately empty, and `load()` must stay offline), so instead
3976/// the fact is *carried* to whoever fails for want of a machine. See
3977/// [`FleetInventory::describe_sources`].
3978#[derive(Debug, Clone, PartialEq, Eq)]
3979pub struct SourceContribution {
3980 /// [`InfraSource::owner`] — the name this camp knows the fleet by.
3981 pub owner: String,
3982 /// The source, rendered — see [`InfraSource::describe`].
3983 pub source: String,
3984 /// Where the link resolved to, i.e. the foreign `.yah/infra/`.
3985 pub root: std::path::PathBuf,
3986 /// Whether `root` exists on disk. `false` for a `kind = "path"` link
3987 /// aimed at a directory that is not a camp, and for a `kind = "git"`
3988 /// source that `yah infra sync` has never fetched.
3989 pub root_exists: bool,
3990 /// How many machines this source actually added to the inventory — after
3991 /// [`InfraSource::select`] filtering and after losing every name a
3992 /// camp-local entry or an earlier source already claimed.
3993 pub machines: usize,
3994}
3995
3996/// A camp's resolved machine inventory: **the** answer to "which machines does
3997/// this camp have", with exactly one implementation
3998/// ([`resolve_fleet_inventory`]) behind it (R870-B13).
3999///
4000/// A borrowing camp — one whose own `.yah/infra/machines/` is empty and which
4001/// declares `[[source]]` links to another camp's fleet in
4002/// `.yah/infra/sources.toml` — is the case this type exists for. Before it,
4003/// the overlay was applied inline inside [`CloudConfig::load`], so the two
4004/// callers that resolve a *machine name to a machine* (ingress collation and
4005/// the sovereign apex render) read a camp-local-only loader and saw an empty
4006/// fleet. There was no bug in either of them; the inventory simply had two
4007/// readers that disagreed about what the inventory was.
4008#[derive(Debug)]
4009pub struct FleetInventory {
4010 /// Camp-local machines first, then each source's contribution in
4011 /// declaration order. Camp-local wins any name collision; among sources,
4012 /// the earlier-declared one wins.
4013 pub machines: Vec<MachineConfig>,
4014 /// Provenance for the borrowed entries, keyed by [`MachineConfig::name`].
4015 /// A name absent here is camp-local. Same shape and meaning as
4016 /// [`CloudConfig::machine_origins`], which is populated from this.
4017 pub origins: BTreeMap<String, InfraOrigin>,
4018 /// The parsed `.yah/infra/sources.toml`, kept so a caller that already has
4019 /// an inventory in hand does not re-read it (`CloudConfig::load` overlays
4020 /// providers from the same list).
4021 pub sources: SourcesConfig,
4022 /// Per-source accounting — see [`SourceContribution`].
4023 pub contributions: Vec<SourceContribution>,
4024}
4025
4026impl FleetInventory {
4027 /// One line per declared `[[source]]`, for attaching to the error a caller
4028 /// raises when a machine name does not resolve (R870-B13).
4029 ///
4030 /// The failure being diagnosed is always "I was told about machine X and
4031 /// cannot find it", and the three ways a borrowing camp gets there — no
4032 /// link declared, a link pointing somewhere that is not a camp, a link
4033 /// whose `select` filtered X out — are indistinguishable from the name
4034 /// alone. Empty string when the camp declares no sources, so the caller
4035 /// can append it unconditionally without emitting a dangling header.
4036 pub fn describe_sources(&self) -> String {
4037 if self.contributions.is_empty() {
4038 return String::new();
4039 }
4040 let mut out = String::from("linked infra sources consulted:");
4041 for c in &self.contributions {
4042 out.push_str(&format!(
4043 "\n {} ({}) -> {}{} — contributed {} machine(s)",
4044 c.owner,
4045 c.source,
4046 c.root.display(),
4047 if c.root_exists {
4048 ""
4049 } else {
4050 " [ABSENT: not a camp, or an unsynced git source]"
4051 },
4052 c.machines,
4053 ));
4054 }
4055 out
4056 }
4057}
4058
4059/// Resolve a camp's machine inventory: camp-local `.yah/infra/machines/`, the
4060/// pre-R215 `.yah/cloud/machines/` tree, then every machine borrowed through
4061/// `.yah/infra/sources.toml` (R870-B13, on R615-F2's mechanism).
4062///
4063/// **How a camp names another camp's fleet**, decided here rather than
4064/// invented: through the `[[source]]` entry R615-F1 already defines — `owner`
4065/// is the logical name an operator sees, `kind = "path"` resolves against the
4066/// borrowing camp's own root and `kind = "git"` against `yah infra sync`'s
4067/// cache. There is deliberately no second naming scheme: a camp that could
4068/// name a foreign fleet two ways would be a camp whose inventory can drift
4069/// from itself, which is the thing this ticket rejected.
4070///
4071/// **There is exactly one copy.** A `kind = "path"` source reads the owner's
4072/// live tree at `<path>/.yah/infra/` on every load — the borrowing camp
4073/// persists nothing, so the two can never disagree. `kind = "git"` reads a
4074/// synced checkout, which *is* a copy, but an explicit one with a named
4075/// refresh verb (`yah infra sync`) and a pinned `ref`; that is the cache with
4076/// an invalidation story, as against a hand-maintained second inventory.
4077///
4078/// Camp-local files are strict (a malformed TOML this camp owns is a hard
4079/// error) and foreign files are tolerant per-file (R615-F2: a foreign entry
4080/// whose schema this binary predates must not sink the load). A foreign
4081/// machine skipped that way is not silently lost — it fails loudly at the
4082/// point some caller needs it, with [`FleetInventory::describe_sources`]
4083/// naming the link it should have come from.
4084///
4085/// Deliberately *without* [`CloudConfig::load`]'s R844-B7 wrong-root guard: a
4086/// missing `.yah/infra/machines/` is an empty inventory here, because the
4087/// callers that resolve against it (ingress collation, apex render) are handed
4088/// a root that a `CloudConfig::load` already accepted.
4089pub fn resolve_fleet_inventory(workspace_root: &Path) -> Result<FleetInventory> {
4090 let mut machines = load_dir::<MachineConfig>(crate::paths::machines_dir(workspace_root))?;
4091
4092 // Pre-R215 `.yah/cloud/machines/`. Shouldn't have anything since R215-B1
4093 // moved them, but if it does we dedupe by name — R215+ wins.
4094 let cloud_dir = crate::paths::legacy_cloud_dir(workspace_root);
4095 if cloud_dir.exists() {
4096 let names: std::collections::HashSet<String> =
4097 machines.iter().map(|m| m.name.clone()).collect();
4098 for m in load_dir::<MachineConfig>(cloud_dir.join("machines"))? {
4099 if !names.contains(&m.name) {
4100 machines.push(m);
4101 }
4102 }
4103 }
4104
4105 // `SourcesConfig::load` never touches the network — git sources are read
4106 // from `yah infra sync`'s cache (R615-T3) — so this keeps the whole
4107 // offline contract `CloudConfig::load` has always had.
4108 let sources = SourcesConfig::load(&crate::paths::infra_dir(workspace_root))?;
4109 let mut origins = BTreeMap::new();
4110 let contributions = overlay_source_machines(workspace_root, &sources, &mut machines, &mut origins);
4111
4112 Ok(FleetInventory {
4113 machines,
4114 origins,
4115 sources,
4116 contributions,
4117 })
4118}
4119
4120/// Overlay every linked `.yah/infra/sources.toml` source's machines into
4121/// `machines`, recording provenance into `machine_origins` (R615-F2 / W274).
4122/// Must be called AFTER camp-local entries are already in the vector:
4123/// collision resolution is "first writer wins," so seeding with camp-local
4124/// first is what makes camp-local win over every source, and an earlier source
4125/// win over a later one.
4126///
4127/// `select` filters which machines a source contributes. Returns one
4128/// [`SourceContribution`] per declared source, in declaration order.
4129fn overlay_source_machines(
4130 workspace_root: &Path,
4131 sources: &SourcesConfig,
4132 machines: &mut Vec<MachineConfig>,
4133 machine_origins: &mut BTreeMap<String, InfraOrigin>,
4134) -> Vec<SourceContribution> {
4135 let mut seen_machine_names: std::collections::HashSet<String> =
4136 machines.iter().map(|m| m.name.clone()).collect();
4137 let mut contributions = Vec::with_capacity(sources.source.len());
4138
4139 for source in &sources.source {
4140 let root = source.infra_root(workspace_root);
4141 let origin = InfraOrigin {
4142 owner: source.owner.clone(),
4143 source: source.describe(),
4144 mode: source.mode,
4145 };
4146
4147 let (foreign_machines, skipped) = load_dir_tolerant::<MachineConfig>(&root.join("machines"));
4148 for (path, e) in skipped {
4149 tracing::warn!(
4150 "infra source {:?} ({}): skipping unparseable machine {}: {e:#}",
4151 source.owner,
4152 root.display(),
4153 path.display()
4154 );
4155 }
4156 let mut added = 0usize;
4157 for m in foreign_machines {
4158 if seen_machine_names.contains(&m.name) {
4159 continue; // camp-local, or an earlier source, already claimed this name
4160 }
4161 if !machine_matches_select(&m, &source.select) {
4162 continue;
4163 }
4164 seen_machine_names.insert(m.name.clone());
4165 machine_origins.insert(m.name.clone(), origin.clone());
4166 machines.push(m);
4167 added += 1;
4168 }
4169
4170 contributions.push(SourceContribution {
4171 owner: source.owner.clone(),
4172 source: source.describe(),
4173 root_exists: root.is_dir(),
4174 root,
4175 machines: added,
4176 });
4177 }
4178
4179 contributions
4180}
4181
4182/// Overlay every linked source's providers into `providers`, recording
4183/// provenance into `provider_origins` (R615-F2 / W274). Same first-writer-wins
4184/// rule as [`overlay_source_machines`], and the same requirement that
4185/// camp-local entries already be in the vector.
4186///
4187/// [`InfraSource::select`] deliberately does not apply: nothing in W274 or
4188/// R615-F1 describes a provider-scoped filter — every provider a source
4189/// declares either overlays whole or, on an id collision, doesn't.
4190fn overlay_source_providers(
4191 workspace_root: &Path,
4192 sources: &SourcesConfig,
4193 providers: &mut Vec<ProviderConfig>,
4194 provider_origins: &mut BTreeMap<String, InfraOrigin>,
4195) {
4196 let mut seen_provider_ids: std::collections::HashSet<String> =
4197 providers.iter().map(|p| p.id.clone()).collect();
4198
4199 for source in &sources.source {
4200 let root = source.infra_root(workspace_root);
4201 let origin = InfraOrigin {
4202 owner: source.owner.clone(),
4203 source: source.describe(),
4204 mode: source.mode,
4205 };
4206
4207 let (foreign_providers, skipped) =
4208 load_dir_tolerant::<ProviderConfig>(&root.join("providers"));
4209 for (path, e) in skipped {
4210 tracing::warn!(
4211 "infra source {:?} ({}): skipping unparseable provider {}: {e:#}",
4212 source.owner,
4213 root.display(),
4214 path.display()
4215 );
4216 }
4217 for p in foreign_providers {
4218 if seen_provider_ids.contains(&p.id) {
4219 continue;
4220 }
4221 seen_provider_ids.insert(p.id.clone());
4222 provider_origins.insert(p.id.clone(), origin.clone());
4223 providers.push(p);
4224 }
4225 }
4226}
4227
4228/// One component of a [`ServiceConfig`]. The `kind` (e.g. `"mesofact-static"`,
4229/// `"almanac"`, `"container"`) selects which reconciler runs against the
4230/// pointed-at workload manifest.
4231#[derive(Debug, Clone, Serialize, Deserialize)]
4232#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4233pub struct ServiceComponent {
4234 pub id: String,
4235 pub kind: String,
4236 /// Path of the directory holding this component's `workload.toml`. Relative
4237 /// to the workspace root for in-tree components, or to the materialized
4238 /// `<checkout>/<subdir>` when [`git`](Self::git) is set.
4239 pub path: String,
4240 /// Optional external git source (R561-F1). When set, the component's code
4241 /// is materialized by shallow-clone before build; see [`GitSource`].
4242 #[serde(default, skip_serializing_if = "Option::is_none")]
4243 pub git: Option<GitSource>,
4244 /// Operator-facing role label, e.g. `"static"`, `"dynamic"`, `"compute"`.
4245 pub role: String,
4246 /// Optional artifact kind this component publishes (`"static"`,
4247 /// `"container-image"`, …). Drives mirror provider-slot routing.
4248 #[serde(default, skip_serializing_if = "Option::is_none")]
4249 pub publishes: Option<String>,
4250 /// URL sub-path a static component's build output is published under,
4251 /// relative to the service's publish prefix (R746). `None` = the service
4252 /// root, which is what every pre-R746 component means.
4253 ///
4254 /// Static publishers lay a component's `out_dir` down at
4255 /// `<bucket>/<service>/<env>/…` and the front door fetches
4256 /// `${ASSET_ORIGIN}/<request path>` — the request path *is* the key. So a
4257 /// service with two static components had them overwrite each other at
4258 /// one prefix, and there was no way to say "this bundle serves under
4259 /// /app". `mount` is that: it appends to the publish prefix, which makes
4260 /// the URL sub-path and the storage sub-path the same string by
4261 /// construction rather than by two manifests agreeing.
4262 ///
4263 /// Cross-checked against the domain route that names the component
4264 /// ([`CloudConfig::cross_ref_validate`]): a component mounted at `/app`
4265 /// must be routed at `/app` or `/app/*`, because a disagreement means
4266 /// requests land on a prefix nothing published to — a 404 whose cause is
4267 /// two files apart.
4268 #[serde(default, skip_serializing_if = "Option::is_none")]
4269 pub mount: Option<String>,
4270 /// Sync-wave index (0-based). Components in wave 0 roll out in parallel
4271 /// first; the reconciler waits for all wave-N components to become healthy
4272 /// before starting wave N+1. Defaults to 0 (all components in one wave).
4273 #[serde(default, skip_serializing_if = "is_zero_u32")]
4274 pub wave: u32,
4275
4276 /// Whether this component ships inside the service's one assembled bundle
4277 /// or as a deployed unit of its own (R870-F23).
4278 ///
4279 /// This is the vocabulary R870-F15's design needed and the config did not
4280 /// have. `[providers.bundle]` is a per-**mirror** slot, so before this
4281 /// there was no way to say "give this one component its own workload" at
4282 /// all — the whole service was one bundle or it was nothing, and a service
4283 /// whose components genuinely release on different cadences had no shape
4284 /// to declare.
4285 ///
4286 /// It is one field rather than a pair of flags on purpose: a component
4287 /// being both bundle-staged and its own workload is the second admission
4288 /// rule R870-F23 was asked to enforce, and an enum makes it unrepresentable
4289 /// instead of merely refused.
4290 #[serde(default, skip_serializing_if = "DeployTier::is_default")]
4291 pub deploy: DeployTier,
4292}
4293
4294/// How one [`ServiceComponent`] reaches a node — see
4295/// [`ServiceComponent::deploy`].
4296#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
4297#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4298#[serde(rename_all = "kebab-case")]
4299pub enum DeployTier {
4300 /// Staged into the service's single assembled W272 bundle under
4301 /// `app/dist/<mount>/` and served by the one bundle workload (R870-B11).
4302 /// The default, and what every component in the tree means today.
4303 #[default]
4304 Bundle,
4305 /// Deployed as its own workload, with its own release cadence, its own
4306 /// address, and its own place in the inner door's mount table.
4307 Workload,
4308}
4309
4310impl DeployTier {
4311 /// Skip serializing the default so existing `service.toml` files
4312 /// round-trip byte-identically.
4313 fn is_default(&self) -> bool {
4314 matches!(self, DeployTier::Bundle)
4315 }
4316}
4317
4318#[inline]
4319fn is_zero_u32(n: &u32) -> bool {
4320 *n == 0
4321}
4322
4323/// A service's declared databases, grouped by environment (W241 §Sections).
4324/// Parsed from the `[db]` table of `service.toml`; each `[[db.<env>]]` array
4325/// entry names one database. The environment tag drives backend selection at
4326/// query time (see the data-workbench's `db.query` / the `sql_*` MCP tools):
4327/// `dev` = local file, `pond` = a DB inside the running pond container stack
4328/// (reached on a declared localhost port), `cloud` = a remote libSQL/Turso or
4329/// Postgres endpoint whose auth comes from an env var (never stored in TOML).
4330#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq, Eq)]
4331#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4332pub struct DbCatalog {
4333 /// Local-file SQLite databases used in dev mode.
4334 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4335 pub dev: Vec<DevDb>,
4336 /// Databases running inside the pond container stack.
4337 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4338 pub pond: Vec<PondDb>,
4339 /// Remote cloud databases (Turso, Postgres).
4340 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4341 pub cloud: Vec<CloudDb>,
4342}
4343
4344impl DbCatalog {
4345 /// True when no database is declared in any environment. Lets
4346 /// [`ServiceConfig`] skip serializing an empty `[db]` table.
4347 pub fn is_empty(&self) -> bool {
4348 self.dev.is_empty() && self.pond.is_empty() && self.cloud.is_empty()
4349 }
4350}
4351
4352/// A dev-mode local SQLite database (`[[db.dev]]`). `path` is resolved
4353/// relative to the workspace root and opened as a local file — read/write, no
4354/// network, no auth.
4355#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
4356#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4357pub struct DevDb {
4358 /// Logical name, unique within the service's `dev` list. Forms the `name`
4359 /// segment of the catalog id `dev:<service>:<name>`.
4360 pub name: String,
4361 /// On-disk SQLite path, relative to the workspace root (or absolute).
4362 pub path: String,
4363}
4364
4365/// A database running inside the pond container stack (`[[db.pond]]`). The
4366/// pond publishes the DB on a localhost TCP port; the hub connects to
4367/// `127.0.0.1:<port>` when the pond is up and returns a clear error when it is
4368/// not. Either `port` (defaulting to a libSQL/`sqld` HTTP endpoint) or a full
4369/// `url` must be given.
4370#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
4371#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4372pub struct PondDb {
4373 /// Logical name, unique within the service's `pond` list.
4374 pub name: String,
4375 /// Localhost TCP port the pond publishes the DB on. Interpreted per
4376 /// [`kind`](Self::kind). Mutually complete with `url` (provide one).
4377 #[serde(default, skip_serializing_if = "Option::is_none")]
4378 pub port: Option<u16>,
4379 /// Full connection URL, overriding `port` when set (e.g. a non-localhost
4380 /// host or an explicit scheme).
4381 #[serde(default, skip_serializing_if = "Option::is_none")]
4382 pub url: Option<String>,
4383 /// Wire protocol the pond DB speaks. Selects how a bare `port` becomes a
4384 /// URL: `turso` → `http://127.0.0.1:<port>` (libSQL/`sqld` over Hrana),
4385 /// `postgres` → `postgres://127.0.0.1:<port>`.
4386 #[serde(default)]
4387 pub kind: PondDbKind,
4388}
4389
4390/// Wire protocol of a [`PondDb`].
4391#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq, Eq)]
4392#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4393#[serde(rename_all = "kebab-case")]
4394pub enum PondDbKind {
4395 /// libSQL / `sqld` over Hrana HTTP — the default.
4396 #[default]
4397 Turso,
4398 /// PostgreSQL wire protocol.
4399 Postgres,
4400}
4401
4402/// A remote cloud database (`[[db.cloud]]`). The connection `url` is stored in
4403/// TOML but the credential never is — `auth_token_env` names an environment
4404/// variable the daemon reads at connect time, so the same declaration works
4405/// whether the token is provisioned service-locally or camp-shared (W241;
4406/// operator confirmed both scopes are needed). A camp-wide cloud DB not owned
4407/// by any single service is declared identically in `.yah/db/cloud.toml`.
4408#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
4409#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4410pub struct CloudDb {
4411 /// Logical name, unique within its `cloud` list.
4412 pub name: String,
4413 /// Connection URL: `libsql://…` / `http(s)://…` (Turso, `sqld`) or
4414 /// `postgres://…`.
4415 pub url: String,
4416 /// Name of the environment variable holding the auth token. Resolved in
4417 /// the daemon at connect time (value never stored on disk). For a libSQL
4418 /// URL the token is threaded as `?auth_token=…`.
4419 #[serde(default, skip_serializing_if = "Option::is_none")]
4420 pub auth_token_env: Option<String>,
4421}
4422
4423/// A camp-shared cloud database catalog, parsed from `.yah/db/cloud.toml`.
4424/// These are cloud DBs not owned by any single service — declared once at camp
4425/// scope and addressed as `cloud:<name>` (two-segment id), distinct from a
4426/// service-local `cloud:<service>:<name>`.
4427#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq, Eq)]
4428#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4429pub struct CampCloudDbs {
4430 #[serde(default, rename = "cloud", skip_serializing_if = "Vec::is_empty")]
4431 pub cloud: Vec<CloudDb>,
4432}
4433
4434impl CampCloudDbs {
4435 /// Load `<camp_root>/.yah/db/cloud.toml`, or an empty catalog if the file
4436 /// is absent (the common case — most camps declare no shared cloud DBs).
4437 pub fn load(camp_root: &Path) -> Result<Self> {
4438 let path = camp_root.join(".yah/db/cloud.toml");
4439 if !path.exists() {
4440 return Ok(Self::default());
4441 }
4442 let src = std::fs::read_to_string(&path)
4443 .with_context(|| format!("reading {}", path.display()))?;
4444 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
4445 }
4446}
4447
4448/// Topological shape of a mirror — how its providers sit relative to each other.
4449#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
4450#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4451#[serde(rename_all = "kebab-case")]
4452pub enum MirrorShape {
4453 /// Single machine hosts compute (and any non-Cloudflare-fronted static).
4454 SingleMachine,
4455 /// Operator-local dev mirror — static via built-in file server, compute
4456 /// via the local container runtime.
4457 Local,
4458 /// Multi-machine deployment (machines listed per provider slot).
4459 MultiMachine,
4460}
4461
4462/// Which public-ingress provider fronts this mirror's compute (W267, R594-F11).
4463///
4464/// Both arms answer exactly one question — *given these local workload ports,
4465/// make them publicly reachable at these hostnames* — and they differ only in
4466/// where the ingress rules live and who supervises the front door:
4467///
4468/// | | [`CloudflareTunnel`](Self::CloudflareTunnel) | [`Passway`](Self::Passway) |
4469/// |---|---|---|
4470/// | Ingress rules live | Cloudflare's API (token-form tunnels are remotely-managed) | the pingora `Backends` set in the proxy process |
4471/// | How they get there | an API call per deployed workload | passway polls `GET /service-records?ready=true` |
4472/// | Front door lifecycle | a kamaji-supervised `cloudflared` appliance | a kamaji-supervised passway appliance |
4473///
4474/// Flipping this field is the whole tier ladder: rented edge → sovereign edge
4475/// is a one-line mirror edit, not a rewrite. The provider owns **addressing**
4476/// and never **rendering** — the W173 render cube stays in mesofact's manifest.
4477#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
4478#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4479#[serde(rename_all = "kebab-case")]
4480pub enum IngressProvider {
4481 /// No public front door for this mirror. The default: a mirror that
4482 /// publishes to R2 behind a Worker, or a mesh-only compute tier, has no
4483 /// ingress provider to reconcile.
4484 #[default]
4485 None,
4486 /// Rented edge — `cloudflared` dials *out* from the node to Cloudflare's
4487 /// edge. Zero inbound ports, no TLS to manage on the box, hostname rules
4488 /// held in Cloudflare's API.
4489 CloudflareTunnel,
4490 /// Sovereign edge — passway terminates TLS on the node and load-balances
4491 /// an upstream set discovered from yubaba's service records.
4492 Passway,
4493}
4494
4495impl IngressProvider {
4496 /// `true` when this mirror declares a front door that has to be reconciled.
4497 pub fn is_declared(self) -> bool {
4498 !matches!(self, Self::None)
4499 }
4500
4501 /// Kebab-case wire name, as it appears in `mirrors/<env>.toml`.
4502 pub fn as_str(self) -> &'static str {
4503 match self {
4504 Self::None => "none",
4505 Self::CloudflareTunnel => "cloudflare-tunnel",
4506 Self::Passway => "passway",
4507 }
4508 }
4509}
4510
4511/// What a `cloudflare-tunnel` edge dials instead of the fronted workload —
4512/// W348 §1.3's **stacked** shape (R910).
4513///
4514/// `via = "passway"` puts the tunnel in front of the same node's passway door
4515/// for the same hostnames: cloudflared → the node's sni-demux on loopback
4516/// `:443` → the per-tenant passway → the workload. Passway keeps everything it
4517/// does on a public door — origin TLS, host routing, `[ingress.auth]`, ACME
4518/// (by DNS-01, the one challenge that reaches a NAT'd node) — and Cloudflare
4519/// owns only the browser-facing handshake. Without it a tunnel edge dials the
4520/// workload directly, and a tunnel edge and a passway edge claiming one
4521/// hostname is a partition conflict.
4522///
4523/// An enum rather than a bool because what it names is a front door, and
4524/// passway is simply the only one a tunnel can stack in front of today.
4525#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
4526#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4527#[serde(rename_all = "kebab-case")]
4528pub enum IngressVia {
4529 /// The node's own passway door, reached through its sni-demux.
4530 Passway,
4531}
4532
4533impl IngressVia {
4534 /// Kebab-case wire name, as it appears in `mirrors/<env>.toml`.
4535 pub fn as_str(self) -> &'static str {
4536 match self {
4537 Self::Passway => "passway",
4538 }
4539 }
4540
4541 /// The provider of the edge a tunnel carrying this `via` stacks in front
4542 /// of — the edge that has to exist, on the same machines, for the pair to
4543 /// plan.
4544 pub fn provider(self) -> IngressProvider {
4545 match self {
4546 Self::Passway => IngressProvider::Passway,
4547 }
4548 }
4549}
4550
4551/// One declared **edge**: a front door, the slots it fronts, and the nodes it
4552/// is placed on (W305 F2).
4553///
4554/// A mirror declares a *list* of these, which is what lets one service mix
4555/// front doors — cloudflare for the public web tier, passway for an internal or
4556/// high-throughput one. Before this, [`MirrorConfig::ingress`] was a single
4557/// [`IngressProvider`], so a mirror could **swap** front doors but never mix
4558/// them.
4559///
4560/// ```toml
4561/// [[ingress]]
4562/// provider = "passway"
4563/// machines = ["us-east-001", "us-south-001"]
4564/// slots = ["bundle"]
4565///
4566/// [[ingress]]
4567/// provider = "cloudflare-tunnel"
4568/// hostnames = ["issues.yah.dev"]
4569/// ```
4570///
4571/// **The per-node appliance is derived from this, never declared beside it.**
4572/// An edge does invoke a cloudflared or passway process on a box, but that is a
4573/// *consequence* of the service's declaration:
4574/// [`collate_front_doors`](crate::reconciler::collate_front_doors) walks every
4575/// service and derives what each node must run. Declaring it node-side too is
4576/// what produces two sources of truth for one fact.
4577#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
4578#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4579pub struct IngressEdge {
4580 /// Which front door this edge is. [`IngressProvider::None`] is rejected at
4581 /// plan time — an edge that fronts with nothing is always a typo, never an
4582 /// intent (write no edge instead).
4583 pub provider: IngressProvider,
4584 /// Nodes this front door is placed on — **independent of where the fronted
4585 /// workload runs** (R330-F37).
4586 ///
4587 /// Empty falls back to the fronted slot's own `machine` / `machines`, which
4588 /// is the co-located shape every mirror had before front-door placement was
4589 /// expressible. Listing several is what lets the ingress tier and the
4590 /// service tier scale independently: **N front doors over ONE deployment**,
4591 /// one rendered copy, so no cache coherence to settle.
4592 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4593 pub machines: Vec<String>,
4594 /// Provider slot roles this edge fronts (`"bundle"`, `"compute"`, …).
4595 ///
4596 /// One of the two selectors. With a single edge both may be empty, meaning
4597 /// "every fronted slot" — the legacy shape. With **several** edges a
4598 /// selector is mandatory on each, and the partition must be total and
4599 /// disjoint: a slot claimed by no edge, or by two, is an error naming it.
4600 /// An implicit catch-all across mixed front doors would silently publish a
4601 /// service through the wrong one.
4602 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4603 pub slots: Vec<String>,
4604 /// Public hostnames this edge fronts — the other selector, for partitioning
4605 /// by what the world dials rather than by which slot serves it.
4606 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4607 pub hostnames: Vec<String>,
4608 /// Cloudflare Tunnel id this edge publishes through, overriding the
4609 /// fronting machine's [`MachineConfig::cloudflared`].
4610 ///
4611 /// This is W267 Gap 3's real fix, and it is the *service* side of it: a node
4612 /// can join two cohorts' orange networks, and since §Granularity argues the
4613 /// tunnel credential **is** the isolation boundary, which cohort a given
4614 /// service fronts through is a property of the service, not of the box.
4615 /// `MachineConfig.cloudflared` stays as the per-node default (one tunnel is
4616 /// the common case, and the credential does live on the node), but it is no
4617 /// longer the only way to say it — so the node never has to enumerate
4618 /// cohorts.
4619 #[serde(default, skip_serializing_if = "Option::is_none")]
4620 pub tunnel_id: Option<String>,
4621 /// Infra provider id whose credentials this edge's front door authenticates
4622 /// with — `use = "cloudflare"`, resolved through
4623 /// `.yah/infra/providers/<id>.toml` exactly as a slot's `use` is.
4624 ///
4625 /// Same split as [`tunnel_id`](Self::tunnel_id), one field over: whose
4626 /// Cloudflare account holds the tunnel is a property of the **front door**,
4627 /// not of the box that runs the compute. Without this the account was read
4628 /// off the fronted slot's own `use`, which conflates two unrelated facts —
4629 /// and is unwritable for a slot whose compute provider is `kind = "static"`
4630 /// (a borrowed bare box: placement only, no credentials). Such a mirror had
4631 /// no way to name a Cloudflare account at all, short of writing
4632 /// `use = "cloudflare"` on the compute slot and lying about what runs it
4633 /// (R845).
4634 ///
4635 /// `None` falls back to the fronted slot's `use`, which is what every
4636 /// mirror written before this field meant.
4637 #[serde(default, rename = "use", skip_serializing_if = "Option::is_none")]
4638 pub provider_id: Option<String>,
4639 /// Digest-pinned image reference for this edge's front-door appliance,
4640 /// e.g. `localhost/passway:tag@sha256:<hex>` (R870-F16).
4641 ///
4642 /// `None` is the state of every mirror on disk today: the passway arm of
4643 /// `yah cloud apply` cannot deploy an appliance the mirror doesn't name an
4644 /// image for, so it renders the manual `yah cloud ingress deploy …
4645 /// --image <passway-ref>` step instead of running it. Declaring this field
4646 /// is what makes the arm self-sufficient, matching the CloudflareTunnel
4647 /// arm's real-API-call shape rather than only printing for an operator to
4648 /// copy by hand.
4649 #[serde(default, skip_serializing_if = "Option::is_none")]
4650 pub image: Option<String>,
4651 /// Cheers bearer-auth for this edge's door, spelled as an `[ingress.auth]`
4652 /// table under the `[[ingress]]` entry (R870-F26).
4653 ///
4654 /// ```toml
4655 /// [[ingress]]
4656 /// provider = "passway"
4657 /// image = "localhost/passway:v1@sha256:…"
4658 ///
4659 /// [ingress.auth]
4660 /// key_secret = "cheers/yah-camp/verify"
4661 /// kid = "YOHV4Riq-g8fX4uYl8rTjQ"
4662 /// iss = "yah-camp"
4663 /// aud = "analytics.yah.dev"
4664 /// require_prefixes = ["/"]
4665 /// ```
4666 ///
4667 /// **This is what makes an apply-driven push FAITHFUL rather than merely
4668 /// blocked.** Before it, `yah cloud apply`'s Passway arm rebuilt the door's
4669 /// spec with `auth: None` because a mirror had no way to say otherwise, and
4670 /// `/workloads/deploy` is a full replace — so pushing at a door someone had
4671 /// deployed with `--auth-key-secret …` took its auth away and brought it
4672 /// back anonymous (R870-B24). That strip is guarded by a read-back in
4673 /// `push_passway_ingress`, and the guard STAYS: it covers a door that
4674 /// acquired auth in a way no mirror can see. This field is what lets the
4675 /// common case sail past that guard by carrying the auth instead of losing
4676 /// it — the guard early-returns on any push that carries auth of its own.
4677 ///
4678 /// [`PasswayAuth`] verbatim, not a config-side copy of its five fields: all
4679 /// five are required by `Deserialize`, so a half-written table is refused
4680 /// by serde naming the missing field, and the renderer that emits the
4681 /// `PASSWAY_AUTH_*` variables reads the very same struct.
4682 ///
4683 /// Only meaningful on a `provider = "passway"` edge — a cloudflare-tunnel
4684 /// edge carrying one is refused by [`MirrorConfig::ingress_edges`] rather
4685 /// than silently ignored, since ignoring it yields exactly the
4686 /// believed-protected-but-public door this vocabulary exists to prevent.
4687 #[serde(default, skip_serializing_if = "Option::is_none")]
4688 pub auth: Option<PasswayAuth>,
4689 /// Stack this edge in front of another front door on the same node instead
4690 /// of dialing the workload (R910) — see [`IngressVia`].
4691 ///
4692 /// ```toml
4693 /// [[ingress]]
4694 /// provider = "passway"
4695 /// machines = ["us-west-011"]
4696 /// hostnames = ["api-staging.noisetable.com"]
4697 ///
4698 /// [[ingress]]
4699 /// provider = "cloudflare-tunnel"
4700 /// via = "passway"
4701 /// use = "cloudflare-tunnel-staging"
4702 /// machines = ["us-west-011"]
4703 /// hostnames = ["api-staging.noisetable.com"]
4704 /// ```
4705 ///
4706 /// Only a `cloudflare-tunnel` edge may carry it, and the mirror must also
4707 /// declare the edge it names, claiming the **same hostnames on the same
4708 /// machines** — the tunnel dials its own node's loopback demux, so a pair
4709 /// split across nodes routes to nothing. Refused otherwise by
4710 /// [`plan_ingress`](crate::reconciler::plan_ingress), naming both edges.
4711 ///
4712 /// `None` is every mirror written before R910.
4713 #[serde(default, skip_serializing_if = "Option::is_none")]
4714 pub via: Option<IngressVia>,
4715 /// The door behind a tunnel, spelled `[ingress.tunnel_door]` on the
4716 /// **passway** edge a `via = "passway"` tunnel stacks in front of (R910-F2).
4717 ///
4718 /// ```toml
4719 /// [ingress.tunnel_door]
4720 /// contact_email = "ops@example.com"
4721 /// zone_id = "<cloudflare zone id>"
4722 /// token_secret = "example/staging/cf-dns-token"
4723 /// ports = { "staging.example.com" = 8445 }
4724 /// ```
4725 ///
4726 /// Required exactly when the edge is derived behind a tunnel, refused
4727 /// otherwise — see `partition`. `yah cloud apply` turns it into one scoped
4728 /// enrollment per hostname, which the tunnel's machines route, arm and
4729 /// issue from; see [`TunnelDoor`].
4730 #[serde(default, skip_serializing_if = "Option::is_none")]
4731 pub tunnel_door: Option<TunnelDoor>,
4732}
4733
4734/// What a passway door behind a cloudflare tunnel needs that a public door
4735/// does not (R910-F2): where its per-hostname loopback listeners are, and the
4736/// inputs to issue its own certificates by DNS-01 — the only ACME challenge
4737/// that works when nothing public reaches the node.
4738///
4739/// One door (one cold passway, one certificate) per hostname, which is the
4740/// per-tenant shape yubaba's tenant tier arms — so each hostname names its own
4741/// port rather than sharing one listener.
4742#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
4743#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4744#[serde(deny_unknown_fields)]
4745pub struct TunnelDoor {
4746 /// ACME account contact.
4747 pub contact_email: String,
4748 /// Cloudflare zone id the `_acme-challenge` TXT records are written into.
4749 pub zone_id: String,
4750 /// Cluster secret holding a `Zone:DNS:Edit` token for that zone, declared
4751 /// in the machines' sovereign group with an access rule naming each
4752 /// hostname's door (`passway.<hostname>`).
4753 pub token_secret: String,
4754 /// Loopback port each hostname's door listens on — the demux backend the
4755 /// tunnel's SNI is spliced to. Every fronted hostname needs one.
4756 pub ports: std::collections::BTreeMap<String, u16>,
4757}
4758
4759impl IngressEdge {
4760 /// An edge with no selector — fronts every fronted slot, legal only when it
4761 /// is the mirror's only edge.
4762 pub fn all_slots(provider: IngressProvider, machines: Vec<String>) -> Self {
4763 Self {
4764 provider,
4765 machines,
4766 slots: Vec::new(),
4767 hostnames: Vec::new(),
4768 tunnel_id: None,
4769 provider_id: None,
4770 image: None,
4771 auth: None,
4772 via: None,
4773 tunnel_door: None,
4774 }
4775 }
4776
4777 /// Refuse `via` on an edge that cannot stack (R910). A passway edge
4778 /// terminates the connection itself, so `via` there would be ignored — and
4779 /// an ignored `via` is a mirror that reads as tunnel-fronted while
4780 /// publishing the door's own address.
4781 pub fn validate_via(&self) -> Result<()> {
4782 if self.via.is_some() && self.provider != IngressProvider::CloudflareTunnel {
4783 bail!(
4784 "{}: declares `via`, but only a `provider = \"cloudflare-tunnel\"` edge can stack \
4785 in front of another front door — a {:?} edge terminates the connection itself. \
4786 Move `via` to the tunnel edge, or drop it.",
4787 self.label(),
4788 self.provider.as_str()
4789 );
4790 }
4791 Ok(())
4792 }
4793
4794 /// Refuse an `[ingress.auth]` table that cannot produce a protected door
4795 /// (R870-F26). Called from [`MirrorConfig::ingress_edges`], so every reader
4796 /// of a mirror — plan, collate, `yah cloud validate`, apply — gets it.
4797 ///
4798 /// Two failures, and they fail in opposite directions, which is why both
4799 /// are here rather than left to the deploy:
4800 ///
4801 /// - **Auth on a non-passway edge.** Nothing downstream would read it, so
4802 /// the operator gets a door they believe is protected and is not. Only
4803 /// passway renders `PASSWAY_AUTH_*`; a cloudflare-tunnel edge publishes
4804 /// through Cloudflare Access instead and has no place to put these.
4805 /// - **A present-but-empty field.** `Deserialize` already refuses a
4806 /// *missing* one by name; [`PasswayAuth::validate`] covers the rest, and
4807 /// is the same implementation `yah cloud ingress deploy` runs on its
4808 /// flags — so the two doors cannot diverge on what counts as configured.
4809 pub fn validate_auth(&self) -> Result<()> {
4810 let Some(auth) = &self.auth else {
4811 return Ok(());
4812 };
4813 if !matches!(self.provider, IngressProvider::Passway) {
4814 bail!(
4815 "{}: declares `[ingress.auth]`, but only `provider = \"passway\"` renders the \
4816 PASSWAY_AUTH_* variables — this edge would deploy a door with NO bearer auth \
4817 while the mirror says otherwise. Move the auth to the passway edge, or drop it.",
4818 self.label()
4819 );
4820 }
4821 auth.validate(AuthSpelling::MirrorTable)
4822 .map_err(|why| anyhow::anyhow!("{}: {why}", self.label()))
4823 }
4824
4825 /// Refuse an `[ingress.tunnel_door]` that cannot produce a working door
4826 /// (R910-F2). Whether the edge is actually behind a tunnel is a property
4827 /// of the pair, so `partition` checks that half.
4828 pub fn validate_tunnel_door(&self) -> Result<()> {
4829 let Some(door) = &self.tunnel_door else {
4830 return Ok(());
4831 };
4832 if self.provider != IngressProvider::Passway {
4833 bail!(
4834 "{}: declares `[ingress.tunnel_door]`, but only the `provider = \"passway\"` edge a \
4835 tunnel stacks in front of has a door to describe. Move it to that edge, or drop it.",
4836 self.label()
4837 );
4838 }
4839 for (field, value) in [
4840 ("contact_email", &door.contact_email),
4841 ("zone_id", &door.zone_id),
4842 ("token_secret", &door.token_secret),
4843 ] {
4844 if value.trim().is_empty() {
4845 bail!("{}: `[ingress.tunnel_door].{field}` is empty", self.label());
4846 }
4847 }
4848 if door.ports.is_empty() {
4849 bail!(
4850 "{}: `[ingress.tunnel_door].ports` is empty — each fronted hostname needs the \
4851 loopback port its door listens on",
4852 self.label()
4853 );
4854 }
4855 let mut seen: std::collections::BTreeMap<u16, &str> = std::collections::BTreeMap::new();
4856 for (host, port) in &door.ports {
4857 if *port == 0 {
4858 bail!("{}: `[ingress.tunnel_door].ports.{host:?}` is 0", self.label());
4859 }
4860 if let Some(other) = seen.insert(*port, host) {
4861 bail!(
4862 "{}: `[ingress.tunnel_door].ports` gives {other:?} and {host:?} the same port \
4863 {port} — one held socket cannot be two hostnames' door",
4864 self.label()
4865 );
4866 }
4867 }
4868 Ok(())
4869 }
4870
4871 /// `true` when this edge names which slots/hostnames it fronts.
4872 pub fn has_selector(&self) -> bool {
4873 !self.slots.is_empty() || !self.hostnames.is_empty()
4874 }
4875
4876 /// Does this edge claim the rule derived from `slot` publishing `hostname`?
4877 ///
4878 /// A selectorless edge claims everything; that is checked to be
4879 /// unambiguous (one edge only) before this is consulted.
4880 pub fn claims(&self, slot: &str, hostname: &str) -> bool {
4881 if !self.has_selector() {
4882 return true;
4883 }
4884 self.slots.iter().any(|s| s == slot) || self.hostnames.iter().any(|h| h == hostname)
4885 }
4886
4887 /// Human-readable identity for an error message — the provider plus
4888 /// whichever selector was written.
4889 pub fn label(&self) -> String {
4890 let sel = match (self.slots.is_empty(), self.hostnames.is_empty()) {
4891 (true, true) => "no selector".to_string(),
4892 (false, true) => format!("slots = {:?}", self.slots),
4893 (true, false) => format!("hostnames = {:?}", self.hostnames),
4894 (false, false) => format!("slots = {:?} + hostnames = {:?}", self.slots, self.hostnames),
4895 };
4896 format!("[[ingress]] provider = {:?} ({sel})", self.provider.as_str())
4897 }
4898}
4899
4900/// A mirror's `ingress` declaration, in either spelling.
4901///
4902/// The list is the general form; the bare provider is shorthand for the single
4903/// edge fronting everything, and is kept rather than migrated because it is the
4904/// honest spelling for the common case — one service, one front door. Both
4905/// normalize to the same `Vec<IngressEdge>` through
4906/// [`MirrorConfig::ingress_edges`], so nothing downstream branches on which was
4907/// written.
4908#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
4909#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4910#[serde(untagged)]
4911pub enum IngressDecl {
4912 /// `ingress = "passway"` — one edge fronting every fronted slot, placed by
4913 /// the sibling [`MirrorConfig::ingress_machines`].
4914 Provider(IngressProvider),
4915 /// `[[ingress]]` — one entry per declared edge.
4916 Edges(Vec<IngressEdge>),
4917}
4918
4919/// Hand-written because `#[serde(untagged)]` throws the real error away.
4920///
4921/// A derived untagged `Deserialize` tries each variant and, on failure, reports
4922/// only `data did not match any variant of untagged enum IngressDecl` — so a
4923/// misspelled `provider = "passwya"` says nothing about providers, nothing about
4924/// the legal values, and points at the `[[ingress]]` header rather than the
4925/// field. Dispatching on the input shape first means each arm's own error
4926/// survives: a bad string names the legal provider vocabulary, a bad edge table
4927/// names the offending field.
4928impl<'de> Deserialize<'de> for IngressDecl {
4929 fn deserialize<D: serde::Deserializer<'de>>(d: D) -> std::result::Result<Self, D::Error> {
4930 struct DeclVisitor;
4931
4932 impl<'de> serde::de::Visitor<'de> for DeclVisitor {
4933 type Value = IngressDecl;
4934
4935 fn expecting(&self, f: &mut std::fmt::Formatter) -> std::fmt::Result {
4936 f.write_str(
4937 "a provider name (`ingress = \"passway\"`) or a list of edge tables \
4938 (`[[ingress]]`)",
4939 )
4940 }
4941
4942 fn visit_str<E: serde::de::Error>(self, v: &str) -> std::result::Result<Self::Value, E> {
4943 IngressProvider::deserialize(serde::de::value::StrDeserializer::new(v))
4944 .map(IngressDecl::Provider)
4945 }
4946
4947 fn visit_seq<A: serde::de::SeqAccess<'de>>(
4948 self,
4949 seq: A,
4950 ) -> std::result::Result<Self::Value, A::Error> {
4951 Vec::<IngressEdge>::deserialize(serde::de::value::SeqAccessDeserializer::new(seq))
4952 .map(IngressDecl::Edges)
4953 }
4954 }
4955
4956 d.deserialize_any(DeclVisitor)
4957 }
4958}
4959
4960/// No front door — the shape of every mirror that publishes to R2 behind a
4961/// Worker, or runs a mesh-only compute tier.
4962impl Default for IngressDecl {
4963 fn default() -> Self {
4964 Self::Provider(IngressProvider::None)
4965 }
4966}
4967
4968impl IngressDecl {
4969 /// `true` when this mirror declares no front door at all.
4970 pub fn is_absent(&self) -> bool {
4971 match self {
4972 Self::Provider(p) => !p.is_declared(),
4973 Self::Edges(e) => e.is_empty(),
4974 }
4975 }
4976}
4977
4978impl From<IngressProvider> for IngressDecl {
4979 fn from(p: IngressProvider) -> Self {
4980 Self::Provider(p)
4981 }
4982}
4983
4984/// A service mirror — the projection of a [`ServiceConfig`] onto concrete
4985/// infra. Lives at `.yah/services/<svc>/mirrors/<env>.toml`.
4986#[derive(Debug, Clone, Serialize, Deserialize)]
4987#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4988pub struct MirrorConfig {
4989 pub schema_version: u32,
4990 pub shape: MirrorShape,
4991 /// Public-ingress edges fronting this mirror (W267, W305 F2). Defaults to
4992 /// none.
4993 ///
4994 /// Two spellings, one meaning — see [`IngressDecl`]. `ingress = "passway"`
4995 /// is one edge fronting everything; `[[ingress]]` entries declare several,
4996 /// each naming its provider plus the slots or hostnames it fronts. Read it
4997 /// through [`ingress_edges`](Self::ingress_edges), never by matching on the
4998 /// enum, so the two spellings cannot drift apart.
4999 ///
5000 /// Declared at mirror scope rather than per provider slot because a front
5001 /// door does **fan-in**: one `cloudflared` (or one passway) on a node
5002 /// multiplexes every hostname→port rule it fronts, so pinning one to a
5003 /// single slot would mint one edge connection per slot for no gain. An
5004 /// edge's `slots` selector is the general form of that — it groups slots
5005 /// behind one front door, it does not split a front door per slot.
5006 #[serde(default, skip_serializing_if = "IngressDecl::is_absent")]
5007 pub ingress: IngressDecl,
5008 /// Machines the front door is placed on — **independent of where the
5009 /// fronted workload runs** (R330-F37).
5010 ///
5011 /// The single-edge spelling of [`IngressEdge::machines`]: it applies to the
5012 /// one edge `ingress = "<provider>"` declares, and combining it with
5013 /// `[[ingress]]` entries is an error rather than a silent precedence rule.
5014 ///
5015 /// Empty (the default) keeps the pre-existing behaviour: the front door is
5016 /// co-located with the fronted slot's own `machine` / `machines`. That was
5017 /// never a design choice, it was an artifact of bundles binding
5018 /// `127.0.0.1` — nothing off-node could reach a workload, so a proxy had to
5019 /// sit on top of it. R599-F12 landed mesh binding, which removes the
5020 /// constraint: passway is a reverse proxy, and a valid front door needs a
5021 /// cert and an upstream it can *reach*, not a local copy of the service.
5022 ///
5023 /// Listing several machines is what lets the ingress tier and the service
5024 /// tier scale independently — **N front doors over ONE deployment**. There
5025 /// is still exactly one rendered copy of the site, so fanning the front door
5026 /// out introduces no cache-coherence problem; that only appears if you
5027 /// deploy the *workload* to every node instead.
5028 ///
5029 /// ```toml
5030 /// ingress = "passway"
5031 /// ingress_machines = ["us-east-001", "us-west-001"]
5032 /// ```
5033 ///
5034 /// Declaring this without [`ingress`](Self::ingress) is an error, not a
5035 /// no-op — it always means the operator expected a front door somewhere.
5036 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5037 pub ingress_machines: Vec<String>,
5038 /// Provider slots, keyed by role (`"static"`, `"compute"`, …). Each value
5039 /// either references a provider declared under `.yah/infra/providers/` or
5040 /// inlines a local-only provider (no creds, no infra file).
5041 ///
5042 /// A role is normally service-wide — one slot serves every component that
5043 /// shares it — but [`ReconcileCtx::slot`](crate::reconciler::ReconcileCtx::slot)
5044 /// looks up the component-qualified key `"<role>:<component id>"` first.
5045 /// A service with two components of the same role (e.g. two
5046 /// `mesofact-static` components under one mirror) declares
5047 /// `providers."static:<id>"` per component to give each its own port;
5048 /// omitting the qualifier keeps the pre-existing single-slot behavior.
5049 #[serde(default)]
5050 pub providers: BTreeMap<String, MirrorProviderSlot>,
5051 /// Capability→driver bindings, keyed by **capability** (`"pg"`, `"s3"`, …)
5052 /// rather than by slot role (W265 §Drivers).
5053 ///
5054 /// This is the generalization of [`Self::providers`]: `providers.static` /
5055 /// `providers.object_store` are the special case where the slot name and
5056 /// the capability happen to coincide, and keying by capability is what stops
5057 /// the slot enum growing one arm per tier-specific implementation. A service
5058 /// says "I need pg"; the mirror says which implementation of pg *this tier*
5059 /// uses; the app talks the same wire protocol either way and never forks.
5060 ///
5061 /// ```toml
5062 /// [drivers.pg]
5063 /// kind = "local-pg-dev" # dev — kamaji-supervised loopback postgres
5064 ///
5065 /// [drivers.smtp]
5066 /// kind = "local-mailcrab" # dev/pond — a catcher with a browsable inbox
5067 /// ```
5068 ///
5069 /// Additive in P1: `drivers` lands *alongside* `providers`, and migrating
5070 /// the existing `providers.static` / `providers.object_store` declarations
5071 /// over is a separate pass (W265 §"Open follow-ups"). A mirror that declares
5072 /// neither is unchanged.
5073 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
5074 pub drivers: BTreeMap<String, MirrorProviderSlot>,
5075 /// Per-environment alias overrides for `kind = "static-asset"` components.
5076 ///
5077 /// Keys are logical names (e.g. `"whisper-default"`); values must be
5078 /// filenames present in the component's `workload.toml` catalog.
5079 /// **Resolution only** — this table may never introduce a filename absent
5080 /// from the catalog. Validated against the workload catalog at sync time.
5081 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
5082 pub asset_aliases: BTreeMap<String, String>,
5083 /// Per-environment build overrides, keyed by **component id** (R905).
5084 ///
5085 /// A `[build]` block lives on the component's `workload.toml`, and a
5086 /// component is declared exactly once in `service.toml` — so without this
5087 /// table a service that deploys the same component to two environments
5088 /// builds it identically for both. That is wrong for any bundle whose
5089 /// contents depend on the tier it is being built *for*: noisetable's
5090 /// landing site bakes `NOISETABLE_API_ORIGIN` into the shipped JS, so the
5091 /// staging site was served a production API origin and every call from it
5092 /// was blocked by production CORS.
5093 ///
5094 /// ```toml
5095 /// # .yah/services/noisetable-marketing/mirrors/staging.toml
5096 /// [build.site]
5097 /// command = "bun run build:staging"
5098 ///
5099 /// # or, without a sibling script per environment:
5100 /// [build.site.env]
5101 /// NOISETABLE_API_ORIGIN = "https://api-staging.noisetable.com"
5102 /// ```
5103 ///
5104 /// The environment axis stays on the mirror, where `providers`, `drivers`
5105 /// and `ingress` already live, rather than growing an `env`-keyed table on
5106 /// the component's own `BuildConfig` — a per-component struct is the wrong
5107 /// place to enumerate environments, and doing it there would have made the
5108 /// mirror the *second* per-environment surface instead of the only one.
5109 ///
5110 /// Deliberately NOT overridable here: `out_dir`. Where a bundler writes is
5111 /// a property of the project's own toolchain, not of the tier it is built
5112 /// for, and it is read independently of `[build]` by the publish path
5113 /// (`read_workload_out_dir`, `collect_component_files`) — making it
5114 /// per-environment would mean threading the mirror into every one of those
5115 /// readers to buy a knob no tier needs.
5116 ///
5117 /// Keys are validated against the service's declared component ids at
5118 /// config load (`cross_ref_validate`), so a typo is a refusal rather than
5119 /// an override that silently never fires.
5120 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
5121 pub build: BTreeMap<String, MirrorBuildOverride>,
5122}
5123
5124/// One component's per-environment build override — the value type of
5125/// [`MirrorConfig::build`] (R905).
5126///
5127/// Every field is additive-or-replacing against the component's own
5128/// `workload.toml [build]` table; an empty override is indistinguishable from
5129/// declaring none.
5130#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
5131#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5132#[serde(deny_unknown_fields)]
5133pub struct MirrorBuildOverride {
5134 /// Replaces `build.command` for this environment.
5135 ///
5136 /// Absent leaves the workload's own command in place — including absent,
5137 /// which means "this project has no external bundler step" and must keep
5138 /// meaning that (R838-B1). An override may therefore *introduce* a command
5139 /// where the workload's `[build]` table declares none — a deliberate
5140 /// per-tier opt-in to a bundler, since the only way to write it is to name
5141 /// one. A workload with no `[build]` table at all is still skipped
5142 /// wholesale: there is no `out_dir` to publish from, so an override there
5143 /// would have nothing to hand the publish step.
5144 #[serde(default, skip_serializing_if = "Option::is_none")]
5145 pub command: Option<String>,
5146 /// Replaces `build.render_command` — the data-only re-render
5147 /// (`revalidate_static`). Overridden separately from `command` because the
5148 /// two run at different times against different inputs; a tier that needs
5149 /// a different bundler command usually needs the same renderer.
5150 #[serde(default, skip_serializing_if = "Option::is_none")]
5151 pub render_command: Option<String>,
5152 /// Environment variables exported to the build (and render) subprocess,
5153 /// applied on top of the inherited environment.
5154 ///
5155 /// This is the knob for the common case — the command is the same, only a
5156 /// baked-in origin/flag differs — and it is what keeps a project from
5157 /// having to pre-declare one `build:<env>` script per environment before
5158 /// any environment can exist.
5159 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
5160 pub env: BTreeMap<String, String>,
5161}
5162
5163impl MirrorBuildOverride {
5164 /// `true` when this override would change nothing.
5165 pub fn is_empty(&self) -> bool {
5166 self.command.is_none() && self.render_command.is_none() && self.env.is_empty()
5167 }
5168
5169 /// The override's `env` table as the `Vec<(String, String)>` that both
5170 /// `ExecContext::with_env` and `std::process::Command::envs` want.
5171 pub fn env_pairs(&self) -> Vec<(String, String)> {
5172 self.env
5173 .iter()
5174 .map(|(k, v)| (k.clone(), v.clone()))
5175 .collect()
5176 }
5177}
5178
5179impl MirrorConfig {
5180 /// This environment's build override for `component_id`, if any.
5181 ///
5182 /// Returns `None` for an override that exists but changes nothing, so
5183 /// callers can treat `Some(_)` as "something differs here" without
5184 /// re-checking each field.
5185 pub fn build_override(&self, component_id: &str) -> Option<&MirrorBuildOverride> {
5186 self.build.get(component_id).filter(|o| !o.is_empty())
5187 }
5188}
5189
5190impl MirrorConfig {
5191 /// This mirror's declared edges, with both spellings normalized (W305 F2).
5192 ///
5193 /// The single place `ingress` + `ingress_machines` are reconciled, so no
5194 /// consumer has to know which spelling was written. Returns an empty vec
5195 /// when the mirror declares no front door.
5196 ///
5197 /// Errors are the declarations that cannot mean anything:
5198 ///
5199 /// - `ingress_machines` with no `ingress` — front-door placement with no
5200 /// front door to place, always a typo (R330-F37);
5201 /// - `ingress_machines` alongside `[[ingress]]` — placement declared twice,
5202 /// in a form where one silently wins;
5203 /// - `provider = "none"` on an edge — an edge that fronts with nothing.
5204 /// The `[[ingress]]` entries exactly as written, without normalizing the
5205 /// scalar spelling or validating anything.
5206 ///
5207 /// [`ingress_edges`](Self::ingress_edges) is the one to reach for; this
5208 /// exists for the checks that must run *before* a mirror is known to be
5209 /// well-formed — cross-reference validation walks every mirror in the
5210 /// workspace, and hard-failing there on an unrelated mirror's shape error
5211 /// would report the wrong file. Empty for the scalar spelling, which has no
5212 /// edge table to carry per-edge fields.
5213 pub fn ingress_edge_slice(&self) -> &[IngressEdge] {
5214 match &self.ingress {
5215 IngressDecl::Edges(edges) => edges,
5216 IngressDecl::Provider(_) => &[],
5217 }
5218 }
5219
5220 pub fn ingress_edges(&self) -> Result<Vec<IngressEdge>> {
5221 match &self.ingress {
5222 IngressDecl::Provider(p) if !p.is_declared() => {
5223 if !self.ingress_machines.is_empty() {
5224 bail!(
5225 "mirror declares `ingress_machines = {:?}` but no `ingress` provider — \
5226 front-door placement with no front door to place. Add \
5227 `ingress = \"passway\"` (or \"cloudflare-tunnel\"), or drop \
5228 `ingress_machines`.",
5229 self.ingress_machines
5230 );
5231 }
5232 Ok(Vec::new())
5233 }
5234 IngressDecl::Provider(p) => Ok(vec![IngressEdge::all_slots(
5235 *p,
5236 self.ingress_machines.clone(),
5237 )]),
5238 IngressDecl::Edges(edges) => {
5239 if !self.ingress_machines.is_empty() {
5240 bail!(
5241 "mirror declares both `[[ingress]]` edges and the single-edge \
5242 `ingress_machines = {:?}` — front-door placement stated twice. Move \
5243 those names onto the edge they place: `machines = [...]` inside the \
5244 `[[ingress]]` entry.",
5245 self.ingress_machines
5246 );
5247 }
5248 for edge in edges {
5249 if !edge.provider.is_declared() {
5250 bail!(
5251 "{}: `provider = \"none\"` fronts nothing. An edge exists to name a \
5252 front door — delete the entry instead.",
5253 edge.label()
5254 );
5255 }
5256 edge.validate_auth()?;
5257 edge.validate_via()?;
5258 edge.validate_tunnel_door()?;
5259 }
5260 Ok(edges.clone())
5261 }
5262 }
5263 }
5264
5265 /// Nodes this mirror's **passway** front doors are placed on, in
5266 /// declaration order and de-duplicated — or `None` when the mirror declares
5267 /// no passway edge at all.
5268 ///
5269 /// `Some(vec![])` is a real and different answer from `None`: a passway edge
5270 /// is declared but names no machine, so its placement falls back to the
5271 /// fronted slot's own. That fallback is placement *resolution* — it belongs
5272 /// to [`IngressRule::machines`](crate::reconciler::IngressRule::machines)
5273 /// and the plan it is built from, not to a mirror read in isolation — so it
5274 /// is reported as "declared, placement unknown from here" rather than
5275 /// half-derived. A caller that needs a node to dial has to say so.
5276 ///
5277 /// Passway-only because the caller is tenant DNS onboarding: only a passway
5278 /// node serves yubaba's `GET /domains/{domain}/onboarding`. A
5279 /// cloudflare-tunnel edge publishes through Cloudflare's own DNS and has no
5280 /// such record to hand a tenant, so folding its machines in would point the
5281 /// UI at a node that cannot answer.
5282 ///
5283 /// Read through [`ingress_edges`](Self::ingress_edges), so both spellings
5284 /// are covered by construction. A declaration that cannot mean anything
5285 /// (`ingress_machines` with no `ingress`, or both spellings at once) reads
5286 /// as `None` rather than propagating an error: those are reported by
5287 /// cross-reference validation, which can name the offending file.
5288 pub fn passway_machines(&self) -> Option<Vec<String>> {
5289 let edges = self.ingress_edges().ok()?;
5290 let mut declared = false;
5291 let mut machines: Vec<String> = Vec::new();
5292 for edge in edges
5293 .iter()
5294 .filter(|e| matches!(e.provider, IngressProvider::Passway))
5295 {
5296 declared = true;
5297 for m in &edge.machines {
5298 if !machines.iter().any(|seen| seen == m) {
5299 machines.push(m.clone());
5300 }
5301 }
5302 }
5303 declared.then_some(machines)
5304 }
5305
5306 /// Parse a single `mirrors/<env>.toml` file.
5307 pub fn load(path: &Path) -> Result<Self> {
5308 let src =
5309 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
5310 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
5311 }
5312
5313 /// Persist to `.yah/services/<service>/mirrors/<env>.toml`, creating the
5314 /// `mirrors/` directory if needed. Create-or-overwrite. The mirror file is
5315 /// named by `env` (its stem); `service` selects the owning service dir.
5316 pub fn save(&self, workspace_root: &Path, service: &str, env: &str) -> Result<()> {
5317 let dir = crate::paths::service_mirrors_dir(workspace_root, service);
5318 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
5319 let path = crate::paths::service_mirror_toml(workspace_root, service, env);
5320 let s = toml::to_string_pretty(self)
5321 .with_context(|| format!("serializing mirror {service}/{env}"))?;
5322 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
5323 }
5324
5325 /// Remove `.yah/services/<service>/mirrors/<env>.toml`. Returns `false`
5326 /// when the file was already absent. Leaves the service and its other
5327 /// mirrors untouched.
5328 ///
5329 /// Also checks legacy stems (e.g. `local-sim` when `env = "pond"`) so
5330 /// deleting a canonical tier name removes whichever file exists on disk.
5331 pub fn delete(workspace_root: &Path, service: &str, env: &str) -> Result<bool> {
5332 let path = crate::paths::service_mirror_toml(workspace_root, service, env);
5333 if path.exists() {
5334 std::fs::remove_file(&path).with_context(|| format!("removing {}", path.display()))?;
5335 return Ok(true);
5336 }
5337 // Try legacy file stems for canonical tier names.
5338 let legacy: &[&str] = match env {
5339 "dev" => &["local"],
5340 "pond" => &["local-sim", "sim"],
5341 "prod" => &["cloud"],
5342 _ => &[],
5343 };
5344 for stem in legacy {
5345 let alt = crate::paths::service_mirror_toml(workspace_root, service, stem);
5346 if alt.exists() {
5347 std::fs::remove_file(&alt)
5348 .with_context(|| format!("removing {}", alt.display()))?;
5349 return Ok(true);
5350 }
5351 }
5352 Ok(false)
5353 }
5354}
5355
5356/// A provider slot inside a [`MirrorConfig`]. Two shapes:
5357/// - **Reference** (`use = "<provider-id>"`) — point at an infra-declared
5358/// provider; extra fields are slot-specific (bucket, zone, dns, …).
5359/// - **Inline** (`kind = "local-*"`) — for providers that need no infra
5360/// declaration because they carry no credentials.
5361#[derive(Debug, Clone, Serialize, Deserialize)]
5362#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5363#[serde(untagged)]
5364pub enum MirrorProviderSlot {
5365 Reference {
5366 #[serde(rename = "use")]
5367 provider_id: String,
5368 #[serde(flatten)]
5369 #[cfg_attr(
5370 feature = "json-schema",
5371 schemars(with = "std::collections::BTreeMap<String, serde_json::Value>")
5372 )]
5373 fields: BTreeMap<String, toml::Value>,
5374 },
5375 Inline {
5376 kind: Provider,
5377 #[serde(flatten)]
5378 #[cfg_attr(
5379 feature = "json-schema",
5380 schemars(with = "std::collections::BTreeMap<String, serde_json::Value>")
5381 )]
5382 fields: BTreeMap<String, toml::Value>,
5383 },
5384}
5385
5386impl MirrorProviderSlot {
5387 /// Provider id this slot references, or `None` for inline slots.
5388 pub fn provider_id(&self) -> Option<&str> {
5389 match self {
5390 Self::Reference { provider_id, .. } => Some(provider_id),
5391 Self::Inline { .. } => None,
5392 }
5393 }
5394
5395 /// Provider kind for inline slots, or `None` for reference slots
5396 /// (resolve via the referenced [`ProviderConfig`]).
5397 pub fn inline_kind(&self) -> Option<Provider> {
5398 match self {
5399 Self::Reference { .. } => None,
5400 Self::Inline { kind, .. } => Some(*kind),
5401 }
5402 }
5403
5404 pub fn fields(&self) -> &BTreeMap<String, toml::Value> {
5405 match self {
5406 Self::Reference { fields, .. } | Self::Inline { fields, .. } => fields,
5407 }
5408 }
5409
5410 /// F16 placement: parse the optional `required = { … }` sub-table on this
5411 /// slot. Returns `None` when absent or unparseable (callers treat as no
5412 /// constraint). See [`RequiredSpec`] for the field grammar.
5413 pub fn required(&self) -> Option<RequiredSpec> {
5414 let v = self.fields().get("required")?.clone();
5415 v.try_into().ok()
5416 }
5417}
5418
5419/// F16 placement constraints declared on a [`MirrorProviderSlot`], lives under
5420/// `[providers.<role>] required = { regions = [...], mesh_tags = [...] }` in
5421/// `mirrors/<env>.toml`.
5422///
5423/// Hard (must-satisfy) axes, all AND-ed together:
5424/// - `regions` / `zones` / `providers` — *membership*: the machine's
5425/// `region` / `zone` / `provider` must be one of the listed values.
5426/// - `mesh_tags` — *superset*: the machine's `mesh_tags` must contain every
5427/// listed tag.
5428/// - `memory_mb` / `cpu_millis` — *capacity floor* (R572-F5): the machine's
5429/// `allocatable` budget must cover the demand. `0` = no constraint.
5430/// - *taint repulsion* — **unconditional** (R876-B7): the machine must not
5431/// carry any taint that [`taint_effect`] classifies as
5432/// [`TaintEffect::Repels`], unless that exact key is listed in
5433/// [`Self::tolerates`]. This axis is not declared; it applies to every spec.
5434/// - `requires_taint` — *taint affinity* (R572-F5): the machine must carry
5435/// this taint key (in `taints` or `mesh_tags`). `None` = no affinity.
5436///
5437/// These are the **only** readers of [`MachineConfig::taints`], which is
5438/// what makes [`taint_effect`]'s closed vocabulary well-founded.
5439///
5440/// [`MachineConfig::sovereign_group`] is deliberately **not** an axis here and
5441/// must not become one (W305/R742-F1). A sovereign group is a blast radius,
5442/// not a filter: which quorum a box votes in says nothing about whether a
5443/// workload may run on it, and a dev-group node exists precisely so dev-mode
5444/// services — stateful ones included — can be scheduled onto it. Filtering on
5445/// it would re-make the mistake W305 exists to undo, where one mechanism
5446/// silently carried three unrelated properties.
5447///
5448/// An empty / zero / None on every axis means "no constraint on that axis".
5449/// A fully-unconstrained `RequiredSpec` matches every machine (see
5450/// [`RequiredSpec::is_unconstrained`]).
5451#[derive(Debug, Clone, Default, Serialize, Deserialize)]
5452#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5453pub struct RequiredSpec {
5454 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5455 pub regions: Vec<String>,
5456 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5457 pub zones: Vec<String>,
5458 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5459 pub providers: Vec<String>,
5460 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5461 pub mesh_tags: Vec<String>,
5462
5463 /// R833-F8: **imperative** placement — the machine must be one of these by
5464 /// `name`. Empty (the default) = no constraint, which is every pre-R833-F8
5465 /// caller.
5466 ///
5467 /// This is the one axis that is not a *capability* the scheduler infers.
5468 /// The operator typed `--where=node:us-west-003`, so it composes with the
5469 /// other axes exactly like the rest — a named node that fails the capacity
5470 /// floor or carries a repelling taint still does not match, and the refusal
5471 /// names why rather than silently placing the work somewhere else.
5472 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5473 pub nodes: Vec<String>,
5474
5475 /// R572-F5: minimum memory (MiB) the target node must have in its
5476 /// declared `allocatable` budget. `0` = no constraint. Filled by
5477 /// [`CloudConfig::admit_workload`] from the workload's
5478 /// `memory_request_mb()` — its placement **request**, which is not the
5479 /// same number as the `resources.memory_mb` cgroup **ceiling**.
5480 #[serde(default, skip_serializing_if = "is_zero_u32")]
5481 pub memory_mb: u32,
5482 /// R572-F5: minimum CPU (millicores) the target node must have in its
5483 /// declared `allocatable` budget. `0` = no constraint. Filled by
5484 /// [`CloudConfig::admit_workload`] from the workload's `resources.cpu_millis`.
5485 #[serde(default, skip_serializing_if = "is_zero_u32")]
5486 pub cpu_millis: u32,
5487 /// R876-B7: repelling node taints this placement **opts back in to**.
5488 /// Each entry is a machine taint key spelled exactly as it appears in
5489 /// [`MachineConfig::taints`] — `"no-appliance"`, not `"appliance"` — so the
5490 /// node side and the workload side share one vocabulary and nothing has to
5491 /// translate between them.
5492 ///
5493 /// # Why this replaced `repel_archetypes`
5494 ///
5495 /// Repulsion used to be **opt-in-to-be-repelled**: the spec named the
5496 /// archetypes it was, and only a `no-<that archetype>` taint blocked it.
5497 /// That field was `#[serde(skip)]`, so a slot declared as
5498 /// `required = { ... }` in a mirror TOML always deserialized with it empty
5499 /// and [`Self::matches`] never read [`MachineConfig::taints`] at all. Node
5500 /// taints were therefore structurally inert for every mirror-declared
5501 /// placement, and inert *silently* — `no-server` is a legal key, so
5502 /// `yah cloud validate` passed and an operator draining a node before
5503 /// maintenance got a green run and a workload that never moved (R876-S2's
5504 /// drill measured exactly this against the real tree).
5505 ///
5506 /// The sense is now inverted, which is the only shape that can survive a
5507 /// field the wire does not carry: **repulsion is unconditional and
5508 /// toleration is declared.** A spec that says nothing is repelled by every
5509 /// repelling taint — the reading an operator writing `taints = ["no-server"]`
5510 /// on a machine already assumed they were getting.
5511 ///
5512 /// Toleration is per-key and absolute; there is no wildcard. Listing a key
5513 /// no machine declares is harmless and matches nothing.
5514 ///
5515 /// [`admission_spec`] fills this from the placement group's archetypes —
5516 /// every repelling key that is *not* the group's own class — which is what
5517 /// makes the `admit_workload` path behave identically across this change
5518 /// (R860-T4 / W338 §Placement consequences 2 still hold: the group's
5519 /// archetypes are the union over `local` requirement edges, so a `Server`
5520 /// bound to an `Appliance` tolerates neither `no-server` nor
5521 /// `no-appliance`).
5522 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5523 pub tolerates: Vec<String>,
5524 /// R572-F5: taint the workload requires the target node to carry
5525 /// (annotation `yah.placement.requires-taint`). The node must have the
5526 /// key in its `taints` list or `mesh_tags`. `None` = no affinity constraint.
5527 #[serde(skip)]
5528 pub requires_taint: Option<String>,
5529
5530 /// R844-F8: **how many** machines this constraint places onto. `None` — the
5531 /// only shape on disk before this field — means one, so every mirror in the
5532 /// tree resolves byte-identically across the change.
5533 ///
5534 /// This is not a match axis: it never appears in [`Self::matches`] and never
5535 /// changes whether a given machine qualifies. It is the *cardinality* of the
5536 /// answer, which is why it lives here rather than as another filter — the
5537 /// operator declares what is required and how many of it, and the scheduler
5538 /// picks which.
5539 ///
5540 /// **Declared, never inferred.** The count is emphatically not "how many
5541 /// machines happen to match": deriving it that way would make adding a box
5542 /// to the fleet silently scale a production front door. A constraint that
5543 /// matches four machines and asks for two places on two.
5544 ///
5545 /// **Fewer matches than asked is an error** ([`select_matching`]), not a
5546 /// partial placement. Placing one of two and reporting success is the
5547 /// subset-that-looks-like-it-worked failure R844 exists to close.
5548 ///
5549 /// Deliberately absent from [`Self::is_unconstrained`], which answers "does
5550 /// every machine match" — a question about the predicate, not the count. A
5551 /// `required = { replicas = 2 }` with no axis is therefore still
5552 /// unconstrained, and the deploy side still refuses it as an
5553 /// underspecified placement.
5554 #[serde(default, skip_serializing_if = "Option::is_none")]
5555 pub replicas: Option<u32>,
5556}
5557
5558impl RequiredSpec {
5559 /// How many machines this constraint places onto — [`Self::replicas`],
5560 /// resolving the absent case to the pre-R844-F8 answer of one.
5561 ///
5562 /// The single place that default is spelled, so the ingress planner and the
5563 /// deploy resolver cannot disagree about what "no replica count" means.
5564 pub fn replica_count(&self) -> usize {
5565 self.replicas.unwrap_or(1) as usize
5566 }
5567
5568 /// True when no *declared* axis carries a constraint — every untainted
5569 /// machine matches.
5570 ///
5571 /// R876-B7: taint repulsion is deliberately absent from this conjunction,
5572 /// unlike the `repel_archetypes` it replaced. Repulsion is no longer an axis
5573 /// a spec declares — it applies to every spec — so including it would make
5574 /// the answer a property of the fleet rather than of the constraint. Nor
5575 /// does [`Self::tolerates`] belong here: a toleration *widens* the candidate
5576 /// set, and the callers of this predicate ask "did the operator narrow
5577 /// anything" in order to refuse an underspecified placement. A slot that
5578 /// declares only a toleration has still narrowed nothing.
5579 pub fn is_unconstrained(&self) -> bool {
5580 self.regions.is_empty()
5581 && self.zones.is_empty()
5582 && self.providers.is_empty()
5583 && self.mesh_tags.is_empty()
5584 && self.nodes.is_empty()
5585 && self.memory_mb == 0
5586 && self.cpu_millis == 0
5587 && self.requires_taint.is_none()
5588 }
5589
5590 /// Whether `machine` satisfies every hard axis.
5591 ///
5592 /// - Membership axes (region/zone/provider): machine must carry the field
5593 /// and it must appear in the constraint list.
5594 /// - `mesh_tags`: machine tags must be a superset of the required set.
5595 /// - **R572-F5 capacity floor**: `machine.allocatable.{memory,cpu}` must
5596 /// cover `self.{memory,cpu}`. A machine with no `allocatable` block passes
5597 /// unconditionally (capacity unknown → no constraint enforced).
5598 /// - **Taint repulsion (R876-B7)**: machine must not carry *any* taint that
5599 /// [`taint_effect`] classifies as [`TaintEffect::Repels`], unless that key
5600 /// is listed in [`Self::tolerates`]. Applied unconditionally — this is the
5601 /// axis no spec has to declare, and the one that makes a node drainable.
5602 /// - **R572-F5 taint affinity**: if `requires_taint` is set, the machine
5603 /// must carry that key in its `taints` list or `mesh_tags`.
5604 ///
5605 /// A [`TaintEffect::Attracts`] key (today just `public-ip`) does **not**
5606 /// repel: it is the affinity vocabulary, so reading it as repulsion would
5607 /// evict every workload from the three nodes that carry it. Only the
5608 /// `no-<archetype>` class repels, and [`taint_effect`] is the single
5609 /// authority on which is which — which is why
5610 /// [`crate::validate::check_inert_taints`] refuses to let an unclassifiable
5611 /// key be declared: it would read as a constraint and be none.
5612 pub fn matches(&self, machine: &MachineConfig) -> bool {
5613 let member_ok = |constraint: &[String], value: Option<&str>| -> bool {
5614 constraint.is_empty() || value.map_or(false, |v| constraint.iter().any(|c| c == v))
5615 };
5616
5617 // R833-F8: imperative node pin, checked first because it is the axis a
5618 // human asserted rather than one the scheduler derived — a refusal
5619 // should read "us-west-003 does not match" and not lead with a tag set
5620 // the operator never typed.
5621 if !member_ok(&self.nodes, Some(machine.name.as_str())) {
5622 return false;
5623 }
5624
5625 // Membership + mesh-tags (pre-existing axes).
5626 if !member_ok(&self.regions, machine.region.as_deref())
5627 || !member_ok(&self.zones, machine.zone.as_deref())
5628 || !member_ok(&self.providers, Some(machine.provider.as_str()))
5629 || !self
5630 .mesh_tags
5631 .iter()
5632 .all(|t| machine.mesh_tags.iter().any(|mt| mt == t))
5633 {
5634 return false;
5635 }
5636
5637 // R572-F5: capacity floor. Skipped when machine has no allocatable
5638 // declaration (unknown capacity → passes, consistent with pre-F5 behaviour).
5639 if self.memory_mb > 0 || self.cpu_millis > 0 {
5640 if let Some(alloc) = &machine.allocatable {
5641 if self.memory_mb > alloc.memory_mb || self.cpu_millis > alloc.cpu_millis {
5642 return false;
5643 }
5644 }
5645 }
5646
5647 // R876-B7: taint repulsion, repel-by-default. Every repelling taint on
5648 // the machine blocks placement unless this spec names it in
5649 // `tolerates`. Driven off `machine.taints` rather than off a field of
5650 // `self`, which is the whole point: a spec that arrives by deserializing
5651 // a mirror's `required = {...}` carries no repulsion declaration and
5652 // never could, so making repulsion conditional on one made node taints
5653 // structurally unreadable on that path (R876-S2).
5654 for taint in &machine.taints {
5655 if !matches!(taint_effect(taint), TaintEffect::Repels(_)) {
5656 continue;
5657 }
5658 if !self.tolerates.iter().any(|t| t == taint) {
5659 return false;
5660 }
5661 }
5662
5663 // R572-F5: taint affinity. Machine must carry the required taint key
5664 // in either its `taints` list or `mesh_tags`.
5665 if let Some(req) = &self.requires_taint {
5666 let has_it = machine.taints.iter().any(|t| t == req)
5667 || machine.mesh_tags.iter().any(|t| t == req);
5668 if !has_it {
5669 return false;
5670 }
5671 }
5672
5673 true
5674 }
5675
5676 /// Human-readable summary of the constraints, for fail-loud error messages.
5677 /// Example: `required.regions=[us-west] + required.mesh_tags=[tag:cloud-runner]`.
5678 pub fn describe(&self) -> String {
5679 let mut parts = Vec::new();
5680 let mut push = |label: &str, vals: &[String]| {
5681 if !vals.is_empty() {
5682 parts.push(format!("required.{label}=[{}]", vals.join(",")));
5683 }
5684 };
5685 push("nodes", &self.nodes);
5686 push("regions", &self.regions);
5687 push("zones", &self.zones);
5688 push("providers", &self.providers);
5689 push("mesh_tags", &self.mesh_tags);
5690 // Kept with the other list axes, and NOT moved below: `push` borrows
5691 // `parts` mutably for as long as it is live, so interleaving it with the
5692 // direct `parts.push` calls under it does not compile.
5693 push("tolerates", &self.tolerates);
5694 if self.memory_mb > 0 {
5695 parts.push(format!("memory_mb>={}", self.memory_mb));
5696 }
5697 if self.cpu_millis > 0 {
5698 parts.push(format!("cpu_millis>={}", self.cpu_millis));
5699 }
5700 if let Some(req) = &self.requires_taint {
5701 parts.push(format!("requires_taint={req}"));
5702 }
5703 if parts.is_empty() {
5704 "no constraints".to_string()
5705 } else {
5706 parts.join(" + ")
5707 }
5708 }
5709}
5710
5711/// Which front door actually serves a domain's requests (R594-F12).
5712///
5713/// Every domain manifest must say this out loud. Before it existed the
5714/// difference between "R2 serves this hostname directly" and "a Worker
5715/// serves it" was expressed *only* by whether the file happened to carry
5716/// `[[routes]]` — so binding a route-carrying domain straight to R2 was
5717/// accepted silently and served 200s on its SSG half while losing clean
5718/// URLs, SPA shell fallback, deferred-route pointers and branded error
5719/// pages. All of those live in the Worker
5720/// (`oss/mesofact/packages/mesofact-edge/src/router.ts`) or in
5721/// mesofact-serve; an R2 custom domain has none of them.
5722///
5723/// The vocabulary mirrors `scripts/cf-apex-mode.sh` (worker | grey | orange)
5724/// — this moves the choice into the config where it can be checked instead
5725/// of living in one bash script.
5726///
5727/// A front door does **fan-in** only. The render cube (SSG / SPA / SSR /
5728/// deferred / 404) is mesofact's manifest, not this one — see W173 and
5729/// `.yah/docs/working/W267-sovereign-public-ingress.md`
5730/// §"Two front doors, one render contract".
5731#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
5732#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5733#[serde(rename_all = "kebab-case")]
5734pub enum FrontDoor {
5735 /// Cloudflare R2 custom domain. Requests hit R2 objects with edge
5736 /// caching and nothing else — no clean URLs, no SPA fallback, no
5737 /// branded errors. Correct for a pure asset tier (W175's verdict for
5738 /// `cdn.yah.dev`) and wrong for anything that renders pages.
5739 /// Implies zero `[[routes]]` and no `worker_bundle_path`.
5740 BucketDirect,
5741 /// Cloudflare Worker generated from this manifest's route table.
5742 Worker,
5743 /// Sovereign L7 ingress — the `passway` proxy on yah-owned metal
5744 /// (`oss/passway`, W267). Same route table as `worker`; different
5745 /// machine terminates TLS.
5746 Passway,
5747}
5748
5749impl FrontDoor {
5750 /// Whether this front door consumes the manifest's `[[routes]]` table.
5751 /// `bucket-direct` does not; the other two are nothing without it.
5752 pub fn is_route_driven(self) -> bool {
5753 matches!(self, FrontDoor::Worker | FrontDoor::Passway)
5754 }
5755
5756 /// The manifest spelling, for error messages.
5757 pub fn as_str(self) -> &'static str {
5758 match self {
5759 FrontDoor::BucketDirect => "bucket-direct",
5760 FrontDoor::Worker => "worker",
5761 FrontDoor::Passway => "passway",
5762 }
5763 }
5764}
5765
5766/// A routing manifest for one domain, from `.yah/domains/<name>.toml`.
5767///
5768/// The domain manifest is the *only* place that knows about path routing:
5769/// services declare static/backend components by opaque ID, and this
5770/// manifest binds those components to URL paths on a public-facing
5771/// domain. Generated Worker bundles consume this. See
5772/// `.yah/docs/working/W118-yah-domain-tiers.md` (R347).
5773#[derive(Debug, Clone, Serialize, Deserialize)]
5774#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5775pub struct DomainConfig {
5776 pub schema_version: u32,
5777 /// Stable identifier for this domain (file stem of the manifest).
5778 /// Example: `"yah-dev"` for the `yah.dev` zone.
5779 pub name: String,
5780 /// The fully-qualified domain this manifest routes for. Example:
5781 /// `"yah.dev"`, `"app.yah.dev"`.
5782 pub domain: String,
5783 /// Which front door serves this domain (R594-F12). **Required** — a
5784 /// default here would silently re-create the defect the field exists to
5785 /// close. Cross-checked against `routes` / `worker_bundle_path` by
5786 /// [`DomainConfig::validate_front_door`] at load time.
5787 pub front_door: FrontDoor,
5788 /// Public CDN bucket name. Static-mode route components publish into
5789 /// this bucket. Owned by the domain, *not* by any single service.
5790 pub cdn_bucket: String,
5791 /// Optional path (relative to workspace root) where the generated
5792 /// Worker bundle lands. `None` while the bundle generator (R347-F4)
5793 /// is still being wired up.
5794 #[serde(default, skip_serializing_if = "Option::is_none")]
5795 pub worker_bundle_path: Option<String>,
5796 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5797 pub routes: Vec<DomainRoute>,
5798}
5799
5800/// One entry in a [`DomainConfig`]'s route table.
5801///
5802/// The `mode` discriminator picks the variant's body via serde's
5803/// internally-tagged enum representation. Path patterns follow the
5804/// Worker convention: a trailing `*` matches everything underneath.
5805#[derive(Debug, Clone, Serialize, Deserialize)]
5806#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5807pub struct DomainRoute {
5808 /// URL pattern this route matches. Examples: `"/"`, `"/dashboard/*"`,
5809 /// `"/camp/ws"`.
5810 pub path: String,
5811 /// Response headers the front door sets on every response served under
5812 /// this route (R746). Empty by default.
5813 ///
5814 /// This is the manifest's answer to "who decides a path's response
5815 /// headers". Before it existed the answer was *nobody*: a `_headers` file
5816 /// is a Cloudflare Pages / Netlify convention, and neither of this
5817 /// repo's front doors reads one — a Worker returns what it fetched from
5818 /// R2, and R2 serves only the object's own httpMetadata. So a site could
5819 /// carry a `_headers` file declaring COOP/COEP and ship without them,
5820 /// which is exactly how it was found: `SharedArrayBuffer` is simply
5821 /// absent in a document served cross-origin-isolation-free, with no
5822 /// error anywhere to say why.
5823 ///
5824 /// Deliberately a free-form `name -> value` map rather than named fields
5825 /// for the isolation headers: the domain manifest has no business
5826 /// knowing which headers a route's payload happens to need. Ordering
5827 /// follows the route table's own rule — first matching route wins, no
5828 /// merging across routes (see the Worker's `applyRouteHeaders`).
5829 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
5830 pub headers: BTreeMap<String, String>,
5831 #[serde(flatten)]
5832 pub mode: RouteMode,
5833}
5834
5835/// Body of a [`DomainRoute`]. Three modes on the wire, four shapes here:
5836/// - **Static** (`mode = "static"`, `component = …`) — the door serves a
5837/// published component's bytes from its CDN prefix. Component ref points at
5838/// a `kind = "mesofact-static"` (or similar) service component.
5839/// - **StaticBucket** (`mode = "static"`, `bucket = …`, R560-F13) — the door
5840/// serves an R2 bucket's ROOT, keyed by the request path minus its leading
5841/// slash with nothing stripped or prepended. For a CDN whose route prefixes
5842/// ARE its buckets' top-level key prefixes (cdn.noisetable.com), where xlb
5843/// derives a blob's key from the very URL path it serves at, so path == key
5844/// is an invariant rather than a convention.
5845/// - **Backend** — Worker proxies to an HTTP origin owned by a backend
5846/// component (yubaba workload, gateway, etc.).
5847/// - **Redirect** — Worker emits a 30x to the target URL. Used to keep
5848/// old paths alive during domain refactors.
5849///
5850/// The two static shapes are separate variants rather than one variant with
5851/// two optional fields, so "both" and "neither" have no spelling in Rust. The
5852/// TOML keeps one `mode = "static"` and the choice between `component` and
5853/// `bucket` is refused at parse time unless exactly one is present — see
5854/// [`RouteModeWire`].
5855#[derive(Debug, Clone, Serialize, Deserialize)]
5856#[serde(try_from = "RouteModeWire", into = "RouteModeWire")]
5857pub enum RouteMode {
5858 Static {
5859 /// Component reference `"<service>/<component-id>"`. Validated
5860 /// at [`CloudConfig::load`] time.
5861 component: String,
5862 },
5863 StaticBucket {
5864 /// R2 bucket name. Validated against R2's naming rule at parse time,
5865 /// which is also what makes [`crate::route_table::r2_binding_name`]
5866 /// injective.
5867 bucket: String,
5868 },
5869 Backend {
5870 /// Component reference `"<service>/<component-id>"`. Validated
5871 /// at [`CloudConfig::load`] time.
5872 component: String,
5873 /// Origin URL the Worker `fetch()`es. Schema-permissive — could
5874 /// be `https://...`, `wss://...`, or a yah-internal mesh URL
5875 /// resolved by yubaba.
5876 origin: String,
5877 /// The path prefix this route's `path` becomes at the origin, when the
5878 /// two differ (R898-F3). `/api/issues*` with `origin_path = "/issues"`
5879 /// proxies `/api/issues/42` to `<origin>/issues/42`.
5880 ///
5881 /// Absent means an identity proxy — the public path reaches the origin
5882 /// unchanged. It is declared here rather than derived because it is a
5883 /// fact about the *upstream's* path layout, which this repo does not
5884 /// own: the issue tracker serves `/issues` and the almanac serves
5885 /// `/releases`, and both are live contracts that predate the domain
5886 /// manifest. Compiled into [`RouteRewrite`](crate::route_table::RouteRewrite)
5887 /// by [`DomainConfig::route_table`](crate::route_table).
5888 #[serde(default, skip_serializing_if = "Option::is_none")]
5889 origin_path: Option<String>,
5890 },
5891 Redirect {
5892 /// Absolute URL or path the Worker emits a 30x to.
5893 target: String,
5894 /// HTTP status code. Defaults to 308 (permanent + method-preserving)
5895 /// so deprecations don't silently turn POSTs into GETs.
5896 #[serde(default = "default_redirect_status")]
5897 status: u16,
5898 },
5899}
5900
5901fn default_redirect_status() -> u16 {
5902 308
5903}
5904
5905/// The TOML/JSON shape of a [`RouteMode`]: one `mode = "static"` whose body
5906/// names a `component` OR a `bucket`.
5907///
5908/// Private to parsing. The two optional fields exist only here, and
5909/// `TryFrom` turns them into exactly one [`RouteMode`] variant or refuses the
5910/// manifest naming both keys — so the contradictory route never reaches a
5911/// consumer. It is also the JSON schema's source, since the schema describes
5912/// what an author writes.
5913#[derive(Debug, Clone, Serialize, Deserialize)]
5914#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5915#[serde(tag = "mode", rename_all = "kebab-case")]
5916enum RouteModeWire {
5917 Static {
5918 /// Component reference `"<service>/<component-id>"`: serve that
5919 /// component's published bytes. Exactly one of `component` / `bucket`.
5920 #[serde(default, skip_serializing_if = "Option::is_none")]
5921 component: Option<String>,
5922 /// R2 bucket name: serve the bucket ROOT, the key being the request
5923 /// path minus its leading slash with nothing stripped or prepended
5924 /// (R560-F13). Requires `front_door = "worker"`. Exactly one of
5925 /// `component` / `bucket`.
5926 #[serde(default, skip_serializing_if = "Option::is_none")]
5927 bucket: Option<String>,
5928 },
5929 // Field docs below are the schema's text; they restate [`RouteMode`]'s.
5930 Backend {
5931 /// Component reference `"<service>/<component-id>"`. Validated
5932 /// at [`CloudConfig::load`] time.
5933 component: String,
5934 /// Origin URL the Worker `fetch()`es. Schema-permissive — could
5935 /// be `https://...`, `wss://...`, or a yah-internal mesh URL
5936 /// resolved by yubaba.
5937 origin: String,
5938 /// The path prefix this route's `path` becomes at the origin, when the
5939 /// two differ (R898-F3). `/api/issues*` with `origin_path = "/issues"`
5940 /// proxies `/api/issues/42` to `<origin>/issues/42`.
5941 ///
5942 /// Absent means an identity proxy — the public path reaches the origin
5943 /// unchanged. It is declared here rather than derived because it is a
5944 /// fact about the *upstream's* path layout, which this repo does not
5945 /// own: the issue tracker serves `/issues` and the almanac serves
5946 /// `/releases`, and both are live contracts that predate the domain
5947 /// manifest. Compiled into [`RouteRewrite`](crate::route_table::RouteRewrite)
5948 /// by [`DomainConfig::route_table`](crate::route_table).
5949 #[serde(default, skip_serializing_if = "Option::is_none")]
5950 origin_path: Option<String>,
5951 },
5952 Redirect {
5953 /// Absolute URL or path the Worker emits a 30x to.
5954 target: String,
5955 /// HTTP status code. Defaults to 308 (permanent + method-preserving)
5956 /// so deprecations don't silently turn POSTs into GETs.
5957 #[serde(default = "default_redirect_status")]
5958 status: u16,
5959 },
5960}
5961
5962impl TryFrom<RouteModeWire> for RouteMode {
5963 type Error = String;
5964
5965 fn try_from(wire: RouteModeWire) -> std::result::Result<Self, String> {
5966 Ok(match wire {
5967 RouteModeWire::Static {
5968 component: Some(component),
5969 bucket: None,
5970 } => Self::Static { component },
5971 RouteModeWire::Static {
5972 component: None,
5973 bucket: Some(bucket),
5974 } => {
5975 validate_r2_bucket_name(&bucket)?;
5976 Self::StaticBucket { bucket }
5977 }
5978 RouteModeWire::Static {
5979 component: Some(component),
5980 bucket: Some(bucket),
5981 } => {
5982 return Err(format!(
5983 "a static route declares both component = \"{component}\" and bucket = \
5984 \"{bucket}\" — declare exactly one: `component` serves a published \
5985 component from its CDN prefix, `bucket` serves an R2 bucket root with \
5986 the request path as the key"
5987 ))
5988 }
5989 RouteModeWire::Static {
5990 component: None,
5991 bucket: None,
5992 } => {
5993 return Err("a static route declares neither `component` nor `bucket` — \
5994 declare exactly one"
5995 .to_string())
5996 }
5997 RouteModeWire::Backend {
5998 component,
5999 origin,
6000 origin_path,
6001 } => Self::Backend {
6002 component,
6003 origin,
6004 origin_path,
6005 },
6006 RouteModeWire::Redirect { target, status } => Self::Redirect { target, status },
6007 })
6008 }
6009}
6010
6011impl From<RouteMode> for RouteModeWire {
6012 fn from(mode: RouteMode) -> Self {
6013 match mode {
6014 RouteMode::Static { component } => Self::Static {
6015 component: Some(component),
6016 bucket: None,
6017 },
6018 RouteMode::StaticBucket { bucket } => Self::Static {
6019 component: None,
6020 bucket: Some(bucket),
6021 },
6022 RouteMode::Backend {
6023 component,
6024 origin,
6025 origin_path,
6026 } => Self::Backend {
6027 component,
6028 origin,
6029 origin_path,
6030 },
6031 RouteMode::Redirect { target, status } => Self::Redirect { target, status },
6032 }
6033 }
6034}
6035
6036// Hand-written only because schemars 0.8 does not follow `#[serde(try_from)]`:
6037// a derive would describe the Rust variants (`static-bucket`), which no
6038// manifest may spell. The schema is the wire shape an author writes.
6039#[cfg(feature = "json-schema")]
6040impl schemars::JsonSchema for RouteMode {
6041 fn schema_name() -> String {
6042 "RouteMode".to_string()
6043 }
6044
6045 fn json_schema(gen: &mut schemars::gen::SchemaGenerator) -> schemars::schema::Schema {
6046 <RouteModeWire as schemars::JsonSchema>::json_schema(gen)
6047 }
6048}
6049
6050/// R2's bucket naming rule: 3–63 characters of lowercase letters, digits and
6051/// hyphens, starting and ending with a letter or digit.
6052///
6053/// Checked at parse time rather than left to the deploy's 400, and load-bearing
6054/// beyond that: the Worker binding name derived from a bucket
6055/// ([`crate::route_table::r2_binding_name`]) only maps `-` to `_`, which is
6056/// collision-free exactly because a bucket name can carry no `_` of its own.
6057fn validate_r2_bucket_name(bucket: &str) -> std::result::Result<(), String> {
6058 let valid_chars = bucket
6059 .bytes()
6060 .all(|b| b.is_ascii_lowercase() || b.is_ascii_digit() || b == b'-');
6061 let valid_ends = bucket.starts_with(|c: char| c.is_ascii_alphanumeric())
6062 && bucket.ends_with(|c: char| c.is_ascii_alphanumeric());
6063 if (3..=63).contains(&bucket.len()) && valid_chars && valid_ends {
6064 Ok(())
6065 } else {
6066 Err(format!(
6067 "bucket = \"{bucket}\" is not an R2 bucket name — 3 to 63 characters of \
6068 lowercase letters, digits and hyphens, starting and ending with a letter or digit"
6069 ))
6070 }
6071}
6072
6073/// Normalize a component `mount` to a storage/URL key prefix: strip the
6074/// surrounding slashes. `"/app"`, `"app/"`, `"/app/"` → `"app"`; `"/"`, `""`
6075/// → `""` (the service root).
6076///
6077/// One producer on purpose — the publisher's key prefix, the route-path
6078/// cross-check and the front door's key lookup must all agree on what `/app`
6079/// means down to the byte, and three copies of `trim_matches('/')` is how they
6080/// stop agreeing.
6081pub fn normalize_mount(raw: &str) -> String {
6082 raw.trim_matches('/').to_string()
6083}
6084
6085/// The key prefix a domain route pattern serves under: `"/*"` → `""`,
6086/// `"/app/*"` and `"/app"` → `"app"`. The twin of [`normalize_mount`] on the
6087/// routing side.
6088pub fn route_path_prefix(path: &str) -> String {
6089 normalize_mount(path.strip_suffix('*').unwrap_or(path))
6090}
6091
6092/// The route-driven domain whose route table binds a component of `service`,
6093/// if any. Used by static publishers to pick up the per-route response
6094/// headers a service's paths were declared with.
6095///
6096/// Deterministic by `BTreeMap` key order when more than one domain routes the
6097/// same service (a legitimate shape: an apex and a staging host serving one
6098/// bundle). Returning the first is a real limitation, not a considered
6099/// choice — the day two such domains want *different* headers for one
6100/// component, this needs the domain identity threaded in rather than inferred.
6101pub fn domain_serving_service<'a>(
6102 domains: &'a BTreeMap<String, DomainConfig>,
6103 service: &str,
6104) -> Option<&'a DomainConfig> {
6105 domains
6106 .values()
6107 .find(|d| d.front_door.is_route_driven() && d.serves_service(service))
6108}
6109
6110/// The `ROUTE_HEADERS` Worker-binding value for `service`, read from the
6111/// workspace's domain manifests. `"[]"` when no route-driven domain routes the
6112/// service, or when the one that does declares no headers.
6113///
6114/// R898-F1 widened this into
6115/// [`route_table_for_service`](crate::route_table::route_table_for_service):
6116/// same lookup, same `.yah/domains/` read, the whole compiled table
6117/// (`{path, mode, resolved origin, headers, auth}`) instead of its header
6118/// column. This spelling survives because it is what the *bindings* carry until
6119/// R898-F2/F3 widen those consumers — and because it needs no placement, which
6120/// the reconcilers calling it do not have.
6121///
6122/// Reads `.yah/domains/` directly rather than taking a loaded [`CloudConfig`]:
6123/// the static reconcilers are handed a per-component [`ReconcileCtx`], not the
6124/// whole workspace config, and threading a config reference through all 22 of
6125/// its construction sites to reach one string would be a wide change for a
6126/// narrow read. Manifest parse errors propagate — a domain file that no longer
6127/// loads is a deploy-stopping fact, not a reason to ship a Worker with the
6128/// headers quietly missing.
6129pub fn route_headers_for_service(workspace_root: &Path, service: &str) -> Result<String> {
6130 Ok(domain_for_service(workspace_root, service)?
6131 .as_ref()
6132 .map(DomainConfig::route_headers_json)
6133 .unwrap_or_else(|| "[]".to_string()))
6134}
6135
6136/// The route-driven domain manifest serving `service`, loaded from
6137/// `.yah/domains/`.
6138///
6139/// The whole manifest rather than one projection of it, for the consumer that
6140/// needs more than the header column: R898-F3's Worker reconciler compiles the
6141/// full route table (`{path, mode, origin, rewrite, headers, auth}`) and cannot
6142/// re-derive modes and origins from `route_headers_json`'s output. Same lookup
6143/// [`route_headers_for_service`] makes — it is now a caller of this.
6144pub fn domain_for_service(workspace_root: &Path, service: &str) -> Result<Option<DomainConfig>> {
6145 let domains = load_domains(&crate::paths::domains_dir(workspace_root))?;
6146 Ok(domain_serving_service(&domains, service).cloned())
6147}
6148
6149impl DomainConfig {
6150 /// Parse a single `.yah/domains/<name>.toml`, rejecting a manifest whose
6151 /// declared front door contradicts its route table
6152 /// ([`Self::validate_front_door`]).
6153 pub fn load(path: &Path) -> Result<Self> {
6154 let src =
6155 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
6156 let dom: Self =
6157 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
6158 dom.validate_front_door()
6159 .with_context(|| format!("validating {}", path.display()))?;
6160 dom.validate_route_headers()
6161 .with_context(|| format!("validating {}", path.display()))?;
6162 Ok(dom)
6163 }
6164
6165 /// R594-F12 — the front door must agree with the rest of the manifest.
6166 ///
6167 /// - `bucket-direct` is an R2 custom domain: a Worker route table would
6168 /// never be consulted, so declaring one means the author expected
6169 /// Worker behaviour (clean URLs, SPA fallback, branded errors) from a
6170 /// surface that cannot provide it. Rejected rather than silently
6171 /// ignored. Same for `worker_bundle_path` — nothing would deploy it.
6172 /// - `worker` / `passway` with an empty route table is a silent 404
6173 /// machine: the front door exists, has nothing to serve, and every
6174 /// request falls through to the catch-all.
6175 ///
6176 /// Called from [`Self::load`], so both [`CloudConfig::load`] and
6177 /// [`CloudConfig::load_from_config_dir`] enforce it.
6178 pub fn validate_front_door(&self) -> Result<()> {
6179 match self.front_door {
6180 FrontDoor::BucketDirect => {
6181 if let Some(route) = self.routes.first() {
6182 anyhow::bail!(
6183 "front_door = \"bucket-direct\" but routes[0].path = \"{}\" — \
6184 an R2 custom domain never consults a route table, so this \
6185 route would silently do nothing (no clean URLs, no SPA \
6186 fallback, no branded errors). Set front_door = \"worker\" \
6187 (or \"passway\") to keep the routes, or drop the [[routes]] \
6188 to keep the bucket-direct binding.",
6189 route.path
6190 );
6191 }
6192 if let Some(path) = &self.worker_bundle_path {
6193 anyhow::bail!(
6194 "front_door = \"bucket-direct\" but worker_bundle_path = \
6195 \"{path}\" — nothing deploys a Worker bundle for a domain \
6196 bound straight to R2"
6197 );
6198 }
6199 }
6200 FrontDoor::Worker | FrontDoor::Passway => {
6201 // R560-F13: a bucket route is read through a Worker R2 binding.
6202 // Passway has no R2 read path, so accepting one there would
6203 // compile an entry the door cannot serve — refused naming the
6204 // route instead.
6205 if self.front_door == FrontDoor::Passway {
6206 if let Some(route) = self
6207 .routes
6208 .iter()
6209 .find(|r| matches!(r.mode, RouteMode::StaticBucket { .. }))
6210 {
6211 anyhow::bail!(
6212 "front_door = \"passway\" but routes path = \"{}\" declares \
6213 `bucket` — a bucket route is served through a Cloudflare Worker \
6214 R2 binding, and passway has no R2 read path. Set front_door = \
6215 \"worker\", or serve the path from a published `component`.",
6216 route.path
6217 );
6218 }
6219 }
6220 if self.routes.is_empty() {
6221 anyhow::bail!(
6222 "front_door = \"{}\" but [[routes]] is empty — a front door \
6223 with no route table is a silent 404 machine. Declare at \
6224 least one route, or set front_door = \"bucket-direct\" if \
6225 this domain really is served straight from R2.",
6226 self.front_door.as_str()
6227 );
6228 }
6229 }
6230 }
6231 Ok(())
6232 }
6233
6234 // `route_headers_json` — the header column of the compiled route table —
6235 // lives in `crate::route_table` alongside `route_table`, `RouteTable` and
6236 // the one serializer both projections share (R898-F1).
6237
6238 /// R749-T5 — everything [`Self::route_headers_json`] emits must be
6239 /// *applicable*, checked here where the table is PRODUCED.
6240 ///
6241 /// That method serializes a typed struct, so the table's JSON *shape* is
6242 /// sound by construction. Its contents are not: a route's `headers` map is
6243 /// a free-form `name -> value` read verbatim out of hand-written TOML, so
6244 /// `"Cross Origin Opener Policy"` (spaces instead of hyphens) or a value
6245 /// carrying a newline ships a structurally-valid table that neither front
6246 /// door can apply — and they fail *differently*, neither naming the
6247 /// manifest line responsible:
6248 ///
6249 /// - **passway** — `mesofact::route_headers::RouteHeaderTable::parse`
6250 /// refuses the start, so the origin is simply down.
6251 /// - **worker** — `validateRouteHeaderTable` accepts it (it checks shape,
6252 /// not header validity) and `applyRouteHeaders` then throws inside the
6253 /// exported `fetch`, which is a 500 on every request, not the
6254 /// serve-without-the-headers degradation that code intends.
6255 ///
6256 /// So the strictness lives at the producer: a table that cannot be applied
6257 /// fails `yah cloud apply` at manifest load, naming domain, route and
6258 /// header. This is deliberately *not* a second parser — the check is
6259 /// `HeaderName`/`HeaderValue`'s own, the very constructors the passway door
6260 /// runs on the far side, and route *matching* semantics stay defined once,
6261 /// at the doors. Only routes that contribute to the table are checked, so
6262 /// the invariant is exactly "`route_headers_json`'s output parses".
6263 ///
6264 /// Called from [`Self::load`], alongside [`Self::validate_front_door`].
6265 pub fn validate_route_headers(&self) -> Result<()> {
6266 use axum::http::{HeaderName, HeaderValue};
6267
6268 for route in self.routes.iter().filter(|r| !r.headers.is_empty()) {
6269 if route.path.is_empty() {
6270 anyhow::bail!(
6271 "domain \"{}\" declares response headers on a route whose `path` is \
6272 empty — a rule that matches nothing (or everything, depending on \
6273 which front door reads it) is not a policy",
6274 self.name
6275 );
6276 }
6277 for (name, value) in &route.headers {
6278 HeaderName::try_from(name.as_str()).with_context(|| {
6279 format!(
6280 "domain \"{}\" route \"{}\" declares {name:?}, which is not a valid \
6281 HTTP header name — names are token characters only, so it is \
6282 `Cross-Origin-Opener-Policy`, never `Cross Origin Opener Policy`",
6283 self.name, route.path
6284 )
6285 })?;
6286 HeaderValue::try_from(value.as_str()).with_context(|| {
6287 format!(
6288 "domain \"{}\" route \"{}\" declares {name} = {value:?}, which is not \
6289 a valid HTTP header value — no newlines and no control characters",
6290 self.name, route.path
6291 )
6292 })?;
6293 }
6294 }
6295 Ok(())
6296 }
6297
6298 /// Whether this domain's route table binds any component of `service`.
6299 pub fn serves_service(&self, service: &str) -> bool {
6300 self.routes.iter().any(|r| {
6301 r.mode
6302 .component()
6303 .and_then(split_component_ref)
6304 .is_some_and(|(svc, _)| svc == service)
6305 })
6306 }
6307
6308 /// Persist to `.yah/domains/<name>.toml`, creating the domains
6309 /// directory if needed. Create-or-overwrite.
6310 pub fn save(&self, workspace_root: &Path) -> Result<()> {
6311 let dir = crate::paths::domains_dir(workspace_root);
6312 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
6313 let path = crate::paths::domain_toml(workspace_root, &self.name);
6314 let s = toml::to_string_pretty(self)
6315 .with_context(|| format!("serializing domain {}", self.name))?;
6316 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
6317 }
6318
6319 /// Remove `.yah/domains/<name>.toml`. Returns `false` when the file
6320 /// was already absent.
6321 pub fn delete(workspace_root: &Path, name: &str) -> Result<bool> {
6322 let path = crate::paths::domain_toml(workspace_root, name);
6323 if !path.exists() {
6324 return Ok(false);
6325 }
6326 std::fs::remove_file(&path).with_context(|| format!("removing {}", path.display()))?;
6327 Ok(true)
6328 }
6329}
6330
6331impl RouteMode {
6332 /// Component reference for static/backend modes; `None` for redirects and
6333 /// bucket-root static routes, which reference no component.
6334 pub fn component(&self) -> Option<&str> {
6335 match self {
6336 Self::Static { component } | Self::Backend { component, .. } => Some(component),
6337 Self::StaticBucket { .. } | Self::Redirect { .. } => None,
6338 }
6339 }
6340}
6341
6342// ─── Service-group vault (R706 / W294) ───────────────────────────────────────
6343
6344/// A camp's declaration of one cluster secret, from
6345/// `.yah/infra/secrets/<slug>.toml`.
6346///
6347/// This is the *authoring* side of the fleet's cluster-secret store: it names
6348/// where the value lives in the camp (a `fob` vault slot), what the fleet should
6349/// call it, and — the point of R706 — which workloads are allowed to mount it.
6350///
6351/// The declaration is not itself the enforcement point. `yah cloud secret put`
6352/// reads this file, seals the vault value under the cluster KEK, and ships the
6353/// ciphertext **with its access rule** into raft; yubaba's `ClusterResolver`
6354/// evaluates the rule on the node at mount time. Deleting this file does not
6355/// revoke anything — the record in raft is the live authority. That asymmetry is
6356/// deliberate: a rule that lived only in a git-tracked camp file would be
6357/// trivially bypassed by anyone who could reach the fleet without the camp.
6358///
6359/// ```toml
6360/// #:schema ../../schema/secret.toml.schema.json
6361/// schema_version = 2
6362/// name = "cheers/cloud-admin/verify-key"
6363/// vault_slot = "cheers-cloud-admin-verify-key"
6364/// groups = ["prod"]
6365/// description = "Ed25519 public key yah-cloud-admin verifies operator PASETOs with"
6366///
6367/// [access]
6368/// workloads = [{ workload = "yah-cloud-admin" }]
6369///
6370/// [target]
6371/// kind = "file"
6372/// path = "/run/secrets/cheers-verify.key"
6373/// mode = 0o400
6374/// ```
6375#[derive(Debug, Clone, Serialize, Deserialize)]
6376#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
6377pub struct SecretConfig {
6378 pub schema_version: u32,
6379
6380 /// Logical cluster-secret key, as `SecretRef::Cluster { name }` spells it —
6381 /// e.g. `"tls/yah.dev/cert"`, `"cheers/cloud-admin/verify-key"`. May contain
6382 /// `/`; the file stem is a filesystem-safe slug and carries no meaning.
6383 pub name: String,
6384
6385 /// The `fob` vault slot in this camp holding the plaintext value. Read by
6386 /// `yah cloud secret put` at ship time and never recorded anywhere else — in
6387 /// particular the value is not in this file, so the declaration is safe to
6388 /// commit.
6389 pub vault_slot: String,
6390
6391 /// The sovereign groups this secret belongs to (R911-F8). Required and
6392 /// non-empty; each entry must be a `sovereign_group` that some
6393 /// `.yah/infra/machines/*.toml` declares ([`SecretConfig::load`] checks).
6394 ///
6395 /// Cluster secrets live per group in the fleet object store
6396 /// (`secrets/<group>/<name>.sealed`), so a declaration with no group has
6397 /// nowhere to land and nothing to be compared against. `yah cloud secret
6398 /// put` refuses a node whose `/raft/status` group is not listed here, and
6399 /// `status` counts a declaration only against nodes in one of these groups.
6400 pub groups: Vec<String>,
6401
6402 /// Human note for `yah cloud secret ls`. What this secret is and who minted
6403 /// it — the thing nobody remembers 6 months later.
6404 #[serde(default, skip_serializing_if = "Option::is_none")]
6405 pub description: Option<String>,
6406
6407 /// How the vault slot's text decodes into the bytes the consumer expects.
6408 ///
6409 /// `fob` slots hold strings, but plenty of real secrets are **binary** — an
6410 /// Ed25519 key is exactly 32 raw bytes, and `yah-cloud-admin` rejects a key
6411 /// file of any other length. Without this field the only way to ship such a
6412 /// key would be to hope its bytes happened to be valid UTF-8, which for a
6413 /// random key they are not.
6414 ///
6415 /// Defaults to [`SecretEncoding::Utf8`] — the right answer for tokens,
6416 /// passwords, and PEM, which is most secrets.
6417 #[serde(default)]
6418 pub encoding: SecretEncoding,
6419
6420 /// Who may mount it. Stamped onto the raft record verbatim.
6421 ///
6422 /// Defaults to [`SecretAccess::default`] — the deny-all empty allow-list. A
6423 /// declaration that forgets this field produces a secret nobody can mount,
6424 /// which is the correct direction to fail in.
6425 ///
6426 /// Three forms:
6427 ///
6428 /// ```toml
6429 /// access = "allow_any" # explicit escape hatch
6430 ///
6431 /// [access] # named workloads
6432 /// workloads = [{ workload = "yah-cloud-admin" }]
6433 ///
6434 /// [access] # signed recipes (R555-F5)
6435 /// recipes = [{ recipe = "rusty-v8-musl", key = "3d40…" }]
6436 /// ```
6437 ///
6438 /// Use the `recipes` form for a credential a **dispatched build** needs (the
6439 /// R2 write key, the cosign signing key). A remote QED run's workload name
6440 /// is a fresh `forge-<uuid>` every time, so `workloads` cannot name it and
6441 /// `allow_any` over-answers — see W235 §Seam (c) secret scoping. `key` is
6442 /// the hex Ed25519 public key from the recipe's `[admission]` block.
6443 #[serde(default)]
6444 pub access: SecretAccess,
6445
6446 /// Advisory: the mount shape a consuming workload should declare. Not
6447 /// enforced — yubaba honours whatever the `WorkloadSpec` asks for — but it
6448 /// lets `yah cloud secret put` print the exact `SecretMount` to paste, so
6449 /// the consumer and the declaration can't drift on path or mode.
6450 #[serde(default, skip_serializing_if = "Option::is_none")]
6451 pub target: Option<SecretTargetDecl>,
6452}
6453
6454/// How a [`SecretConfig`]'s vault text becomes the bytes delivered to the
6455/// container (R706 / W294).
6456#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
6457#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
6458#[serde(rename_all = "kebab-case")]
6459pub enum SecretEncoding {
6460 /// Ship the vault string's UTF-8 bytes verbatim. Tokens, passwords, PEM.
6461 #[default]
6462 Utf8,
6463 /// The vault string is hex; ship the decoded bytes. Use for binary key
6464 /// material — e.g. a raw Ed25519 key, which must land as exactly 32 bytes.
6465 Hex,
6466}
6467
6468/// Advisory mount shape on a [`SecretConfig`]. Mirrors
6469/// `workload_spec::SecretTarget` in a TOML-friendly, externally-tagged-free
6470/// shape (a `kind` discriminator reads better in a hand-written manifest than
6471/// serde's default enum encoding).
6472#[derive(Debug, Clone, Serialize, Deserialize)]
6473#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
6474#[serde(tag = "kind", rename_all = "kebab-case")]
6475pub enum SecretTargetDecl {
6476 /// Mounted as a tmpfs-backed file inside the container.
6477 File {
6478 /// Absolute path inside the container.
6479 path: String,
6480 /// Unix permission bits. Defaults to `0o400` (owner-read-only).
6481 #[serde(default = "default_secret_mode")]
6482 mode: u32,
6483 },
6484 /// Injected as an environment variable. Prefer `file` — env vars leak
6485 /// through subprocess environments and log dumps.
6486 EnvVar { name: String },
6487}
6488
6489fn default_secret_mode() -> u32 {
6490 0o400
6491}
6492
6493impl SecretTargetDecl {
6494 /// The `workload_spec` target this declaration describes.
6495 pub fn to_target(&self) -> workload_spec::SecretTarget {
6496 match self {
6497 Self::File { path, mode } => workload_spec::SecretTarget::File {
6498 path: path.into(),
6499 mode: *mode,
6500 },
6501 Self::EnvVar { name } => workload_spec::SecretTarget::EnvVar { name: name.clone() },
6502 }
6503 }
6504}
6505
6506/// The `schema_version` every [`SecretConfig`] must carry. Version 2 added the
6507/// required `groups` (R911-F8).
6508pub const SECRET_CONFIG_SCHEMA_VERSION: u32 = 2;
6509
6510impl SecretConfig {
6511 /// Parse a single `.yah/infra/secrets/<slug>.toml`.
6512 ///
6513 /// `sovereign_groups` is the camp's group vocabulary
6514 /// ([`CloudConfig::declared_sovereign_groups`]); every entry in `groups`
6515 /// must be one of them.
6516 pub fn load(path: &Path, sovereign_groups: &[&str]) -> Result<Self> {
6517 let src =
6518 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
6519 let cfg: Self =
6520 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
6521 cfg.validate()
6522 .and_then(|()| cfg.validate_groups(sovereign_groups))
6523 .with_context(|| format!("validating {}", path.display()))?;
6524 Ok(cfg)
6525 }
6526
6527 /// Load every declaration in `dir`, keyed by logical secret name. A missing
6528 /// directory is an empty map (a camp with no cluster secrets is normal).
6529 ///
6530 /// Two files declaring the same `name` is a hard error, not a last-writer-
6531 /// wins merge: they would race to define the access rule for one record, and
6532 /// whichever lost would look correct in git while being inert on the fleet.
6533 pub fn load_dir(dir: &Path, sovereign_groups: &[&str]) -> Result<BTreeMap<String, Self>> {
6534 let mut out: BTreeMap<String, Self> = BTreeMap::new();
6535 if !dir.exists() {
6536 return Ok(out);
6537 }
6538 for entry in std::fs::read_dir(dir).with_context(|| format!("reading {}", dir.display()))? {
6539 let path = entry?.path();
6540 if path.extension().is_none_or(|e| e != "toml") {
6541 continue;
6542 }
6543 let cfg = Self::load(&path, sovereign_groups)?;
6544 if let Some(prev) = out.insert(cfg.name.clone(), cfg) {
6545 anyhow::bail!(
6546 "two secret declarations both claim name {:?} (one of them is {}); \
6547 a cluster secret must have exactly one declaration so its access \
6548 rule has one author",
6549 prev.name,
6550 path.display()
6551 );
6552 }
6553 }
6554 Ok(out)
6555 }
6556
6557 /// Reject declarations that would produce an unusable or dangerous record.
6558 ///
6559 /// Structural only: whether each of `groups` names a real sovereign group
6560 /// needs the camp's machines, so that half is [`Self::validate_groups`].
6561 pub fn validate(&self) -> Result<()> {
6562 if self.schema_version != SECRET_CONFIG_SCHEMA_VERSION {
6563 anyhow::bail!(
6564 "`schema_version` is {}, expected {SECRET_CONFIG_SCHEMA_VERSION}: version 2 \
6565 added the required `groups = [\"<sovereign group>\", ...]` (R911-F8)",
6566 self.schema_version
6567 );
6568 }
6569 if self.name.trim().is_empty() {
6570 anyhow::bail!("`name` must not be empty");
6571 }
6572 if self.groups.is_empty() {
6573 anyhow::bail!(
6574 "`groups` must name at least one sovereign group: cluster secrets live per \
6575 group (secrets/<group>/), so a declaration in no group has nowhere to land"
6576 );
6577 }
6578 if let Some(bad) = self.groups.iter().find(|g| g.trim().is_empty()) {
6579 anyhow::bail!("`groups` has an empty entry: {bad:?}");
6580 }
6581 let mut seen = std::collections::BTreeSet::new();
6582 if let Some(dup) = self.groups.iter().find(|g| !seen.insert(g.as_str())) {
6583 anyhow::bail!("`groups` lists {dup:?} twice");
6584 }
6585 if self.vault_slot.trim().is_empty() {
6586 anyhow::bail!(
6587 "`vault_slot` must not be empty — it names the fob slot holding the value"
6588 );
6589 }
6590 // A deny-all rule is a *valid* record (it is the fail-closed default the
6591 // resolver relies on) but it is never a useful thing to deliberately
6592 // ship, so catching it here saves an operator the round-trip of
6593 // deploying a workload that mysteriously can't see its own secret.
6594 if let SecretAccess::Workloads(entries) = &self.access {
6595 if entries.is_empty() {
6596 anyhow::bail!(
6597 "`[access]` admits nobody: list the workloads allowed to mount {:?} \
6598 (e.g. `workloads = [{{ workload = \"my-service\" }}]`), or set \
6599 `access = \"allow_any\"` to store it unrestricted",
6600 self.name
6601 );
6602 }
6603 if let Some(bad) = entries.iter().find(|e| e.workload.trim().is_empty()) {
6604 anyhow::bail!("`[access]` entry has an empty `workload` name: {bad:?}");
6605 }
6606 }
6607 Ok(())
6608 }
6609
6610 /// Every entry in `groups` must be a sovereign group the camp's machines
6611 /// declare. A typo would otherwise produce a declaration no node ever
6612 /// matches, which `put` would refuse everywhere and `status` would never
6613 /// count, without either one saying why.
6614 pub fn validate_groups(&self, sovereign_groups: &[&str]) -> Result<()> {
6615 if let Some(unknown) = self
6616 .groups
6617 .iter()
6618 .find(|g| !sovereign_groups.contains(&g.as_str()))
6619 {
6620 anyhow::bail!(
6621 "`groups` names {unknown:?}, which no .yah/infra/machines/*.toml declares as a \
6622 `sovereign_group` (declared: {})",
6623 if sovereign_groups.is_empty() {
6624 "(none)".to_string()
6625 } else {
6626 sovereign_groups.join(", ")
6627 }
6628 );
6629 }
6630 Ok(())
6631 }
6632}
6633
6634#[cfg(test)]
6635mod secret_config_tests {
6636 use super::*;
6637
6638 fn parse(body: &str) -> Result<SecretConfig> {
6639 let cfg: SecretConfig = toml::from_str(body)?;
6640 cfg.validate()?;
6641 Ok(cfg)
6642 }
6643
6644 #[test]
6645 fn minimal_declaration_parses_with_narrow_defaults() {
6646 let cfg = parse(
6647 r#"
6648schema_version = 2
6649name = "svc/token"
6650vault_slot = "svc-token"
6651groups = ["prod"]
6652[access]
6653workloads = [{ workload = "svc" }]
6654"#,
6655 )
6656 .unwrap();
6657
6658 assert_eq!(cfg.groups, vec!["prod".to_string()]);
6659 assert_eq!(cfg.encoding, SecretEncoding::Utf8, "text is the default");
6660 assert!(cfg.target.is_none());
6661 // The omitted tenant/namespace must narrow to the singletons, not widen
6662 // to a wildcard.
6663 assert!(cfg
6664 .access
6665 .admits(&workload_spec::secrets::SecretConsumer::workload("svc")));
6666 assert!(!cfg
6667 .access
6668 .admits(&workload_spec::secrets::SecretConsumer::workload("other")));
6669 }
6670
6671 #[test]
6672 fn allow_any_is_spelled_as_a_bare_string() {
6673 // The operator-facing spelling, pinned: `access = "allow_any"`.
6674 let cfg = parse(
6675 r#"
6676schema_version = 2
6677name = "public/thing"
6678vault_slot = "slot"
6679groups = ["prod"]
6680access = "allow_any"
6681"#,
6682 )
6683 .unwrap();
6684 assert_eq!(cfg.access, SecretAccess::AllowAny);
6685 }
6686
6687 #[test]
6688 fn a_declaration_with_no_access_block_is_rejected() {
6689 // Omitting `[access]` defaults to deny-all, which is the correct
6690 // *runtime* default but never a correct authoring intent — so it must
6691 // not silently produce a secret nobody can mount.
6692 let err = parse(
6693 r#"
6694schema_version = 2
6695name = "svc/token"
6696vault_slot = "svc-token"
6697groups = ["prod"]
6698"#,
6699 )
6700 .unwrap_err()
6701 .to_string();
6702 assert!(err.contains("admits nobody"), "got {err}");
6703 }
6704
6705 #[test]
6706 fn empty_name_or_slot_is_rejected() {
6707 assert!(parse(
6708 r#"
6709schema_version = 2
6710name = ""
6711vault_slot = "slot"
6712groups = ["prod"]
6713access = "allow_any"
6714"#
6715 )
6716 .is_err());
6717 assert!(parse(
6718 r#"
6719schema_version = 2
6720name = "x"
6721vault_slot = " "
6722groups = ["prod"]
6723access = "allow_any"
6724"#
6725 )
6726 .is_err());
6727 }
6728
6729 #[test]
6730 fn target_declaration_maps_onto_the_workload_spec_type() {
6731 let cfg = parse(
6732 r#"
6733schema_version = 2
6734name = "svc/token"
6735vault_slot = "slot"
6736groups = ["prod"]
6737access = "allow_any"
6738[target]
6739kind = "file"
6740path = "/run/secrets/t"
6741"#,
6742 )
6743 .unwrap();
6744 match cfg.target.unwrap().to_target() {
6745 workload_spec::SecretTarget::File { path, mode } => {
6746 assert_eq!(path, std::path::PathBuf::from("/run/secrets/t"));
6747 assert_eq!(mode, 0o400, "owner-read-only by default");
6748 }
6749 other => panic!("expected File, got {other:?}"),
6750 }
6751 }
6752
6753 #[test]
6754 fn load_dir_is_empty_for_a_camp_with_no_secrets() {
6755 let tmp = tempfile::TempDir::new().unwrap();
6756 assert!(SecretConfig::load_dir(&tmp.path().join("nope"), &["prod"])
6757 .unwrap()
6758 .is_empty());
6759 }
6760
6761 /// Write `body` to a temp file and run the real loader over it, so the
6762 /// assertions see the error chain an operator would.
6763 fn load_body(body: &str, sovereign_groups: &[&str]) -> Result<SecretConfig> {
6764 let tmp = tempfile::TempDir::new().unwrap();
6765 let path = tmp.path().join("decl.toml");
6766 std::fs::write(&path, body).unwrap();
6767 SecretConfig::load(&path, sovereign_groups)
6768 }
6769
6770 #[test]
6771 fn a_declaration_without_groups_fails_to_load_naming_the_field() {
6772 let err = format!(
6773 "{:#}",
6774 load_body(
6775 "schema_version = 2\nname = \"svc/token\"\nvault_slot = \"slot\"\n\
6776 access = \"allow_any\"\n",
6777 &["prod"],
6778 )
6779 .unwrap_err()
6780 );
6781 assert!(err.contains("groups"), "must name the field: {err}");
6782 assert!(err.contains("decl.toml"), "must name the file: {err}");
6783 }
6784
6785 #[test]
6786 fn an_empty_or_duplicated_group_list_is_rejected() {
6787 let err = format!(
6788 "{:#}",
6789 load_body(
6790 "schema_version = 2\nname = \"s\"\nvault_slot = \"slot\"\ngroups = []\n\
6791 access = \"allow_any\"\n",
6792 &["prod"],
6793 )
6794 .unwrap_err()
6795 );
6796 assert!(err.contains("at least one sovereign group"), "got {err}");
6797
6798 let err = format!(
6799 "{:#}",
6800 load_body(
6801 "schema_version = 2\nname = \"s\"\nvault_slot = \"slot\"\n\
6802 groups = [\"prod\", \"prod\"]\naccess = \"allow_any\"\n",
6803 &["prod"],
6804 )
6805 .unwrap_err()
6806 );
6807 assert!(err.contains("twice"), "got {err}");
6808 }
6809
6810 #[test]
6811 fn a_group_no_machine_declares_is_rejected_naming_the_vocabulary() {
6812 let err = format!(
6813 "{:#}",
6814 load_body(
6815 "schema_version = 2\nname = \"s\"\nvault_slot = \"slot\"\n\
6816 groups = [\"stagin\"]\naccess = \"allow_any\"\n",
6817 &["dev", "prod"],
6818 )
6819 .unwrap_err()
6820 );
6821 assert!(err.contains("\"stagin\""), "names the bad entry: {err}");
6822 assert!(err.contains("dev, prod"), "names what is declared: {err}");
6823 }
6824
6825 #[test]
6826 fn a_version_1_declaration_is_refused_naming_the_migration() {
6827 let err = format!(
6828 "{:#}",
6829 load_body(
6830 "schema_version = 1\nname = \"s\"\nvault_slot = \"slot\"\n\
6831 groups = [\"prod\"]\naccess = \"allow_any\"\n",
6832 &["prod"],
6833 )
6834 .unwrap_err()
6835 );
6836 assert!(err.contains("schema_version") && err.contains("groups"), "got {err}");
6837 }
6838}
6839
6840/// Split a `"<service>/<component-id>"` ref. Returns `None` if the ref
6841/// isn't shaped like `service/component`.
6842pub(crate) fn split_component_ref(s: &str) -> Option<(&str, &str)> {
6843 let (svc, comp) = s.split_once('/')?;
6844 if svc.is_empty() || comp.is_empty() || comp.contains('/') {
6845 return None;
6846 }
6847 Some((svc, comp))
6848}
6849
6850#[cfg(test)]
6851mod tests {
6852 use super::*;
6853 use std::path::PathBuf;
6854
6855 fn make_machine(name: &str, mesh_tags: Vec<&str>) -> MachineConfig {
6856 MachineConfig {
6857 name: name.into(),
6858 provider: "hetzner".into(),
6859 location: Some("hil".into()),
6860 server_type: Some("ccx13".into()),
6861 hosts_mirrors: vec![],
6862 mesh_tags: mesh_tags.into_iter().map(String::from).collect(),
6863 region: None,
6864 zone: None,
6865 arch: None,
6866 bucket: None,
6867 vendor: None,
6868 nickname: None,
6869 legacy_hostkey_fingerprint: None,
6870 registration: Default::default(),
6871 ssh_keys: vec![],
6872 cloudflared: None,
6873 hosts_operator_bridge: false,
6874 connect: None,
6875 allocatable: None,
6876 taints: vec![],
6877 sovereign_group: None,
6878 sovereign_role: None,
6879 ingress_floating_ip: None,
6880 }
6881 }
6882
6883 /// Like [`make_machine`] but with explicit topology axes for F16 tests.
6884 fn make_machine_topo(
6885 name: &str,
6886 provider: &str,
6887 region: &str,
6888 mesh_tags: Vec<&str>,
6889 ) -> MachineConfig {
6890 MachineConfig {
6891 provider: provider.into(),
6892 region: Some(region.into()),
6893 zone: Some(region.into()),
6894 ..make_machine(name, mesh_tags)
6895 }
6896 }
6897
6898 fn make_empty_cfg(machines: Vec<MachineConfig>) -> CloudConfig {
6899 CloudConfig {
6900 workspace_root: PathBuf::new(),
6901 machines,
6902 providers: vec![],
6903 machine_origins: BTreeMap::new(),
6904 provider_origins: BTreeMap::new(),
6905 recovery_measurements: BTreeMap::new(),
6906 services: BTreeMap::new(),
6907 domains: BTreeMap::new(),
6908 legacy_mirrors: vec![],
6909 workloads: vec![],
6910 topology: TopologyConfig::default(),
6911 }
6912 }
6913
6914 #[test]
6915 fn required_spec_parses_from_provider_fields() {
6916 let toml_src = r#"
6917use = "hetzner-primary"
6918[required]
6919mesh_tags = ["tag:cloud-runner"]
6920"#;
6921 let slot: MirrorProviderSlot = toml::from_str(toml_src).unwrap();
6922 let req = slot.required().expect("required block present");
6923 assert_eq!(req.mesh_tags, vec!["tag:cloud-runner"]);
6924 }
6925
6926 #[test]
6927 fn required_spec_absent_when_field_missing() {
6928 let slot: MirrorProviderSlot = toml::from_str(r#"use = "hetzner-primary""#).unwrap();
6929 assert!(slot.required().is_none());
6930 }
6931
6932 #[test]
6933 fn db_catalog_parses_all_env_blocks() {
6934 // W241 / R571-F8: a service.toml [db] table with dev/pond/cloud.
6935 let toml_src = r#"
6936schema_version = 1
6937name = "scrabcake"
6938[address]
6939kind = "front-door"
6940domain = "scrabcake.net.yah.dev"
6941
6942[[db.dev]]
6943name = "main"
6944path = "data/dev.sqlite"
6945
6946[[db.pond]]
6947name = "main"
6948port = 5433
6949
6950[[db.pond]]
6951name = "pg"
6952port = 5432
6953kind = "postgres"
6954
6955[[db.cloud]]
6956name = "main"
6957url = "libsql://scrabcake.turso.io"
6958auth_token_env = "SCRABCAKE_TURSO_TOKEN"
6959"#;
6960 let svc: ServiceConfig = toml::from_str(toml_src).unwrap();
6961 assert_eq!(svc.db.dev.len(), 1);
6962 assert_eq!(svc.db.dev[0].path, "data/dev.sqlite");
6963 assert_eq!(svc.db.pond.len(), 2);
6964 assert_eq!(svc.db.pond[0].port, Some(5433));
6965 assert_eq!(svc.db.pond[0].kind, PondDbKind::Turso); // default
6966 assert_eq!(svc.db.pond[1].kind, PondDbKind::Postgres);
6967 assert_eq!(
6968 svc.db.cloud[0].auth_token_env.as_deref(),
6969 Some("SCRABCAKE_TURSO_TOKEN")
6970 );
6971 }
6972
6973 #[test]
6974 fn service_without_db_table_has_empty_catalog() {
6975 let svc: ServiceConfig =
6976 toml::from_str("schema_version = 1\nname = \"s\"\n[address]\nkind = \"front-door\"\ndomain = \"s.dev\"\n").unwrap();
6977 assert!(svc.db.is_empty());
6978 // And an empty [db] must not appear when re-serialized.
6979 let out = toml::to_string(&svc).unwrap();
6980 assert!(
6981 !out.contains("[db"),
6982 "empty db table should be skipped: {out}"
6983 );
6984 }
6985
6986 #[test]
6987 fn camp_shared_cloud_toml_parses() {
6988 let src = r#"
6989[[cloud]]
6990name = "analytics"
6991url = "postgres://shared/analytics"
6992"#;
6993 let shared: CampCloudDbs = toml::from_str(src).unwrap();
6994 assert_eq!(shared.cloud.len(), 1);
6995 assert_eq!(shared.cloud[0].name, "analytics");
6996 }
6997
6998 #[test]
6999 fn resolve_machine_by_mesh_tags_superset_match() {
7000 let cfg = make_empty_cfg(vec![
7001 make_machine("yah-bnt-1", vec!["tag:primary-yah", "tag:tier-scratch"]),
7002 make_machine("us-west-001", vec!["tag:primary-yah", "tag:cloud-runner"]),
7003 ]);
7004 let picked = cfg
7005 .resolve_machine_by_mesh_tags(&["tag:cloud-runner".into()])
7006 .map(|m| m.name.as_str());
7007 assert_eq!(picked, Some("us-west-001"));
7008 }
7009
7010 #[test]
7011 fn resolve_machine_by_mesh_tags_returns_none_when_no_match() {
7012 let cfg = make_empty_cfg(vec![make_machine("yah-bnt-1", vec!["tag:primary-yah"])]);
7013 assert!(cfg
7014 .resolve_machine_by_mesh_tags(&["tag:cloud-runner".into()])
7015 .is_none());
7016 }
7017
7018 // ─── R590-F1 mesh-tag node-selector admission ───────────────────────────
7019
7020 /// Build a forge WorkloadSpec carrying the R594 node-selector annotation.
7021 /// `selector` is the comma-joined mesh-tag set; `None` omits the annotation
7022 /// entirely (pre-R594 "no constraint").
7023 fn ws_with_selector(selector: Option<&str>) -> WorkloadSpec {
7024 use workload_spec::{ImageRef, TierTag};
7025 let mut ws = WorkloadSpec::for_forge(
7026 "R590-F1-test",
7027 ImageRef {
7028 registry: "docker.io".into(),
7029 repository: "library/busybox".into(),
7030 tag: "latest".into(),
7031 digest: workload_spec::testing::test_digest(),
7032 },
7033 TierTag("infra".into()),
7034 vec![],
7035 );
7036 if let Some(sel) = selector {
7037 ws.annotations.insert(
7038 velveteen_exec::remote::NODE_SELECTOR_MESH_TAGS_ANNOTATION.into(),
7039 sel.into(),
7040 );
7041 }
7042 ws
7043 }
7044
7045 /// The build-worker fleet shape: one x86 node (us-west-002) and one arm
7046 /// node (a Pi5), both carrying `tag:build-worker`.
7047 fn build_worker_fleet() -> CloudConfig {
7048 make_empty_cfg(vec![
7049 make_machine("us-west-002", vec!["tag:build-worker", "arch:x86"]),
7050 make_machine("pi5-001", vec!["tag:build-worker", "arch:arm"]),
7051 ])
7052 }
7053
7054 #[test]
7055 fn admit_workload_routes_amd64_to_x86_worker() {
7056 let cfg = build_worker_fleet();
7057 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
7058 let picked = cfg.admit_workload(&ws).unwrap();
7059 assert_eq!(picked.name, "us-west-002");
7060 }
7061
7062 #[test]
7063 fn admit_workload_routes_arm64_to_pi5_worker() {
7064 let cfg = build_worker_fleet();
7065 let ws = ws_with_selector(Some("tag:build-worker,arch:arm"));
7066 let picked = cfg.admit_workload(&ws).unwrap();
7067 assert_eq!(picked.name, "pi5-001");
7068 }
7069
7070 /// A forge run must be admissible on a build-worker smaller than its own
7071 /// cgroup ceiling.
7072 ///
7073 /// The fleet's arm build-workers are 8 GiB Pi-5s and `for_forge` sets a
7074 /// 32 GiB ceiling, so while admission read `resources.memory_mb` as the
7075 /// capacity floor this returned "no candidates" and *every* offloaded qed
7076 /// step to those nodes failed at dispatch — measured on desktop-release run
7077 /// b04cef47, where the aarch64-linux row died in 1.6s. The other
7078 /// build-workers (16 GiB us-west-003, and the arm Pi-5s) were excluded the
7079 /// same way, leaving one 47 GiB node as the fleet's only legal target for
7080 /// remote CI.
7081 #[test]
7082 fn admit_workload_places_a_forge_run_on_a_worker_smaller_than_its_ceiling() {
7083 let mut pi = make_machine("pi5-001", vec!["tag:build-worker", "arch:arm"]);
7084 pi.allocatable = Some(NodeAllocatable {
7085 memory_mb: 8192,
7086 cpu_millis: 4000,
7087 });
7088 let cfg = make_empty_cfg(vec![pi]);
7089
7090 let ws = ws_with_selector(Some("tag:build-worker,arch:arm"));
7091 assert!(
7092 ws.resources.memory_mb > 8192,
7093 "precondition: the ceiling must exceed the node, or this proves nothing"
7094 );
7095
7096 let picked = cfg
7097 .admit_workload(&ws)
7098 .expect("an 8 GiB build-worker must admit a forge run");
7099 assert_eq!(picked.name, "pi5-001");
7100 }
7101
7102 /// The floor is still enforced — the fix separates two numbers, it does not
7103 /// disable the R572-F5 capacity check.
7104 #[test]
7105 fn admit_workload_still_rejects_a_node_below_the_declared_request() {
7106 let mut tiny = make_machine("tiny-001", vec!["tag:build-worker", "arch:arm"]);
7107 tiny.allocatable = Some(NodeAllocatable {
7108 memory_mb: 512,
7109 cpu_millis: 4000,
7110 });
7111 let cfg = make_empty_cfg(vec![tiny]);
7112
7113 let ws = ws_with_selector(Some("tag:build-worker,arch:arm"));
7114 assert!(
7115 cfg.admit_workload(&ws).is_err(),
7116 "a 512 MiB node cannot satisfy a 2 GiB forge request"
7117 );
7118 }
7119
7120 // ─── R833-F8 imperative node-selector admission ─────────────────────────
7121
7122 /// Build a forge WorkloadSpec carrying the R833-F8 imperative node
7123 /// selector — the operator's `--where=node:<machine>`.
7124 fn ws_pinned_to(node: &str) -> WorkloadSpec {
7125 let mut ws = ws_with_selector(None);
7126 ws.annotations.insert(
7127 velveteen_exec::remote::NODE_SELECTOR_NODE_ANNOTATION.into(),
7128 node.into(),
7129 );
7130 ws
7131 }
7132
7133 /// The ticket's acceptance shape: a named node wins over the
7134 /// declaration-order tie-break that would otherwise decide placement.
7135 /// `us-west-002` is declared first and carries every tag, so an inferred
7136 /// placement lands there; the pin must reach `pi5-001` regardless.
7137 #[test]
7138 fn admit_workload_honours_an_explicitly_named_node() {
7139 let cfg = build_worker_fleet();
7140 assert_eq!(
7141 cfg.admit_workload(&ws_with_selector(Some("tag:build-worker")))
7142 .unwrap()
7143 .name,
7144 "us-west-002",
7145 "precondition: inference elects the first-declared node",
7146 );
7147 assert_eq!(
7148 cfg.admit_workload(&ws_pinned_to("pi5-001")).unwrap().name,
7149 "pi5-001",
7150 );
7151 }
7152
7153 /// A pin at a machine that is not declared fails loud, naming the
7154 /// constraint and the pool — the operator mistyped a node, and silently
7155 /// running the build somewhere else is the one outcome that must not
7156 /// happen.
7157 #[test]
7158 fn admit_workload_refuses_a_node_that_is_not_declared() {
7159 let cfg = build_worker_fleet();
7160 let err = cfg
7161 .admit_workload(&ws_pinned_to("us-west-404"))
7162 .unwrap_err()
7163 .to_string();
7164 assert!(err.contains("required.nodes=[us-west-404]"), "{err}");
7165 assert!(err.contains("us-west-002"), "the pool must be named: {err}");
7166 }
7167
7168 /// The pin narrows the candidate set; it does not suspend the other axes.
7169 /// A named node that cannot fit the workload still refuses, rather than
7170 /// being handed work it has no room for.
7171 #[test]
7172 fn a_pinned_node_is_still_checked_against_capacity() {
7173 let mut tiny = make_machine("tiny-001", vec!["tag:build-worker", "arch:arm"]);
7174 tiny.allocatable = Some(NodeAllocatable {
7175 memory_mb: 512,
7176 cpu_millis: 4000,
7177 });
7178 let cfg = make_empty_cfg(vec![tiny]);
7179 assert!(cfg.admit_workload(&ws_pinned_to("tiny-001")).is_err());
7180 }
7181
7182 /// Inference is untouched: with no node annotation the `nodes` axis is
7183 /// empty, which is "no constraint" — every pre-R833-F8 workload is admitted
7184 /// exactly as before.
7185 #[test]
7186 fn an_unpinned_workload_carries_no_node_constraint() {
7187 assert!(node_selector_node(&ws_with_selector(Some("arch:x86"))).is_none());
7188 assert_eq!(
7189 node_selector_node(&ws_pinned_to("us-west-003")).as_deref(),
7190 Some("us-west-003")
7191 );
7192 assert!(RequiredSpec::default().is_unconstrained());
7193 assert!(!RequiredSpec {
7194 nodes: vec!["us-west-003".into()],
7195 ..Default::default()
7196 }
7197 .is_unconstrained());
7198 }
7199
7200 #[test]
7201 fn admit_workload_rejects_node_missing_required_tag() {
7202 // Only an arm worker exists; an x86 build must NOT land on it.
7203 let cfg = make_empty_cfg(vec![make_machine(
7204 "pi5-001",
7205 vec!["tag:build-worker", "arch:arm"],
7206 )]);
7207 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
7208 assert!(cfg.admit_workload(&ws).is_err());
7209 }
7210
7211 /// R555-S1 regression: with TWO nodes carrying the same tag set, which one
7212 /// admits must be decided by *declaration order* (file name), which is the
7213 /// contract `admit_workload` documents — not by `read_dir` order, which is
7214 /// filesystem-dependent and can change when an unrelated file appears in
7215 /// the directory. Written creation-order-reversed so a filesystem that
7216 /// yields creation order (rather than sorted order) trips it without the
7217 /// sort in `load_dir`.
7218 ///
7219 /// Live consequence this guards: `.yah/infra/machines/` carries both
7220 /// us-west-002 and us-west-003 on `[tag:build-worker, arch:x86, os:linux]`,
7221 /// so an x86 QED offload has two equal candidates. Unstable selection means
7222 /// a retried build cannot be relied on to land back on the node whose
7223 /// working state it left behind.
7224 #[test]
7225 fn equally_matching_machines_admit_in_file_name_order() {
7226 let tmp = tempfile::TempDir::new().unwrap();
7227 let machines = tmp.path().join(".yah").join("infra").join("machines");
7228 std::fs::create_dir_all(&machines).unwrap();
7229 let toml_for = |name: &str| {
7230 format!(
7231 r#"name = "{name}"
7232provider = "static"
7233mesh_tags = ["tag:build-worker", "arch:x86"]
7234"#
7235 )
7236 };
7237 // Reverse-of-sorted creation order on purpose.
7238 std::fs::write(machines.join("b-second.toml"), toml_for("b-second")).unwrap();
7239 std::fs::write(machines.join("a-first.toml"), toml_for("a-first")).unwrap();
7240
7241 let cfg = CloudConfig::load(tmp.path()).unwrap();
7242 assert_eq!(
7243 cfg.machines.iter().map(|m| m.name.as_str()).collect::<Vec<_>>(),
7244 vec!["a-first", "b-second"],
7245 "machines must load in file-name order, not read_dir order"
7246 );
7247
7248 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
7249 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "a-first");
7250
7251 // R605-T14: the same two nodes, seen as the pool they are. The head is
7252 // what `admit_workload` returns, and the tail is what the dispatcher
7253 // fails over to when the head does not answer — so these two views must
7254 // come from one predicate, not two.
7255 assert_eq!(
7256 cfg.admit_workload_candidates(&ws)
7257 .unwrap()
7258 .iter()
7259 .map(|m| m.name.as_str())
7260 .collect::<Vec<_>>(),
7261 vec!["a-first", "b-second"],
7262 "the pool must be every admissible node, in the same declaration order"
7263 );
7264 }
7265
7266 /// A pool of one is still a pool, and a pool of none is an `Err` that reads
7267 /// exactly like `admit_workload`'s — "nothing admits this" is one failure
7268 /// with one wording, not two.
7269 #[test]
7270 fn admit_workload_candidates_matches_admit_workload_on_the_edges() {
7271 let cfg = make_empty_cfg(vec![
7272 make_machine("x86-box", vec!["tag:build-worker", "arch:x86"]),
7273 make_machine("arm-box", vec!["tag:build-worker", "arch:arm"]),
7274 ]);
7275
7276 let one = ws_with_selector(Some("arch:arm"));
7277 assert_eq!(
7278 cfg.admit_workload_candidates(&one)
7279 .unwrap()
7280 .iter()
7281 .map(|m| m.name.as_str())
7282 .collect::<Vec<_>>(),
7283 vec!["arm-box"],
7284 "only one node carries arch:arm, so the pool is that one node"
7285 );
7286
7287 let none = ws_with_selector(Some("arch:riscv"));
7288 let pool_err = cfg.admit_workload_candidates(&none).unwrap_err().to_string();
7289 let single_err = cfg.admit_workload(&none).unwrap_err().to_string();
7290 assert_eq!(
7291 pool_err, single_err,
7292 "an empty pool must be refused in the same words as an unadmitted workload"
7293 );
7294 }
7295
7296 /// R844-B7 — the wrong-root half of the distinction. A directory with no
7297 /// `.yah/` at all used to load as a valid config with zero machines, so a
7298 /// caller pointed at the wrong directory got a green result that measured
7299 /// nothing. Asserting `load` merely *succeeds* is what let that through;
7300 /// the shape that catches it is a non-zero machine count, or — here — an
7301 /// `Err` naming the path that was looked for.
7302 #[test]
7303 fn loading_a_directory_that_is_not_a_yah_workspace_is_an_error() {
7304 let tmp = tempfile::TempDir::new().unwrap();
7305 // A plausible-looking package root: real files, real subdirectories,
7306 // no `.yah/`. This is exactly what `load_cloud(".")` reads when a test
7307 // runs under `cargo test` from a member crate.
7308 std::fs::create_dir_all(tmp.path().join("src")).unwrap();
7309 std::fs::write(tmp.path().join("Cargo.toml"), "[package]\nname = \"x\"\n").unwrap();
7310
7311 let err = CloudConfig::load(tmp.path()).expect_err(
7312 "a directory with no .yah/ is the WRONG DIRECTORY, not a fleet with no machines",
7313 );
7314 let msg = format!("{err:#}");
7315 assert!(
7316 msg.contains("not a yah workspace"),
7317 "error must say the root is not a workspace, got: {msg}"
7318 );
7319 assert!(
7320 msg.contains(&tmp.path().join(".yah").display().to_string()),
7321 "error must name the path it looked for so an operator sees the \
7322 wrong-root immediately, got: {msg}"
7323 );
7324 }
7325
7326 /// R844-B7 — the other half, and the reason the check is drawn at `.yah/`
7327 /// rather than at the machine list: a camp that declares no machines is a
7328 /// real workspace and must keep loading. Blanket-erroring on an empty
7329 /// fleet would conflate `unknown` with `answered with none`, which is the
7330 /// exact confusion the check exists to remove.
7331 #[test]
7332 fn a_workspace_with_no_machines_declared_still_loads() {
7333 let tmp = tempfile::TempDir::new().unwrap();
7334 std::fs::create_dir_all(tmp.path().join(".yah")).unwrap();
7335
7336 let cfg = CloudConfig::load(tmp.path())
7337 .expect("a `.yah/` with no infra/machines/ is an empty fleet, not a wrong root");
7338 assert!(cfg.machines.is_empty(), "nothing was declared");
7339 assert!(cfg.services.is_empty());
7340 assert!(cfg.providers.is_empty());
7341
7342 // And an existing-but-empty machines dir is the same answer, not a
7343 // second special case.
7344 std::fs::create_dir_all(crate::paths::machines_dir(tmp.path())).unwrap();
7345 let cfg = CloudConfig::load(tmp.path()).expect("an empty machines/ dir still loads");
7346 assert!(cfg.machines.is_empty());
7347 }
7348
7349 #[test]
7350 fn admit_workload_empty_selector_is_unconstrained() {
7351 // Absent annotation ⇒ no mesh-tag constraint ⇒ first declared machine
7352 // (pre-R594 behavior preserved).
7353 let cfg = build_worker_fleet();
7354 let ws = ws_with_selector(None);
7355 let picked = cfg.admit_workload(&ws).unwrap();
7356 assert_eq!(picked.name, "us-west-002");
7357 }
7358
7359 #[test]
7360 fn node_selector_mesh_tags_trims_and_drops_empties() {
7361 let ws = ws_with_selector(Some(" tag:build-worker , arch:x86 ,"));
7362 assert_eq!(
7363 node_selector_mesh_tags(&ws),
7364 vec!["tag:build-worker".to_string(), "arch:x86".to_string()]
7365 );
7366 assert!(node_selector_mesh_tags(&ws_with_selector(None)).is_empty());
7367 }
7368
7369 // ─── F16 topology-aware resolver ────────────────────────────────────────
7370
7371 fn two_region_fleet() -> CloudConfig {
7372 make_empty_cfg(vec![
7373 make_machine_topo(
7374 "us-west-001",
7375 "hetzner",
7376 "us-west",
7377 vec!["tag:cloud-runner"],
7378 ),
7379 make_machine_topo(
7380 "eu-west-001",
7381 "hetzner",
7382 "eu-west",
7383 vec!["tag:cloud-runner"],
7384 ),
7385 ])
7386 }
7387
7388 #[test]
7389 fn resolve_machine_matches_on_region_plus_mesh_tags() {
7390 let cfg = two_region_fleet();
7391 let req = RequiredSpec {
7392 regions: vec!["us-west".into()],
7393 mesh_tags: vec!["tag:cloud-runner".into()],
7394 ..Default::default()
7395 };
7396 let picked = cfg.resolve_machine(&req).unwrap();
7397 assert_eq!(picked.name, "us-west-001");
7398 }
7399
7400 #[test]
7401 fn resolve_machine_region_disambiguates_same_tag() {
7402 // Both boxes carry tag:cloud-runner; the region axis selects eu-west.
7403 let cfg = two_region_fleet();
7404 let req = RequiredSpec {
7405 regions: vec!["eu-west".into()],
7406 mesh_tags: vec!["tag:cloud-runner".into()],
7407 ..Default::default()
7408 };
7409 assert_eq!(cfg.resolve_machine(&req).unwrap().name, "eu-west-001");
7410 }
7411
7412 #[test]
7413 fn resolve_machine_fails_loud_with_constraint_summary() {
7414 let cfg = two_region_fleet();
7415 let req = RequiredSpec {
7416 regions: vec!["us-central".into()],
7417 mesh_tags: vec!["tag:cloud-runner".into()],
7418 ..Default::default()
7419 };
7420 let err = cfg.resolve_machine(&req).unwrap_err().to_string();
7421 assert!(err.contains("required.regions=[us-central]"), "got: {err}");
7422 assert!(
7423 err.contains("required.mesh_tags=[tag:cloud-runner]"),
7424 "got: {err}"
7425 );
7426 // Names the candidates it rejected.
7427 assert!(err.contains("us-west-001"), "got: {err}");
7428 }
7429
7430 #[test]
7431 fn resolve_machine_provider_axis_filters() {
7432 let cfg = make_empty_cfg(vec![
7433 make_machine_topo("aws-west-1", "aws", "us-west", vec!["tag:cloud-runner"]),
7434 make_machine_topo("hz-west-1", "hetzner", "us-west", vec!["tag:cloud-runner"]),
7435 ]);
7436 let req = RequiredSpec {
7437 regions: vec!["us-west".into()],
7438 providers: vec!["hetzner".into()],
7439 ..Default::default()
7440 };
7441 assert_eq!(cfg.resolve_machine(&req).unwrap().name, "hz-west-1");
7442 }
7443
7444 #[test]
7445 fn unconstrained_required_spec_matches_first_machine() {
7446 let cfg = two_region_fleet();
7447 assert!(RequiredSpec::default().is_unconstrained());
7448 assert_eq!(
7449 cfg.resolve_machine(&RequiredSpec::default()).unwrap().name,
7450 "us-west-001"
7451 );
7452 }
7453
7454 #[test]
7455 fn required_spec_parses_topology_axes_from_toml() {
7456 let toml_src = r#"
7457use = "hetzner-primary"
7458[required]
7459regions = ["us-west"]
7460mesh_tags = ["tag:cloud-runner"]
7461"#;
7462 let slot: MirrorProviderSlot = toml::from_str(toml_src).unwrap();
7463 let req = slot.required().expect("required block present");
7464 assert_eq!(req.regions, vec!["us-west"]);
7465 assert_eq!(req.mesh_tags, vec!["tag:cloud-runner"]);
7466 assert!(req.zones.is_empty());
7467 }
7468
7469 // ─── R844-F8 replica count ──────────────────────────────────────────────
7470
7471 fn three_runner_fleet() -> CloudConfig {
7472 make_empty_cfg(vec![
7473 make_machine_topo("us-east-001", "hetzner", "us-east", vec!["tag:cloud-runner"]),
7474 make_machine_topo(
7475 "us-south-001",
7476 "hetzner",
7477 "us-south",
7478 vec!["tag:cloud-runner"],
7479 ),
7480 make_machine_topo(
7481 "us-west-001",
7482 "hetzner",
7483 "us-west",
7484 vec!["tag:cloud-runner"],
7485 ),
7486 ])
7487 }
7488
7489 #[test]
7490 fn an_absent_replica_count_still_places_exactly_one_machine() {
7491 // The migration is additive: every mirror on disk omits `replicas`, and
7492 // must resolve byte-identically to the pre-R844-F8 answer.
7493 let cfg = three_runner_fleet();
7494 let req = RequiredSpec {
7495 mesh_tags: vec!["tag:cloud-runner".into()],
7496 ..Default::default()
7497 };
7498 assert_eq!(req.replica_count(), 1);
7499 let names: Vec<&str> = cfg
7500 .resolve_machines(&req)
7501 .unwrap()
7502 .iter()
7503 .map(|m| m.name.as_str())
7504 .collect();
7505 assert_eq!(names, vec!["us-east-001"]);
7506 assert_eq!(cfg.resolve_machine(&req).unwrap().name, "us-east-001");
7507 }
7508
7509 #[test]
7510 fn a_replica_count_places_that_many_machines_not_every_match() {
7511 // Three machines match; two are asked for; two are placed. Inferring the
7512 // count from the match count would make adding a box to the fleet
7513 // silently scale a production front door.
7514 let cfg = three_runner_fleet();
7515 let req = RequiredSpec {
7516 mesh_tags: vec!["tag:cloud-runner".into()],
7517 replicas: Some(2),
7518 ..Default::default()
7519 };
7520 let names: Vec<&str> = cfg
7521 .resolve_machines(&req)
7522 .unwrap()
7523 .iter()
7524 .map(|m| m.name.as_str())
7525 .collect();
7526 assert_eq!(names, vec!["us-east-001", "us-south-001"]);
7527 }
7528
7529 #[test]
7530 fn fewer_matches_than_replicas_is_an_error_naming_both_numbers() {
7531 // Never a partial placement: one of two reported as success is the
7532 // subset-that-looks-like-it-worked failure in its purest form.
7533 let cfg = three_runner_fleet();
7534 let req = RequiredSpec {
7535 regions: vec!["us-east".into()],
7536 replicas: Some(2),
7537 ..Default::default()
7538 };
7539 let err = cfg.resolve_machines(&req).unwrap_err().to_string();
7540 assert!(err.contains("only 1 of 2"), "got: {err}");
7541 assert!(err.contains("required.regions=[us-east]"), "got: {err}");
7542 // …and names the pool it searched, like every other placement refusal.
7543 assert!(err.contains("declared machines"), "got: {err}");
7544 assert!(err.contains("us-south-001"), "got: {err}");
7545 }
7546
7547 #[test]
7548 fn zero_replicas_is_refused_rather_than_placing_nothing() {
7549 let cfg = three_runner_fleet();
7550 let req = RequiredSpec {
7551 mesh_tags: vec!["tag:cloud-runner".into()],
7552 replicas: Some(0),
7553 ..Default::default()
7554 };
7555 let err = cfg.resolve_machines(&req).unwrap_err().to_string();
7556 assert!(err.contains("replicas = 0"), "got: {err}");
7557 }
7558
7559 #[test]
7560 fn replicas_parses_from_the_inline_required_form() {
7561 // The INLINE form specifically: `[providers.bundle.required]` as a table
7562 // HEADER ends the slot's table and reparents every key below it.
7563 let slot: MirrorProviderSlot = toml::from_str(
7564 r#"
7565use = "hetzner-primary"
7566port = 8080
7567required = { regions = ["us-east"], mesh_tags = ["tag:cloud-runner"], replicas = 2 }
7568"#,
7569 )
7570 .unwrap();
7571 assert_eq!(
7572 slot.fields().get("port").and_then(|v| v.as_integer()),
7573 Some(8080),
7574 "the inline form leaves the slot's other keys where they were"
7575 );
7576 let req = slot.required().expect("required block present");
7577 assert_eq!(req.replicas, Some(2));
7578 assert_eq!(req.replica_count(), 2);
7579 // A count is not a match axis — it says how many, not which.
7580 assert!(!req.is_unconstrained());
7581 assert!(RequiredSpec {
7582 replicas: Some(2),
7583 ..Default::default()
7584 }
7585 .is_unconstrained());
7586 }
7587
7588 #[test]
7589 fn round_trip_machine() {
7590 let cfg = MachineConfig {
7591 name: "test-pdx-1".into(),
7592 provider: "hetzner".into(),
7593 location: Some("pdx".into()),
7594 server_type: Some("cpx22".into()),
7595 hosts_mirrors: vec!["noisetable".into()],
7596 mesh_tags: vec!["region:pdx".into()],
7597 region: Some("us-west".into()),
7598 zone: Some("pdx".into()),
7599 arch: None,
7600 bucket: Some(BucketSpec {
7601 name: "test-assets-pdx-1".into(),
7602 public_read: false,
7603 }),
7604 vendor: None,
7605 nickname: None,
7606 legacy_hostkey_fingerprint: None,
7607 registration: Default::default(),
7608 ssh_keys: vec![],
7609 cloudflared: None,
7610 hosts_operator_bridge: false,
7611 connect: None,
7612 allocatable: None,
7613 taints: vec![],
7614 sovereign_group: None,
7615 sovereign_role: None,
7616 ingress_floating_ip: None,
7617 };
7618 let s = toml::to_string(&cfg).unwrap();
7619 let back: MachineConfig = toml::from_str(&s).unwrap();
7620 assert_eq!(back.name, cfg.name);
7621 assert_eq!(back.location, cfg.location);
7622 assert_eq!(back.region.as_deref(), Some("us-west"));
7623 assert_eq!(back.zone.as_deref(), Some("pdx"));
7624 }
7625
7626 #[test]
7627 fn round_trip_mirror() {
7628 let cfg = LegacyMirrorConfig {
7629 camp: "noisetable".into(),
7630 regions: vec!["pdx".into(), "iad".into()],
7631 workloads: vec!["asset-registry".into()],
7632 cloud_domain: None,
7633 };
7634 let s = toml::to_string(&cfg).unwrap();
7635 let back: LegacyMirrorConfig = toml::from_str(&s).unwrap();
7636 assert_eq!(back.camp, cfg.camp);
7637 assert_eq!(back.regions, cfg.regions);
7638 assert_eq!(back.workloads, cfg.workloads);
7639 }
7640
7641 #[test]
7642 fn mirror_serialises_as_camp_key() {
7643 // Serialised form should use `camp`, not `rig`.
7644 let cfg = LegacyMirrorConfig {
7645 camp: "noisetable".into(),
7646 regions: vec!["pdx".into()],
7647 workloads: vec![],
7648 cloud_domain: None,
7649 };
7650 let s = toml::to_string(&cfg).unwrap();
7651 assert!(
7652 s.contains("camp = "),
7653 "serialised key should be 'camp': {s}"
7654 );
7655 assert!(!s.contains("rig = "), "old key should not appear: {s}");
7656 }
7657
7658 #[test]
7659 fn mirror_rig_alias_still_loads() {
7660 // Old mirrors/*.toml files use `rig = "..."` before the R137 rename;
7661 // the alias keeps them loading until the one-time `sed` migration runs.
7662 let toml_str =
7663 "rig = \"noisetable\"\nregions = [\"pdx\"]\nworkloads = [\"asset-registry\"]\n";
7664 let cfg: LegacyMirrorConfig = toml::from_str(toml_str).unwrap();
7665 assert_eq!(cfg.camp, "noisetable");
7666 }
7667
7668 #[test]
7669 fn mirror_services_alias_still_loads() {
7670 // Old mirrors/*.toml files use `services = [...]`; the alias keeps them
7671 // loading without a migration step.
7672 let toml_str =
7673 "camp = \"noisetable\"\nregions = [\"pdx\"]\nservices = [\"asset-registry\"]\n";
7674 let cfg: LegacyMirrorConfig = toml::from_str(toml_str).unwrap();
7675 assert_eq!(cfg.workloads, vec!["asset-registry"]);
7676 }
7677
7678 #[test]
7679 fn load_dir_missing_is_empty() {
7680 let dir = std::path::PathBuf::from("/nonexistent/path");
7681 let result: Vec<MachineConfig> = load_dir(dir).unwrap();
7682 assert!(result.is_empty());
7683 }
7684
7685 #[test]
7686 fn topology_round_trip() {
7687 let topo = TopologyConfig {
7688 assignments: vec![
7689 MirrorAssignment {
7690 mirror: "noisetable-pdx".into(),
7691 machine: "noisetable-pdx-1".into(),
7692 },
7693 MirrorAssignment {
7694 mirror: "noisetable-iad".into(),
7695 machine: "noisetable-iad-1".into(),
7696 },
7697 ],
7698 buckets: vec![],
7699 };
7700 let s = toml::to_string(&topo).unwrap();
7701 let back: TopologyConfig = toml::from_str(&s).unwrap();
7702 assert_eq!(back.assignments.len(), 2);
7703 assert_eq!(back.assignments[0].mirror, "noisetable-pdx");
7704 assert_eq!(back.assignments[1].machine, "noisetable-iad-1");
7705 }
7706
7707 #[test]
7708 fn topology_absent_returns_default() {
7709 let tmp = tempfile::TempDir::new().unwrap();
7710 let path = tmp.path().join("topology.toml");
7711 // file doesn't exist
7712 let topo = load_topology(path).unwrap();
7713 assert!(topo.assignments.is_empty());
7714 }
7715
7716 /// Helper: lay out a `<workspace_root>/.yah/cloud/` legacy tree for the
7717 /// pre-R215 cargo tests below; returns the legacy cloud_dir for writes.
7718 fn make_legacy_cloud_dir(root: &std::path::Path) -> std::path::PathBuf {
7719 let cloud_dir = root.join(".yah").join("cloud");
7720 std::fs::create_dir_all(&cloud_dir).unwrap();
7721 cloud_dir
7722 }
7723
7724 #[test]
7725 fn cloud_config_load_and_lookup() {
7726 let tmp = tempfile::TempDir::new().unwrap();
7727 let root = tmp.path();
7728 let cloud_dir = make_legacy_cloud_dir(root);
7729
7730 let machine = MachineConfig {
7731 name: "noisetable-pdx-1".into(),
7732 provider: "hetzner".into(),
7733 location: Some("pdx".into()),
7734 server_type: Some("cpx22".into()),
7735 hosts_mirrors: vec!["noisetable".into(), "yah".into()],
7736 mesh_tags: vec!["region:pdx".into(), "tier:t2".into()],
7737 region: None,
7738 zone: None,
7739 arch: None,
7740 bucket: Some(BucketSpec {
7741 name: "noisetable-assets-pdx-1".into(),
7742 public_read: false,
7743 }),
7744 vendor: None,
7745 nickname: None,
7746 legacy_hostkey_fingerprint: None,
7747 registration: Default::default(),
7748 ssh_keys: vec![],
7749 cloudflared: None,
7750 hosts_operator_bridge: false,
7751 connect: None,
7752 allocatable: None,
7753 taints: vec![],
7754 sovereign_group: None,
7755 sovereign_role: None,
7756 ingress_floating_ip: None,
7757 };
7758 // Land in the legacy tree so the legacy machine loader picks it up.
7759 machine.save(&cloud_dir).unwrap();
7760
7761 let mirror_toml = "camp = \"noisetable\"\nregions = [\"pdx\", \"iad\", \"fsn\"]\nworkloads = [\"asset-registry\"]\n";
7762 std::fs::create_dir_all(cloud_dir.join("mirrors")).unwrap();
7763 std::fs::write(cloud_dir.join("mirrors/noisetable.toml"), mirror_toml).unwrap();
7764
7765 let cfg = CloudConfig::load(root).unwrap();
7766
7767 assert_eq!(cfg.machines.len(), 1);
7768 assert_eq!(cfg.legacy_mirrors.len(), 1);
7769 assert_eq!(cfg.workloads.len(), 0); // no workloads/ dir yet
7770 assert!(cfg.services.is_empty(), "no R215+ services/ tree");
7771 assert!(cfg.providers.is_empty(), "no R215+ providers/ tree");
7772
7773 let m = cfg.machine("noisetable-pdx-1").unwrap();
7774 assert_eq!(m.location(), "pdx");
7775 assert_eq!(m.bucket.as_ref().unwrap().name, "noisetable-assets-pdx-1");
7776
7777 let mir = cfg.legacy_mirror("noisetable").unwrap();
7778 assert_eq!(mir.regions, vec!["pdx", "iad", "fsn"]);
7779 assert_eq!(mir.workloads, vec!["asset-registry"]);
7780 }
7781
7782 #[test]
7783 fn mirror_folder_layout_loads() {
7784 // Folder layout: mirrors/<id>/mirror.toml — new preferred form.
7785 let tmp = tempfile::TempDir::new().unwrap();
7786 let root = tmp.path();
7787 let cloud_dir = make_legacy_cloud_dir(root);
7788 let mirror_dir = cloud_dir.join("mirrors").join("yah-com");
7789 std::fs::create_dir_all(&mirror_dir).unwrap();
7790 std::fs::write(
7791 mirror_dir.join("mirror.toml"),
7792 "camp = \"yah\"\nregions = [\"pdx\"]\nworkloads = [\"yah-web\"]\n",
7793 )
7794 .unwrap();
7795
7796 let cfg = CloudConfig::load(root).unwrap();
7797 assert_eq!(cfg.legacy_mirrors.len(), 1);
7798 let mir = cfg.legacy_mirror("yah").unwrap();
7799 assert_eq!(mir.camp, "yah");
7800 assert_eq!(mir.workloads, vec!["yah-web"]);
7801 }
7802
7803 #[test]
7804 fn mirror_folder_and_flat_coexist() {
7805 // Both layouts may coexist in the same mirrors/ directory.
7806 let tmp = tempfile::TempDir::new().unwrap();
7807 let root = tmp.path();
7808 let cloud_dir = make_legacy_cloud_dir(root);
7809 let mirrors_root = cloud_dir.join("mirrors");
7810 std::fs::create_dir_all(&mirrors_root).unwrap();
7811
7812 // Flat legacy mirror
7813 std::fs::write(
7814 mirrors_root.join("noisetable.toml"),
7815 "camp = \"noisetable\"\nregions = [\"pdx\"]\nworkloads = []\n",
7816 )
7817 .unwrap();
7818
7819 // Folder-form mirror
7820 let yah_com_dir = mirrors_root.join("yah-com");
7821 std::fs::create_dir_all(&yah_com_dir).unwrap();
7822 std::fs::write(
7823 yah_com_dir.join("mirror.toml"),
7824 "camp = \"yah\"\nregions = [\"pdx\"]\nworkloads = []\n",
7825 )
7826 .unwrap();
7827
7828 let cfg = CloudConfig::load(root).unwrap();
7829 assert_eq!(cfg.legacy_mirrors.len(), 2);
7830 assert!(cfg.legacy_mirror("noisetable").is_some());
7831 assert!(cfg.legacy_mirror("yah").is_some());
7832 }
7833
7834 #[test]
7835 fn mirror_malformed_fails_with_field_path() {
7836 // A malformed mirror.toml should fail at load with a clear error
7837 // that includes the file path.
7838 let tmp = tempfile::TempDir::new().unwrap();
7839 let root = tmp.path();
7840 let cloud_dir = make_legacy_cloud_dir(root);
7841 let mirror_dir = cloud_dir.join("mirrors").join("bad");
7842 std::fs::create_dir_all(&mirror_dir).unwrap();
7843 // Missing required `camp` field
7844 std::fs::write(
7845 mirror_dir.join("mirror.toml"),
7846 "regions = [\"pdx\"]\nworkloads = []\n",
7847 )
7848 .unwrap();
7849
7850 let err = CloudConfig::load(root).unwrap_err();
7851 let msg = err.to_string();
7852 assert!(
7853 msg.contains("mirror.toml"),
7854 "error should reference the file path, got: {msg}"
7855 );
7856 }
7857
7858 #[test]
7859 fn workload_config_load_and_validate() {
7860 use workload_spec::{
7861 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
7862 RestartPolicy, StopPolicy, TenantId, TierTag, WorkloadSpec,
7863 };
7864
7865 let tmp = tempfile::TempDir::new().unwrap();
7866 let root = tmp.path();
7867 let cloud_dir = make_legacy_cloud_dir(root);
7868 std::fs::create_dir_all(cloud_dir.join("workloads")).unwrap();
7869
7870 let spec = WorkloadSpec {
7871 name: "asset-registry".into(),
7872 image: ImageRef {
7873 registry: "ghcr.io".into(),
7874 repository: "noisetable/asset-registry".into(),
7875 tag: "v1.0.0".into(),
7876 digest: workload_spec::testing::test_digest(),
7877 },
7878 tier: TierTag("tenant".into()),
7879 replicas: 1,
7880 command: None,
7881 entrypoint: None,
7882 workdir: None,
7883 user: None,
7884 env: vec![],
7885 secrets: vec![],
7886 volumes: vec![],
7887 resources: ResourceLimits {
7888 memory_mb: 256,
7889 cpu_millis: 512,
7890 memory_request_mb: None,
7891 cpu_limit_millis: None,
7892 pids_max: None,
7893 scratch_floor_mb: None,
7894 },
7895 depends_on: vec![],
7896 requires: vec![],
7897 healthcheck: None,
7898 restart_policy: RestartPolicy::Always,
7899 archetype: None,
7900 stop_policy: StopPolicy {
7901 signal: 15,
7902 grace_period: workload_spec::Millis::from_secs(10),
7903 },
7904 expose: ExposeSpec {
7905 mesh: MeshExpose {
7906 identity: MeshIdent("asset-registry.pdx".into()),
7907 ports: MeshExpose::anonymous_ports([8080]),
7908 allow_from: vec![],
7909 },
7910 public: None,
7911 operator: None,
7912 },
7913 tenant: TenantId::singleton(),
7914 namespace: NamespaceId::singleton(),
7915 labels: Default::default(),
7916 durability: None,
7917 annotations: Default::default(),
7918 files: Vec::new(),
7919 };
7920
7921 let toml_str = toml::to_string_pretty(&spec).unwrap();
7922 std::fs::write(cloud_dir.join("workloads/asset-registry.toml"), &toml_str).unwrap();
7923
7924 let cfg = CloudConfig::load(root).unwrap();
7925 assert_eq!(cfg.workloads.len(), 1);
7926 assert_eq!(cfg.workloads[0].spec.name, "asset-registry");
7927 assert_eq!(cfg.workload("asset-registry").unwrap().spec.replicas, 1);
7928 }
7929
7930 /// Minimal valid spec for the R215+ loader tests below. Kept as a helper so
7931 /// the two tests differ only in *where* the file lands, which is the whole
7932 /// thing under test.
7933 #[cfg(test)]
7934 fn minimal_spec(name: &str, replicas: u32) -> workload_spec::WorkloadSpec {
7935 use workload_spec::{
7936 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
7937 RestartPolicy, StopPolicy, TenantId, TierTag, WorkloadSpec,
7938 };
7939 WorkloadSpec {
7940 name: name.into(),
7941 image: ImageRef {
7942 registry: "cr.yah.dev".into(),
7943 repository: name.into(),
7944 tag: "v1".into(),
7945 digest: workload_spec::testing::test_digest(),
7946 },
7947 tier: TierTag("infra".into()),
7948 replicas,
7949 command: None,
7950 entrypoint: None,
7951 workdir: None,
7952 user: None,
7953 env: vec![],
7954 secrets: vec![],
7955 volumes: vec![],
7956 resources: ResourceLimits {
7957 memory_mb: 256,
7958 cpu_millis: 250,
7959 memory_request_mb: None,
7960 cpu_limit_millis: None,
7961 pids_max: None,
7962 scratch_floor_mb: None,
7963 },
7964 depends_on: vec![],
7965 requires: vec![],
7966 healthcheck: None,
7967 restart_policy: RestartPolicy::Always,
7968 archetype: None,
7969 stop_policy: StopPolicy {
7970 signal: 15,
7971 grace_period: workload_spec::Millis::from_secs(10),
7972 },
7973 expose: ExposeSpec {
7974 mesh: MeshExpose {
7975 identity: MeshIdent(name.into()),
7976 ports: MeshExpose::anonymous_ports([4325]),
7977 allow_from: vec![],
7978 },
7979 public: None,
7980 operator: None,
7981 },
7982 tenant: TenantId::singleton(),
7983 namespace: NamespaceId::singleton(),
7984 labels: Default::default(),
7985 durability: None,
7986 annotations: Default::default(),
7987 files: Vec::new(),
7988 }
7989 }
7990
7991 /// R568-T7. Workloads must load from the R215+ tree.
7992 ///
7993 /// Before the fix this function tested, `CloudConfig::load` read workloads
7994 /// ONLY from the pre-R215 `.yah/cloud/workloads/` — which R222-B1 emptied —
7995 /// so in any modern camp `cfg.workload(name)` returned `None` for every
7996 /// name and the entire `yah cloud workload …` surface was unreachable. The
7997 /// CLI's own error text has said `.yah/infra/workloads/` throughout, so the
7998 /// bug read as "you must have typoed the filename".
7999 ///
8000 /// Note the fixture writes NO legacy `.yah/cloud/` dir at all: that is the
8001 /// shape of a real post-R215 camp, and it is exactly the shape the old code
8002 /// could not serve.
8003 #[test]
8004 fn workloads_load_from_the_infra_tree() {
8005 let tmp = tempfile::TempDir::new().unwrap();
8006 let root = tmp.path();
8007 let dir = crate::paths::workloads_dir(root);
8008 std::fs::create_dir_all(&dir).unwrap();
8009 std::fs::write(
8010 dir.join("yah-cloud-admin.toml"),
8011 toml::to_string_pretty(&minimal_spec("yah-cloud-admin", 1)).unwrap(),
8012 )
8013 .unwrap();
8014
8015 let cfg = CloudConfig::load(root).unwrap();
8016 assert_eq!(cfg.workloads.len(), 1);
8017 assert_eq!(
8018 cfg.workload("yah-cloud-admin").unwrap().spec.replicas,
8019 1,
8020 "a workload declared under .yah/infra/workloads/ must be resolvable by name"
8021 );
8022 }
8023
8024 /// A camp mid-migration can have both trees. R215+ wins on a name
8025 /// collision — same precedence the machine loader applies — so moving a
8026 /// declaration into `.yah/infra/workloads/` takes effect immediately
8027 /// instead of being silently shadowed by the copy left behind.
8028 #[test]
8029 fn infra_workload_shadows_the_legacy_copy_of_the_same_name() {
8030 let tmp = tempfile::TempDir::new().unwrap();
8031 let root = tmp.path();
8032
8033 let legacy = make_legacy_cloud_dir(root);
8034 std::fs::create_dir_all(legacy.join("workloads")).unwrap();
8035 std::fs::write(
8036 legacy.join("workloads/shared.toml"),
8037 toml::to_string_pretty(&minimal_spec("shared", 9)).unwrap(),
8038 )
8039 .unwrap();
8040 // Legacy-only name, to prove the old tree is still read rather than
8041 // replaced wholesale.
8042 std::fs::write(
8043 legacy.join("workloads/legacy-only.toml"),
8044 toml::to_string_pretty(&minimal_spec("legacy-only", 3)).unwrap(),
8045 )
8046 .unwrap();
8047
8048 let infra = crate::paths::workloads_dir(root);
8049 std::fs::create_dir_all(&infra).unwrap();
8050 std::fs::write(
8051 infra.join("shared.toml"),
8052 toml::to_string_pretty(&minimal_spec("shared", 1)).unwrap(),
8053 )
8054 .unwrap();
8055
8056 let cfg = CloudConfig::load(root).unwrap();
8057 assert_eq!(cfg.workloads.len(), 2, "one `shared`, plus `legacy-only`");
8058 assert_eq!(
8059 cfg.workload("shared").unwrap().spec.replicas,
8060 1,
8061 "the .yah/infra/ copy must win over the legacy one"
8062 );
8063 assert_eq!(cfg.workload("legacy-only").unwrap().spec.replicas, 3);
8064 }
8065
8066 #[test]
8067 fn workload_loader_rejects_bad_spec() {
8068 use workload_spec::{
8069 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
8070 RestartPolicy, StopPolicy, TenantId, TierTag, WorkloadSpec,
8071 };
8072
8073 let tmp = tempfile::TempDir::new().unwrap();
8074 let root = tmp.path();
8075 let cloud_dir = make_legacy_cloud_dir(root);
8076 std::fs::create_dir_all(cloud_dir.join("workloads")).unwrap();
8077
8078 // Construct a spec that round-trips through TOML but fails shape
8079 // validation: replicas = 200 is above the max of 100.
8080 let mut spec = WorkloadSpec {
8081 name: "asset-registry".into(),
8082 image: ImageRef {
8083 registry: "ghcr.io".into(),
8084 repository: "test/app".into(),
8085 tag: "v1".into(),
8086 digest: workload_spec::testing::test_digest(),
8087 },
8088 tier: TierTag("tenant".into()),
8089 replicas: 200, // ← invalid: exceeds max 100
8090 command: None,
8091 entrypoint: None,
8092 workdir: None,
8093 user: None,
8094 env: vec![],
8095 secrets: vec![],
8096 volumes: vec![],
8097 resources: ResourceLimits {
8098 memory_mb: 256,
8099 cpu_millis: 512,
8100 memory_request_mb: None,
8101 cpu_limit_millis: None,
8102 pids_max: None,
8103 scratch_floor_mb: None,
8104 },
8105 depends_on: vec![],
8106 requires: vec![],
8107 healthcheck: None,
8108 restart_policy: RestartPolicy::Always,
8109 archetype: None,
8110 stop_policy: StopPolicy {
8111 signal: 15,
8112 grace_period: workload_spec::Millis::from_secs(10),
8113 },
8114 expose: ExposeSpec {
8115 mesh: MeshExpose {
8116 identity: MeshIdent("asset-registry.pdx".into()),
8117 ports: MeshExpose::anonymous_ports([8080]),
8118 allow_from: vec![],
8119 },
8120 public: None,
8121 operator: None,
8122 },
8123 tenant: TenantId::singleton(),
8124 namespace: NamespaceId::singleton(),
8125 labels: Default::default(),
8126 durability: None,
8127 annotations: Default::default(),
8128 files: Vec::new(),
8129 };
8130
8131 let toml_str = toml::to_string_pretty(&spec).unwrap();
8132 std::fs::write(cloud_dir.join("workloads/bad.toml"), &toml_str).unwrap();
8133
8134 let result = CloudConfig::load(root);
8135 assert!(
8136 result.is_err(),
8137 "loading a WorkloadSpec with replicas=200 should return Err"
8138 );
8139 let msg = result.unwrap_err().to_string();
8140 assert!(
8141 msg.contains("shape validation")
8142 || msg.contains("Replicas")
8143 || msg.contains("replicas"),
8144 "error should mention shape validation or replicas field, got: {msg}"
8145 );
8146
8147 // The `spec` binding is only used for the write — suppress warning.
8148 let _ = &mut spec;
8149 }
8150
8151 #[test]
8152 fn workload_config_save_round_trip() {
8153 use workload_spec::{
8154 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
8155 RestartPolicy, StopPolicy, TenantId, TierTag, WorkloadSpec,
8156 };
8157
8158 let tmp = tempfile::TempDir::new().unwrap();
8159 let root = tmp.path();
8160
8161 let spec = WorkloadSpec {
8162 name: "signing-service".into(),
8163 image: ImageRef {
8164 registry: "ghcr.io".into(),
8165 repository: "noisetable/signing".into(),
8166 tag: "v2.0.0".into(),
8167 digest: workload_spec::testing::test_digest(),
8168 },
8169 tier: TierTag("private".into()),
8170 replicas: 2,
8171 command: None,
8172 entrypoint: None,
8173 workdir: None,
8174 user: None,
8175 env: vec![],
8176 secrets: vec![],
8177 volumes: vec![],
8178 resources: ResourceLimits {
8179 memory_mb: 128,
8180 cpu_millis: 256,
8181 memory_request_mb: None,
8182 cpu_limit_millis: None,
8183 pids_max: None,
8184 scratch_floor_mb: None,
8185 },
8186 depends_on: vec![],
8187 requires: vec![],
8188 healthcheck: None,
8189 restart_policy: RestartPolicy::Always,
8190 archetype: None,
8191 stop_policy: StopPolicy {
8192 signal: 15,
8193 grace_period: workload_spec::Millis::from_secs(5),
8194 },
8195 expose: ExposeSpec {
8196 mesh: MeshExpose {
8197 identity: MeshIdent("signing.pdx".into()),
8198 ports: MeshExpose::anonymous_ports([9090]),
8199 allow_from: vec![],
8200 },
8201 public: None,
8202 operator: None,
8203 },
8204 tenant: TenantId::singleton(),
8205 namespace: NamespaceId::singleton(),
8206 labels: Default::default(),
8207 durability: None,
8208 annotations: Default::default(),
8209 files: Vec::new(),
8210 };
8211
8212 let wc = WorkloadConfig { spec };
8213 let cloud_dir = make_legacy_cloud_dir(root);
8214 wc.save(&cloud_dir).unwrap();
8215
8216 let loaded = CloudConfig::load(root).unwrap();
8217 assert_eq!(loaded.workloads.len(), 1);
8218 assert_eq!(loaded.workloads[0].spec.name, "signing-service");
8219 assert_eq!(loaded.workloads[0].spec.replicas, 2);
8220 }
8221
8222 #[test]
8223 fn machine_save_write_back_fingerprint() {
8224 let tmp = tempfile::TempDir::new().unwrap();
8225 let root = tmp.path();
8226
8227 let mut machine = MachineConfig {
8228 name: "test-pdx-1".into(),
8229 provider: "hetzner".into(),
8230 location: Some("pdx".into()),
8231 server_type: Some("cpx22".into()),
8232 hosts_mirrors: vec![],
8233 mesh_tags: vec![],
8234 region: None,
8235 zone: None,
8236 arch: None,
8237 bucket: None,
8238 vendor: None,
8239 nickname: None,
8240 legacy_hostkey_fingerprint: None,
8241 registration: Default::default(),
8242 ssh_keys: vec![],
8243 cloudflared: None,
8244 hosts_operator_bridge: false,
8245 connect: None,
8246 allocatable: None,
8247 taints: vec![],
8248 sovereign_group: None,
8249 sovereign_role: None,
8250 ingress_floating_ip: None,
8251 };
8252 machine.save(root).unwrap();
8253
8254 // Simulate A4: write back the hostkey fingerprint after provision.
8255 // R707-T1: registration is the write target; the accessor is the read.
8256 machine.registration.hostkey_fingerprint = Some("SHA256:abc123".into());
8257 machine.save(root).unwrap();
8258
8259 let reloaded: Vec<MachineConfig> = load_dir(root.join("machines")).unwrap();
8260 assert_eq!(reloaded.len(), 1);
8261 assert_eq!(reloaded[0].hostkey_fingerprint(), Some("SHA256:abc123"));
8262 }
8263
8264 // ─── New-shape (R222 B2) parse tests ────────────────────────────────────
8265 //
8266 // These mirror the Phase-A manifests committed under `.yah/services/` and
8267 // `.yah/infra/providers/`. Keeping the test strings inline (rather than
8268 // reading the on-disk files) so the loader stays runnable in any workdir
8269 // and so accidental edits to the on-disk files don't silently change
8270 // schema expectations.
8271
8272 #[test]
8273 fn provider_cloudflare_round_trips() {
8274 let src = r#"
8275schema_version = 1
8276id = "cloudflare"
8277kind = "cloudflare"
8278credentials = "keystore://cloudflare/yah"
8279default_zone = "yah.dev"
8280"#;
8281 let cfg: ProviderConfig = toml::from_str(src).unwrap();
8282 assert_eq!(cfg.id, "cloudflare");
8283 assert_eq!(cfg.kind, Provider::Cloudflare);
8284 assert_eq!(
8285 cfg.credentials.as_deref(),
8286 Some("keystore://cloudflare/yah")
8287 );
8288 assert_eq!(
8289 cfg.fields.get("default_zone").and_then(|v| v.as_str()),
8290 Some("yah.dev"),
8291 );
8292 let back = toml::to_string(&cfg).unwrap();
8293 let again: ProviderConfig = toml::from_str(&back).unwrap();
8294 assert_eq!(again.id, cfg.id);
8295 assert_eq!(again.kind, cfg.kind);
8296 }
8297
8298 #[test]
8299 fn provider_hetzner_round_trips() {
8300 let src = r#"
8301schema_version = 1
8302id = "hetzner"
8303kind = "hetzner"
8304credentials = "keystore://hetzner/yah"
8305default_location = "pdx"
8306default_server_type = "cpx11"
8307ssh_keys = []
8308"#;
8309 let cfg: ProviderConfig = toml::from_str(src).unwrap();
8310 assert_eq!(cfg.kind, Provider::Hetzner);
8311 assert_eq!(
8312 cfg.fields.get("default_location").and_then(|v| v.as_str()),
8313 Some("pdx"),
8314 );
8315 assert!(
8316 cfg.fields
8317 .get("ssh_keys")
8318 .map(|v| v.as_array().unwrap().is_empty())
8319 .unwrap_or(false),
8320 "ssh_keys must round-trip as empty array, got {:?}",
8321 cfg.fields.get("ssh_keys"),
8322 );
8323 }
8324
8325 #[test]
8326 fn provider_orbstack_local_container_round_trips() {
8327 let src = r#"
8328schema_version = 1
8329id = "orbstack"
8330kind = "local-container"
8331runtime = "auto"
8332
8333[discovery]
8334orbstack = "~/.orbstack/run/docker.sock"
8335colima = "~/.colima/default/docker.sock"
8336docker = "/var/run/docker.sock"
8337"#;
8338 let cfg: ProviderConfig = toml::from_str(src).unwrap();
8339 assert_eq!(cfg.kind, Provider::LocalContainer);
8340 assert_eq!(
8341 cfg.fields.get("runtime").and_then(|v| v.as_str()),
8342 Some("auto"),
8343 );
8344 let discovery = cfg
8345 .fields
8346 .get("discovery")
8347 .and_then(|v| v.as_table())
8348 .expect("discovery table");
8349 assert!(discovery.contains_key("orbstack"));
8350 assert!(discovery.contains_key("colima"));
8351 assert!(discovery.contains_key("docker"));
8352 }
8353
8354 #[test]
8355 fn provider_unknown_kind_fails() {
8356 let src = r#"
8357schema_version = 1
8358id = "made-up"
8359kind = "fly-io"
8360"#;
8361 let err = toml::from_str::<ProviderConfig>(src).unwrap_err();
8362 let msg = err.to_string();
8363 assert!(
8364 msg.contains("kind") || msg.contains("variant"),
8365 "unknown provider kind should surface as a serde error, got: {msg}"
8366 );
8367 }
8368
8369 #[test]
8370 fn service_dev_yah_round_trips() {
8371 let src = r#"
8372schema_version = 1
8373name = "dev-yah"
8374[address]
8375kind = "front-door"
8376domain = "yah.dev"
8377
8378[[components]]
8379id = "site"
8380kind = "mesofact-static"
8381path = "app/yah/web"
8382role = "static"
8383"#;
8384 let cfg: ServiceConfig = toml::from_str(src).unwrap();
8385 assert_eq!(cfg.name, "dev-yah");
8386 assert_eq!(cfg.domain(), Some("yah.dev"));
8387 assert_eq!(cfg.components.len(), 1);
8388 let c = &cfg.components[0];
8389 assert_eq!(c.id, "site");
8390 assert_eq!(c.kind, "mesofact-static");
8391 assert_eq!(c.path, "app/yah/web");
8392 assert_eq!(c.role, "static");
8393 assert!(c.publishes.is_none());
8394
8395 let back = toml::to_string(&cfg).unwrap();
8396 let again: ServiceConfig = toml::from_str(&back).unwrap();
8397 assert_eq!(again.name, cfg.name);
8398 assert_eq!(again.components[0].kind, c.kind);
8399 }
8400
8401 #[test]
8402 fn mirror_prod_cloudflare_reference_parses() {
8403 let src = r#"
8404schema_version = 1
8405shape = "single-machine"
8406
8407[providers.static]
8408use = "cloudflare"
8409bucket = "yah-dev"
8410zone = "yah.dev"
8411dns = { record = "@", type = "CNAME" }
8412"#;
8413 let cfg: MirrorConfig = toml::from_str(src).unwrap();
8414 assert_eq!(cfg.shape, MirrorShape::SingleMachine);
8415 let slot = cfg.providers.get("static").expect("static slot");
8416 assert_eq!(slot.provider_id(), Some("cloudflare"));
8417 assert!(slot.inline_kind().is_none());
8418 if let MirrorProviderSlot::Reference { fields, .. } = slot {
8419 assert_eq!(
8420 fields.get("bucket").and_then(|v| v.as_str()),
8421 Some("yah-dev")
8422 );
8423 assert_eq!(fields.get("zone").and_then(|v| v.as_str()), Some("yah.dev"));
8424 let dns = fields
8425 .get("dns")
8426 .and_then(|v| v.as_table())
8427 .expect("dns table");
8428 assert_eq!(dns.get("record").and_then(|v| v.as_str()), Some("@"));
8429 assert_eq!(dns.get("type").and_then(|v| v.as_str()), Some("CNAME"));
8430 } else {
8431 panic!("expected Reference slot");
8432 }
8433 }
8434
8435 #[test]
8436 fn mirror_local_inline_static_and_orbstack_compute_parse() {
8437 let src = r#"
8438schema_version = 1
8439shape = "local"
8440
8441[providers.static]
8442kind = "miniflare-native"
8443port = 4321
8444artifact_dir = ".yah/infra/state/local/static"
8445
8446[providers.compute]
8447use = "orbstack"
8448"#;
8449 let cfg: MirrorConfig = toml::from_str(src).unwrap();
8450 assert_eq!(cfg.shape, MirrorShape::Local);
8451
8452 let static_slot = cfg.providers.get("static").expect("static slot");
8453 assert_eq!(static_slot.inline_kind(), Some(Provider::MiniflareNative));
8454 assert!(static_slot.provider_id().is_none());
8455 if let MirrorProviderSlot::Inline { fields, .. } = static_slot {
8456 assert_eq!(fields.get("port").and_then(|v| v.as_integer()), Some(4321));
8457 assert_eq!(
8458 fields.get("artifact_dir").and_then(|v| v.as_str()),
8459 Some(".yah/infra/state/local/static"),
8460 );
8461 } else {
8462 panic!("expected Inline slot for static");
8463 }
8464
8465 let compute_slot = cfg.providers.get("compute").expect("compute slot");
8466 assert_eq!(compute_slot.provider_id(), Some("orbstack"));
8467 }
8468
8469 #[test]
8470 fn mirror_pond_miniflare_minio_parse() {
8471 // pond-tier mirror: miniflare-container + minio, both inline.
8472 // T1 just needs these inline kinds to parse — the reconciler dispatch
8473 // arrives in R256-T3.
8474 let src = r#"
8475schema_version = 1
8476shape = "local"
8477
8478[providers.static]
8479kind = "miniflare-container"
8480port = 4322
8481bucket = "yah-dev"
8482
8483[providers.object_store]
8484kind = "minio-container"
8485api_port = 9000
8486console_port = 9001
8487bucket = "yah-dev"
8488"#;
8489 let cfg: MirrorConfig = toml::from_str(src).unwrap();
8490 assert_eq!(cfg.shape, MirrorShape::Local);
8491
8492 let static_slot = cfg.providers.get("static").expect("static slot");
8493 assert_eq!(
8494 static_slot.inline_kind(),
8495 Some(Provider::MiniflareContainer)
8496 );
8497 if let MirrorProviderSlot::Inline { fields, .. } = static_slot {
8498 assert_eq!(fields.get("port").and_then(|v| v.as_integer()), Some(4322));
8499 assert_eq!(
8500 fields.get("bucket").and_then(|v| v.as_str()),
8501 Some("yah-dev")
8502 );
8503 } else {
8504 panic!("expected Inline slot for miniflare-container static");
8505 }
8506
8507 let object_store_slot = cfg
8508 .providers
8509 .get("object_store")
8510 .expect("object_store slot");
8511 assert_eq!(
8512 object_store_slot.inline_kind(),
8513 Some(Provider::MinioContainer)
8514 );
8515 if let MirrorProviderSlot::Inline { fields, .. } = object_store_slot {
8516 assert_eq!(
8517 fields.get("api_port").and_then(|v| v.as_integer()),
8518 Some(9000)
8519 );
8520 assert_eq!(
8521 fields.get("console_port").and_then(|v| v.as_integer()),
8522 Some(9001)
8523 );
8524 assert_eq!(
8525 fields.get("bucket").and_then(|v| v.as_str()),
8526 Some("yah-dev")
8527 );
8528 } else {
8529 panic!("expected Inline slot for minio-container object_store");
8530 }
8531 }
8532
8533 #[test]
8534 fn provider_miniflare_container_kind_round_trips() {
8535 // Inline-only kind; never declared as a standalone provider file but
8536 // the enum round-trip is still exercised through ProviderConfig because
8537 // schemars/serde share the variant table.
8538 let cfg = MirrorProviderSlot::Inline {
8539 kind: Provider::MiniflareContainer,
8540 fields: BTreeMap::new(),
8541 };
8542 let s = toml::to_string(&cfg).unwrap();
8543 assert!(
8544 s.contains("kind = \"miniflare-container\""),
8545 "kebab-case wire form expected, got: {s}"
8546 );
8547 let back: MirrorProviderSlot = toml::from_str(&s).unwrap();
8548 assert_eq!(back.inline_kind(), Some(Provider::MiniflareContainer));
8549 }
8550
8551 #[test]
8552 fn provider_minio_container_kind_round_trips() {
8553 let cfg = MirrorProviderSlot::Inline {
8554 kind: Provider::MinioContainer,
8555 fields: BTreeMap::new(),
8556 };
8557 let s = toml::to_string(&cfg).unwrap();
8558 assert!(
8559 s.contains("kind = \"minio-container\""),
8560 "kebab-case wire form expected, got: {s}"
8561 );
8562 let back: MirrorProviderSlot = toml::from_str(&s).unwrap();
8563 assert_eq!(back.inline_kind(), Some(Provider::MinioContainer));
8564 }
8565
8566 #[test]
8567 fn mirror_compute_slot_with_machine_reference_parses() {
8568 // The on-disk prod.toml has a commented-out compute slot; this test
8569 // covers the form Phase B will need once yubaba is provisioned.
8570 let src = r#"
8571schema_version = 1
8572shape = "single-machine"
8573
8574[providers.compute]
8575use = "hetzner"
8576machine = "yah-cloud-1"
8577"#;
8578 let cfg: MirrorConfig = toml::from_str(src).unwrap();
8579 let slot = cfg.providers.get("compute").expect("compute slot");
8580 assert_eq!(slot.provider_id(), Some("hetzner"));
8581 if let MirrorProviderSlot::Reference { fields, .. } = slot {
8582 assert_eq!(
8583 fields.get("machine").and_then(|v| v.as_str()),
8584 Some("yah-cloud-1"),
8585 );
8586 }
8587 }
8588
8589 #[test]
8590 fn machine_yah_cloud_1_round_trips_with_existing_shape() {
8591 // The current machine TOML predates B2 — MachineConfig hasn't been
8592 // reshaped yet. This locks the expected shape so we notice if B3
8593 // accidentally regresses it.
8594 let src = r#"
8595name = "yah-cloud-1"
8596provider = "hetzner"
8597location = "pdx"
8598server_type = "cpx11"
8599hosts_mirrors = []
8600mesh_tags = ["tag:tier-scratch", "tag:primary-yah"]
8601ssh_keys = [111513970, 111525493]
8602"#;
8603 let cfg: MachineConfig = toml::from_str(src).unwrap();
8604 assert_eq!(cfg.name, "yah-cloud-1");
8605 assert_eq!(cfg.provider, "hetzner");
8606 assert_eq!(cfg.ssh_keys.len(), 2);
8607 }
8608
8609 #[test]
8610 fn static_node_omits_location_server_type_and_carries_connect() {
8611 // BYO Phase-0: a `static` node we brought up over SSH has no provider
8612 // DC code or SKU; it declares reach in `[connect]` instead. Must load.
8613 let src = r#"
8614name = "us-south-001"
8615provider = "static"
8616region = "us-south"
8617mesh_tags = ["tag:cloud-runner", "tag:voter-candidate"]
8618
8619[connect]
8620address = "45.32.194.254"
8621ssh = "root@45.32.194.254"
8622identity_file = "~/.ssh/yah"
8623yubaba = "http://127.0.0.1:7443"
8624arch = "x86_64"
8625"#;
8626 let cfg: MachineConfig = toml::from_str(src).unwrap();
8627 assert_eq!(cfg.provider, "static");
8628 assert!(cfg.location.is_none());
8629 assert!(cfg.server_type.is_none());
8630 assert_eq!(cfg.location(), ""); // accessor defaults empty
8631 let c = cfg.connect.as_ref().expect("connect block");
8632 assert_eq!(c.ssh, "root@45.32.194.254");
8633 // Loopback is a *declared* reach placeholder, so it stays in [connect]
8634 // verbatim and composes straight through (R707-T1).
8635 assert_eq!(c.yubaba.as_deref(), Some("http://127.0.0.1:7443"));
8636 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://127.0.0.1:7443"));
8637 assert_eq!(cfg.mesh_ipv4(), None);
8638 // Static providers have no driver, so validate() is a no-op pass.
8639 assert!(!provider_has_machine_driver(&cfg.provider));
8640 cfg.validate().unwrap();
8641 }
8642
8643 // ─── R707-T1: declaration / registration split ──────────────────────────
8644
8645 /// The pre-split shape — top-level `hostkey_fingerprint`, mesh IP baked
8646 /// into `[connect].yubaba` — must keep parsing, and must read back through
8647 /// the accessors identically. Every machine TOML in the fleet was written
8648 /// this way, and other camps' inventories still are.
8649 #[test]
8650 fn legacy_shape_still_parses_and_reads_through_accessors() {
8651 let src = r#"
8652name = "us-west-001"
8653provider = "static"
8654region = "us-west"
8655arch = "x86_64"
8656mesh_tags = ["tag:cloud-runner"]
8657hostkey_fingerprint = "SHA256:dmpq"
8658
8659[connect]
8660address = "15.204.89.240"
8661ssh = "debian@15.204.89.240"
8662identity_file = "~/.ssh/yah"
8663yubaba = "http://100.64.0.1:7443"
8664"#;
8665 let cfg: MachineConfig = toml::from_str(src).unwrap();
8666 assert_eq!(cfg.hostkey_fingerprint(), Some("SHA256:dmpq"));
8667 assert_eq!(cfg.mesh_ipv4(), Some("100.64.0.1"));
8668 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://100.64.0.1:7443"));
8669 }
8670
8671 /// The post-split shape reads identically to the legacy one above — same
8672 /// three accessor answers from a file that separates the two halves. This
8673 /// is the "unchanged in meaning" guarantee the fleet migration rests on.
8674 #[test]
8675 fn split_shape_is_equivalent_to_legacy_shape() {
8676 let legacy = r#"
8677name = "m"
8678provider = "static"
8679mesh_tags = []
8680hostkey_fingerprint = "SHA256:dmpq"
8681
8682[connect]
8683address = "15.204.89.240"
8684ssh = "debian@15.204.89.240"
8685identity_file = "~/.ssh/yah"
8686yubaba = "http://100.64.0.1:7443"
8687"#;
8688 let split = r#"
8689name = "m"
8690provider = "static"
8691mesh_tags = []
8692
8693[connect]
8694address = "15.204.89.240"
8695ssh = "debian@15.204.89.240"
8696identity_file = "~/.ssh/yah"
8697
8698[registration]
8699hostkey_fingerprint = "SHA256:dmpq"
8700mesh_ipv4 = "100.64.0.1"
8701"#;
8702 let old: MachineConfig = toml::from_str(legacy).unwrap();
8703 let new: MachineConfig = toml::from_str(split).unwrap();
8704 assert_eq!(old.hostkey_fingerprint(), new.hostkey_fingerprint());
8705 assert_eq!(old.mesh_ipv4(), new.mesh_ipv4());
8706 assert_eq!(old.yubaba_url(), new.yubaba_url());
8707 }
8708
8709 /// A non-default `[connect].yubaba_port` is declared reach and composes
8710 /// with the observed mesh address rather than being pinned into a URL.
8711 #[test]
8712 fn declared_port_composes_with_observed_mesh_address() {
8713 let src = r#"
8714name = "m"
8715provider = "static"
8716mesh_tags = []
8717
8718[connect]
8719address = "10.0.0.1"
8720ssh = "yah@10.0.0.1"
8721identity_file = "~/.ssh/yah"
8722yubaba_port = 9443
8723
8724[registration]
8725mesh_ipv4 = "100.64.0.9"
8726"#;
8727 let cfg: MachineConfig = toml::from_str(src).unwrap();
8728 assert_eq!(cfg.connect.as_ref().unwrap().yubaba_port(), 9443);
8729 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://100.64.0.9:9443"));
8730 }
8731
8732 /// R605-T10 inverts R707-T6 for the private-literal case, and this is the
8733 /// node it was inverted for: us-west-014's shape, mesh-joined AND declaring
8734 /// a LAN `[connect].yubaba`. R707-T6 made the literal win outright so
8735 /// `rollout::yubaba::membership_to_nodes` could match the dev group's
8736 /// LAN-addressed raft membership — which fused identity into reach and made
8737 /// every automated dial go to an address only bldg-2506 can route.
8738 /// `lan_endpoint()` now serves that match, so the mesh address wins the
8739 /// dial and the literal is inert.
8740 #[test]
8741 fn a_private_literal_loses_to_the_registered_mesh_address() {
8742 let src = r#"
8743name = "us-west-014"
8744provider = "static"
8745mesh_tags = []
8746
8747[connect]
8748address = "192.168.10.14"
8749ssh = "yah@192.168.10.14"
8750identity_file = "~/.ssh/yah"
8751yubaba = "http://192.168.10.14:7443"
8752
8753[registration]
8754mesh_ipv4 = "100.64.0.6"
8755"#;
8756 let cfg: MachineConfig = toml::from_str(src).unwrap();
8757 assert_eq!(cfg.mesh_ipv4(), Some("100.64.0.6"), "still mesh-joined");
8758 assert_eq!(
8759 cfg.yubaba_url().as_deref(),
8760 Some("http://100.64.0.6:7443"),
8761 "automation dials the mesh, never the LAN literal"
8762 );
8763 assert_eq!(
8764 cfg.lan_endpoint().as_deref(),
8765 Some("192.168.10.14:7443"),
8766 "the LAN address is still recorded — as identity, not as reach"
8767 );
8768 }
8769
8770 /// The refusal R605-T10 asks for: a node whose ONLY declared reach is a LAN
8771 /// literal is unresolvable, and says so by name rather than returning a URL
8772 /// that will time out. us-west-011's shape before this ticket.
8773 #[test]
8774 fn a_lan_only_node_refuses_with_a_named_reason() {
8775 let src = r#"
8776name = "us-west-011"
8777provider = "static"
8778mesh_tags = []
8779
8780[connect]
8781address = "192.168.10.11"
8782ssh = "yah@192.168.10.11"
8783identity_file = "~/.ssh/yah"
8784yubaba = "http://192.168.10.11:7443"
8785"#;
8786 let cfg: MachineConfig = toml::from_str(src).unwrap();
8787 assert_eq!(cfg.yubaba_url(), None);
8788 let err = cfg.reach().unwrap_err();
8789 assert!(err.contains("us-west-011"), "{err}");
8790 assert!(err.contains("192.168.10.11"), "{err}");
8791 assert!(err.contains("mesh_ipv4"), "{err}");
8792 }
8793
8794 /// The loopback placeholder is a genuine declaration ("reach me through the
8795 /// SSH tunnel"), not a LAN literal — 127/8 is not RFC1918. It must keep
8796 /// resolving verbatim; `hub::coordinator::is_loopback_url` is what judges it
8797 /// downstream.
8798 #[test]
8799 fn a_loopback_placeholder_still_resolves_verbatim() {
8800 let src = r#"
8801name = "m"
8802provider = "static"
8803mesh_tags = []
8804
8805[connect]
8806address = "192.168.10.99"
8807ssh = "yah@192.168.10.99"
8808identity_file = "~/.ssh/yah"
8809yubaba = "http://127.0.0.1:7443"
8810"#;
8811 let cfg: MachineConfig = toml::from_str(src).unwrap();
8812 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://127.0.0.1:7443"));
8813 }
8814
8815 #[test]
8816 fn private_ranges_are_exactly_rfc1918() {
8817 for lan in [
8818 "http://192.168.10.11:7443",
8819 "http://10.0.0.5:7443",
8820 "http://172.16.4.1:7443",
8821 ] {
8822 assert!(private_ipv4_from_url(lan).is_some(), "{lan}");
8823 }
8824 for not_lan in [
8825 "http://100.64.0.6:7443", // mesh
8826 "http://127.0.0.1:7443", // loopback
8827 "http://172.32.0.1:7443", // just past 172.16/12
8828 "http://45.32.194.254:80", // public
8829 "http://us-west-001:7443", // name, not a literal
8830 ] {
8831 assert!(private_ipv4_from_url(not_lan).is_none(), "{not_lan}");
8832 }
8833 }
8834
8835 /// `normalize` migrates in place: the legacy fingerprint moves into
8836 /// `[registration]`, the mesh IP is lifted out of the URL, and the derived
8837 /// `[connect].yubaba` is cleared so the two halves cannot drift.
8838 #[test]
8839 fn normalize_migrates_legacy_fields_and_is_idempotent() {
8840 let src = r#"
8841name = "m"
8842provider = "static"
8843mesh_tags = []
8844hostkey_fingerprint = "SHA256:dmpq"
8845
8846[connect]
8847address = "15.204.89.240"
8848ssh = "debian@15.204.89.240"
8849identity_file = "~/.ssh/yah"
8850yubaba = "http://100.64.0.1:7443"
8851"#;
8852 let mut cfg: MachineConfig = toml::from_str(src).unwrap();
8853 cfg.normalize();
8854 assert!(cfg.legacy_hostkey_fingerprint.is_none());
8855 assert_eq!(
8856 cfg.registration.hostkey_fingerprint.as_deref(),
8857 Some("SHA256:dmpq")
8858 );
8859 assert_eq!(cfg.registration.mesh_ipv4.as_deref(), Some("100.64.0.1"));
8860 assert!(cfg.connect.as_ref().unwrap().yubaba.is_none());
8861 // Accessors still answer the same, and re-running changes nothing.
8862 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://100.64.0.1:7443"));
8863 let once = format!("{cfg:?}");
8864 cfg.normalize();
8865 assert_eq!(once, format!("{cfg:?}"));
8866 }
8867
8868 /// A loopback `[connect].yubaba` is a declaration ("no mesh address yet —
8869 /// reach me through the SSH tunnel"), not a stale observation, so
8870 /// `normalize` must leave it alone. us-west-003/011/013 depend on this.
8871 #[test]
8872 fn normalize_leaves_pre_mesh_loopback_declaration_intact() {
8873 let src = r#"
8874name = "m"
8875provider = "static"
8876mesh_tags = []
8877
8878[connect]
8879address = "192.168.10.11"
8880ssh = "yah@192.168.10.11"
8881identity_file = "~/.ssh/yah"
8882yubaba = "http://127.0.0.1:7443"
8883"#;
8884 let mut cfg: MachineConfig = toml::from_str(src).unwrap();
8885 cfg.normalize();
8886 assert_eq!(
8887 cfg.connect.as_ref().unwrap().yubaba.as_deref(),
8888 Some("http://127.0.0.1:7443")
8889 );
8890 assert!(cfg.registration.is_empty());
8891 assert_eq!(cfg.mesh_ipv4(), None);
8892 }
8893
8894 /// `save` normalizes, so a legacy file that round-trips through the writer
8895 /// comes back on the split shape with nothing lost — the property that
8896 /// keeps `yah cloud machine attach` from re-emitting the old layout.
8897 #[test]
8898 fn save_writes_the_split_shape_from_a_legacy_config() {
8899 let tmp = tempfile::TempDir::new().unwrap();
8900 let root = tmp.path();
8901 let src = r#"
8902name = "m"
8903provider = "static"
8904mesh_tags = []
8905hostkey_fingerprint = "SHA256:dmpq"
8906
8907[connect]
8908address = "15.204.89.240"
8909ssh = "debian@15.204.89.240"
8910identity_file = "~/.ssh/yah"
8911yubaba = "http://100.64.0.1:7443"
8912"#;
8913 let cfg: MachineConfig = toml::from_str(src).unwrap();
8914 cfg.save(root).unwrap();
8915
8916 let written = std::fs::read_to_string(root.join("machines/m.toml")).unwrap();
8917 let reg_at = written
8918 .find("[registration]")
8919 .unwrap_or_else(|| panic!("no [registration] table: {written}"));
8920 let fp_at = written
8921 .find("hostkey_fingerprint")
8922 .unwrap_or_else(|| panic!("fingerprint dropped: {written}"));
8923 assert!(
8924 fp_at > reg_at,
8925 "legacy top-level field must not be re-emitted: {written}"
8926 );
8927 assert!(
8928 !written.contains("yubaba ="),
8929 "derived URL must not be re-emitted alongside mesh_ipv4: {written}"
8930 );
8931
8932 let reloaded: MachineConfig = toml::from_str(&written).unwrap();
8933 assert_eq!(reloaded.hostkey_fingerprint(), Some("SHA256:dmpq"));
8934 assert_eq!(
8935 reloaded.yubaba_url().as_deref(),
8936 Some("http://100.64.0.1:7443")
8937 );
8938 }
8939
8940 /// `[registration]` is omitted entirely for a machine nothing has been
8941 /// observed about — a scaffolded declaration stays clean.
8942 #[test]
8943 fn empty_registration_is_omitted_on_serialize() {
8944 let src = r#"
8945name = "m"
8946provider = "static"
8947mesh_tags = []
8948"#;
8949 let cfg: MachineConfig = toml::from_str(src).unwrap();
8950 assert!(cfg.registration.is_empty());
8951 let out = toml::to_string_pretty(&cfg).unwrap();
8952 assert!(!out.contains("[registration]"), "{out}");
8953 }
8954
8955 #[test]
8956 fn driver_provider_without_location_fails_validate() {
8957 // A driver-backed provider (hetzner/vultr) still MUST carry location +
8958 // server_type — the driver can't create a server without them. The
8959 // contract moved from load-time (required field) to provision-time
8960 // (validate), so the TOML loads but validate() rejects it.
8961 let src = r#"
8962name = "us-west-001"
8963provider = "hetzner"
8964mesh_tags = []
8965"#;
8966 let cfg: MachineConfig = toml::from_str(src).unwrap();
8967 assert!(provider_has_machine_driver(&cfg.provider));
8968 let err = cfg.validate().unwrap_err().to_string();
8969 assert!(
8970 err.contains("location"),
8971 "expected location complaint: {err}"
8972 );
8973 }
8974
8975 /// Helper for the new-tree integration tests below: lay out
8976 /// `<workspace>/.yah/{infra,services}/` with `dev-yah` + its mirrors and
8977 /// the three Phase-A providers (cloudflare, hetzner, orbstack).
8978 fn make_new_tree_with_dev_yah(root: &std::path::Path) {
8979 let infra = root.join(".yah").join("infra");
8980 let providers = infra.join("providers");
8981 std::fs::create_dir_all(&providers).unwrap();
8982 std::fs::write(
8983 providers.join("cloudflare.toml"),
8984 r#"schema_version = 1
8985id = "cloudflare"
8986kind = "cloudflare"
8987credentials = "keystore://cloudflare/yah"
8988default_zone = "yah.dev"
8989"#,
8990 )
8991 .unwrap();
8992 std::fs::write(
8993 providers.join("hetzner.toml"),
8994 r#"schema_version = 1
8995id = "hetzner"
8996kind = "hetzner"
8997credentials = "keystore://hetzner/yah"
8998default_location = "pdx"
8999default_server_type = "cpx11"
9000ssh_keys = []
9001"#,
9002 )
9003 .unwrap();
9004 std::fs::write(
9005 providers.join("orbstack.toml"),
9006 r#"schema_version = 1
9007id = "orbstack"
9008kind = "local-container"
9009runtime = "auto"
9010
9011[discovery]
9012orbstack = "~/.orbstack/run/docker.sock"
9013"#,
9014 )
9015 .unwrap();
9016
9017 let svc = root.join(".yah").join("services").join("dev-yah");
9018 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
9019 std::fs::write(
9020 svc.join("service.toml"),
9021 r#"schema_version = 1
9022name = "dev-yah"
9023[address]
9024kind = "front-door"
9025domain = "yah.dev"
9026
9027[[components]]
9028id = "site"
9029kind = "mesofact-static"
9030path = "app/yah/web"
9031role = "static"
9032"#,
9033 )
9034 .unwrap();
9035 std::fs::write(
9036 svc.join("mirrors/cloud.toml"),
9037 r#"schema_version = 1
9038shape = "single-machine"
9039
9040[providers.static]
9041use = "cloudflare"
9042bucket = "yah-dev"
9043zone = "yah.dev"
9044"#,
9045 )
9046 .unwrap();
9047 std::fs::write(
9048 svc.join("mirrors/local.toml"),
9049 r#"schema_version = 1
9050shape = "local"
9051
9052[providers.static]
9053kind = "miniflare-native"
9054port = 4321
9055
9056[providers.compute]
9057use = "orbstack"
9058"#,
9059 )
9060 .unwrap();
9061 }
9062
9063 #[test]
9064 fn cloud_config_load_new_tree_populates_providers_and_services() {
9065 let tmp = tempfile::TempDir::new().unwrap();
9066 let root = tmp.path();
9067 make_new_tree_with_dev_yah(root);
9068
9069 let cfg = CloudConfig::load(root).unwrap();
9070 assert_eq!(cfg.providers.len(), 3, "three providers loaded");
9071 assert!(cfg.provider("cloudflare").is_some());
9072 assert!(cfg.provider("hetzner").is_some());
9073 assert!(cfg.provider("orbstack").is_some());
9074
9075 let dev = cfg.service("dev-yah").expect("dev-yah service");
9076 assert_eq!(dev.service.domain(), Some("yah.dev"));
9077 assert_eq!(dev.service.components.len(), 1);
9078 assert_eq!(dev.mirrors.len(), 2);
9079 // Legacy file stems "cloud" and "local" are normalised to canonical tier names.
9080 assert!(dev.mirrors.contains_key("prod"), "cloud.toml → prod tier");
9081 assert!(dev.mirrors.contains_key("dev"), "local.toml → dev tier");
9082 assert_eq!(dev.mirrors["prod"].shape, MirrorShape::SingleMachine);
9083 assert_eq!(dev.mirrors["dev"].shape, MirrorShape::Local);
9084
9085 // Legacy fields stay empty when no .yah/cloud/ exists.
9086 assert!(cfg.legacy_mirrors.is_empty());
9087 assert!(cfg.workloads.is_empty());
9088 }
9089
9090 #[test]
9091 fn cloud_config_cross_ref_fails_on_missing_provider() {
9092 // Mirror references a provider id that doesn't exist.
9093 let tmp = tempfile::TempDir::new().unwrap();
9094 let root = tmp.path();
9095 let svc = root.join(".yah").join("services").join("dev-yah");
9096 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
9097 std::fs::write(
9098 svc.join("service.toml"),
9099 "schema_version = 1\nname = \"dev-yah\"\n[address]\nkind = \"front-door\"\ndomain = \"yah.dev\"\n",
9100 )
9101 .unwrap();
9102 std::fs::write(
9103 svc.join("mirrors/prod.toml"),
9104 "schema_version = 1\nshape = \"single-machine\"\n\n[providers.static]\nuse = \"fly-io\"\n",
9105 ).unwrap();
9106
9107 let err = CloudConfig::load(root).unwrap_err();
9108 let msg = err.to_string();
9109 assert!(
9110 msg.contains("fly-io"),
9111 "error should name the missing provider id, got: {msg}"
9112 );
9113 assert!(
9114 msg.contains("providers/fly-io.toml") || msg.contains("no such provider"),
9115 "error should hint at remedy, got: {msg}"
9116 );
9117 }
9118
9119 /// R905. `[build.<id>]` is the per-environment build override, and it is
9120 /// keyed by component id — so a key naming no declared component is a
9121 /// silent no-op: the environment goes on building with the command the
9122 /// operator believed they had replaced.
9123 #[test]
9124 fn cloud_config_cross_ref_fails_on_a_build_override_for_an_unknown_component() {
9125 let tmp = tempfile::TempDir::new().unwrap();
9126 let root = tmp.path();
9127 let svc = root.join(".yah").join("services").join("dev-yah");
9128 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
9129 std::fs::write(
9130 svc.join("service.toml"),
9131 "schema_version = 1\nname = \"dev-yah\"\n[address]\nkind = \"front-door\"\ndomain = \"yah.dev\"\n\n\
9132 [[components]]\nid = \"site\"\nkind = \"mesofact-spa\"\n\
9133 path = \"web/landing\"\nrole = \"static\"\n",
9134 )
9135 .unwrap();
9136 std::fs::write(
9137 svc.join("mirrors/staging.toml"),
9138 "schema_version = 1\nshape = \"single-machine\"\n\n\
9139 [build.sight]\ncommand = \"bun run build:staging\"\n",
9140 )
9141 .unwrap();
9142
9143 let msg = CloudConfig::load(root).unwrap_err().to_string();
9144 assert!(
9145 msg.contains("build.sight") && msg.contains("site"),
9146 "error should name the bad key and the declared ids, got: {msg}"
9147 );
9148 }
9149
9150 /// The same mirror, spelled correctly, loads and resolves — including the
9151 /// `env` half, which is the knob a project uses when it does not want a
9152 /// sibling `build:<env>` script per environment (R905).
9153 #[test]
9154 fn a_mirror_build_override_resolves_by_component_id() {
9155 let tmp = tempfile::TempDir::new().unwrap();
9156 let root = tmp.path();
9157 let svc = root.join(".yah").join("services").join("dev-yah");
9158 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
9159 std::fs::write(
9160 svc.join("service.toml"),
9161 "schema_version = 1\nname = \"dev-yah\"\n[address]\nkind = \"front-door\"\ndomain = \"yah.dev\"\n\n\
9162 [[components]]\nid = \"site\"\nkind = \"mesofact-spa\"\n\
9163 path = \"web/landing\"\nrole = \"static\"\n\n\
9164 [[components]]\nid = \"app\"\nkind = \"mesofact-static\"\n\
9165 path = \"app/browser\"\nrole = \"static\"\nmount = \"/app\"\n",
9166 )
9167 .unwrap();
9168 std::fs::write(
9169 svc.join("mirrors/staging.toml"),
9170 "schema_version = 1\nshape = \"single-machine\"\n\n\
9171 [build.site]\ncommand = \"bun run build:staging\"\n\n\
9172 [build.site.env]\nAPI_ORIGIN = \"https://api-staging.example.com\"\n",
9173 )
9174 .unwrap();
9175
9176 let cfg = CloudConfig::load(root).unwrap();
9177 let mirror = &cfg.service("dev-yah").unwrap().mirrors["staging"];
9178
9179 let site = mirror.build_override("site").expect("site override");
9180 assert_eq!(site.command.as_deref(), Some("bun run build:staging"));
9181 assert_eq!(
9182 site.env_pairs(),
9183 vec![(
9184 "API_ORIGIN".to_string(),
9185 "https://api-staging.example.com".to_string()
9186 )]
9187 );
9188 // A sibling component under the same mirror is untouched — the
9189 // override is per component, not per mirror.
9190 assert!(mirror.build_override("app").is_none());
9191 }
9192
9193 /// An override that names nothing reads as no override at all, so callers
9194 /// can treat `Some(_)` as "something differs here" (R905).
9195 #[test]
9196 fn an_empty_build_override_reads_as_absent() {
9197 let empty = MirrorBuildOverride::default();
9198 assert!(empty.is_empty());
9199 let mut m = mirror("");
9200 m.build.insert("site".into(), empty);
9201 assert!(m.build_override("site").is_none());
9202 }
9203
9204 #[test]
9205 fn cloud_config_cross_ref_fails_on_missing_provider_named_by_an_ingress_edge() {
9206 // R845: the edge's own `use` is a provider reference like any other, so
9207 // a typo has to fail here rather than at the Cloudflare arm of apply.
9208 let tmp = tempfile::TempDir::new().unwrap();
9209 let root = tmp.path();
9210 let svc = root.join(".yah").join("services").join("dev-yah");
9211 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
9212 std::fs::write(
9213 svc.join("service.toml"),
9214 "schema_version = 1\nname = \"dev-yah\"\n[address]\nkind = \"front-door\"\ndomain = \"yah.dev\"\n",
9215 )
9216 .unwrap();
9217 std::fs::write(
9218 svc.join("mirrors/prod.toml"),
9219 "schema_version = 1\nshape = \"single-machine\"\n\n\
9220 [providers.compute]\nkind = \"static\"\nmachine = \"borrowed-01\"\n\
9221 zone = \"a.yah.dev\"\nport = 8080\n\n\
9222 [[ingress]]\nprovider = \"cloudflare-tunnel\"\nuse = \"cloudflar\"\n",
9223 )
9224 .unwrap();
9225
9226 let msg = CloudConfig::load(root).unwrap_err().to_string();
9227 assert!(
9228 msg.contains("ingress[0].use") && msg.contains("cloudflar"),
9229 "error should name the edge and the typo'd id, got: {msg}"
9230 );
9231 }
9232
9233 #[test]
9234 fn cloud_config_cross_ref_passes_on_inline_only_mirror() {
9235 // Inline `kind = "miniflare-native"` doesn't require an infra provider.
9236 let tmp = tempfile::TempDir::new().unwrap();
9237 let root = tmp.path();
9238 let svc = root.join(".yah").join("services").join("local-only");
9239 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
9240 std::fs::write(
9241 svc.join("service.toml"),
9242 "schema_version = 1\nname = \"local-only\"\n[address]\nkind = \"front-door\"\ndomain = \"local.test\"\n",
9243 )
9244 .unwrap();
9245 std::fs::write(
9246 svc.join("mirrors/local.toml"),
9247 "schema_version = 1\nshape = \"local\"\n\n[providers.static]\nkind = \"miniflare-native\"\nport = 8080\n",
9248 ).unwrap();
9249
9250 // Should load fine: no `use=` references, no providers required.
9251 let cfg = CloudConfig::load(root).unwrap();
9252 assert!(cfg.service("local-only").is_some());
9253 }
9254
9255 fn mirror(src: &str) -> MirrorConfig {
9256 toml::from_str(&format!("schema_version = 1\nshape = \"single-machine\"\n{src}"))
9257 .expect("parse mirror")
9258 }
9259
9260 #[test]
9261 fn passway_machines_reads_both_ingress_spellings_the_same_way() {
9262 // The whole reason this is derived in Rust rather than read off a field
9263 // by the UI: these two mirrors say the identical thing, and a consumer
9264 // that reaches for `ingress_machines` sees the second one as empty.
9265 let scalar = mirror("ingress = \"passway\"\ningress_machines = [\"us-east-001\"]\n");
9266 let edges = mirror(
9267 "[[ingress]]\nprovider = \"passway\"\nmachines = [\"us-east-001\"]\n",
9268 );
9269 assert_eq!(scalar.passway_machines(), Some(vec!["us-east-001".into()]));
9270 assert_eq!(scalar.passway_machines(), edges.passway_machines());
9271 }
9272
9273 #[test]
9274 fn passway_machines_skips_a_cloudflare_tunnel_edge() {
9275 // A cloudflared node publishes through Cloudflare's DNS and does not
9276 // serve `GET /domains/{d}/onboarding`, so naming it here would point
9277 // the custom-domain UI at a node that cannot answer.
9278 let cf_only =
9279 mirror("[[ingress]]\nprovider = \"cloudflare-tunnel\"\nmachines = [\"cf-01\"]\n");
9280 assert_eq!(cf_only.passway_machines(), None);
9281
9282 let mixed = mirror(
9283 "[[ingress]]\nprovider = \"cloudflare-tunnel\"\nmachines = [\"cf-01\"]\n\
9284 slots = [\"static\"]\n\n\
9285 [[ingress]]\nprovider = \"passway\"\nmachines = [\"us-east-001\"]\n\
9286 slots = [\"bundle\"]\n",
9287 );
9288 assert_eq!(mixed.passway_machines(), Some(vec!["us-east-001".into()]));
9289 }
9290
9291 #[test]
9292 fn passway_machines_separates_declared_but_unplaced_from_undeclared() {
9293 // Some(vec![]) means "a passway front door exists, but its placement
9294 // falls back to the fronted slot's and is not knowable from the mirror".
9295 // None means there is no passway front door at all. Collapsing the two
9296 // would make a co-located edge indistinguishable from no edge.
9297 assert_eq!(mirror("ingress = \"passway\"\n").passway_machines(), Some(vec![]));
9298 assert_eq!(mirror("").passway_machines(), None);
9299 assert_eq!(mirror("ingress = \"none\"\n").passway_machines(), None);
9300 }
9301
9302 #[test]
9303 fn passway_machines_is_none_for_a_declaration_that_cannot_mean_anything() {
9304 // `ingress_machines` with no `ingress` is an error `ingress_edges` names
9305 // properly; swallowing it to None here is deliberate, because this is
9306 // read while loading every service in the workspace and hard-failing
9307 // would report an unrelated mirror's shape error from the wrong place.
9308 let orphaned = mirror("ingress_machines = [\"us-east-001\"]\n");
9309 assert!(orphaned.ingress_edges().is_err());
9310 assert_eq!(orphaned.passway_machines(), None);
9311 }
9312
9313 /// Same as [`mirror`] but surfacing the parse error instead of panicking —
9314 /// R870-F26's half-written-auth cases are refused BY serde, so the message
9315 /// only exists on this side of the `expect`.
9316 fn try_mirror(src: &str) -> std::result::Result<MirrorConfig, toml::de::Error> {
9317 toml::from_str(&format!("schema_version = 1\nshape = \"single-machine\"\n{src}"))
9318 }
9319
9320 const FULL_AUTH: &str = "[ingress.auth]\n\
9321 key_secret = \"cheers/yah-camp/verify\"\n\
9322 kid = \"YOHV4Riq-g8fX4uYl8rTjQ\"\n\
9323 iss = \"yah-camp\"\n\
9324 aud = \"analytics.yah.dev\"\n\
9325 require_prefixes = [\"/\"]\n";
9326
9327 /// R870-F26 — the vocabulary itself: a complete `[ingress.auth]` table
9328 /// lands on the edge as the renderer's own `PasswayAuth`, which is what
9329 /// `apply` hands to `PasswayIngressSpec` so the push carries the five
9330 /// variables rather than stripping them.
9331 #[test]
9332 fn an_ingress_edge_carries_a_declared_auth_table_through_to_the_renderers_type() {
9333 let m = mirror(&format!(
9334 "[[ingress]]\nprovider = \"passway\"\nmachines = [\"us-east-001\"]\n{FULL_AUTH}"
9335 ));
9336 let edges = m.ingress_edges().expect("a complete auth table is accepted");
9337 let auth = edges[0].auth.as_ref().expect("the table reached the edge");
9338 assert_eq!(auth.key_secret, "cheers/yah-camp/verify");
9339 assert_eq!(auth.kid, "YOHV4Riq-g8fX4uYl8rTjQ");
9340 assert_eq!(auth.iss, "yah-camp");
9341 assert_eq!(auth.aud, "analytics.yah.dev");
9342 assert_eq!(auth.require_prefixes, vec!["/".to_string()]);
9343 }
9344
9345 /// An edge with no auth is byte-identically what it was before the field
9346 /// existed. The default path is the one this must not move.
9347 #[test]
9348 fn an_edge_that_declares_no_auth_is_unchanged() {
9349 let m = mirror("[[ingress]]\nprovider = \"passway\"\nmachines = [\"us-east-001\"]\n");
9350 assert_eq!(m.ingress_edges().unwrap()[0].auth, None);
9351 // And the scalar spelling, which cannot express auth at all.
9352 assert_eq!(
9353 mirror("ingress = \"passway\"\n").ingress_edges().unwrap()[0].auth,
9354 None
9355 );
9356 }
9357
9358 /// A HALF-WRITTEN table is refused at load, naming the field that is
9359 /// missing — never deployed as a half-configured door.
9360 ///
9361 /// This is serde's own doing, and deliberately so: all five fields are
9362 /// required on `PasswayAuth`, so there is no partial value to construct.
9363 /// The dangerous half is the one that fails QUIETLY — `kid`/`iss`/`aud`
9364 /// without `key_secret` makes passway skip the whole feature and come up
9365 /// anonymous, with nothing anywhere complaining.
9366 #[test]
9367 fn a_half_written_auth_table_is_refused_naming_the_missing_field() {
9368 let cases = [
9369 ("key_secret", "kid = \"k\"\niss = \"i\"\naud = \"a\"\nrequire_prefixes = [\"/\"]\n"),
9370 ("kid", "key_secret = \"s\"\niss = \"i\"\naud = \"a\"\nrequire_prefixes = [\"/\"]\n"),
9371 ("iss", "key_secret = \"s\"\nkid = \"k\"\naud = \"a\"\nrequire_prefixes = [\"/\"]\n"),
9372 ("aud", "key_secret = \"s\"\nkid = \"k\"\niss = \"i\"\nrequire_prefixes = [\"/\"]\n"),
9373 ("require_prefixes", "key_secret = \"s\"\nkid = \"k\"\niss = \"i\"\naud = \"a\"\n"),
9374 ];
9375 for (missing, body) in cases {
9376 let err = try_mirror(&format!(
9377 "[[ingress]]\nprovider = \"passway\"\n[ingress.auth]\n{body}"
9378 ))
9379 .expect_err("a partial auth table must not load");
9380 assert!(
9381 err.to_string().contains(missing),
9382 "the refusal must name {missing}, got: {err}"
9383 );
9384 }
9385 }
9386
9387 /// The half serde cannot catch: a field that is PRESENT and empty. Refused
9388 /// by `PasswayAuth::validate`, the same implementation
9389 /// `yah cloud ingress deploy` runs against its flags — so the two authoring
9390 /// routes cannot disagree about what counts as configured.
9391 #[test]
9392 fn a_present_but_empty_auth_field_is_refused_naming_the_toml_key() {
9393 let err = mirror(
9394 "[[ingress]]\nprovider = \"passway\"\n[ingress.auth]\n\
9395 key_secret = \"s\"\nkid = \"\"\niss = \"i\"\naud = \"a\"\nrequire_prefixes = [\"/\"]\n",
9396 )
9397 .ingress_edges()
9398 .expect_err("an empty kid is a boot panic on a remote node")
9399 .to_string();
9400 assert!(err.contains("[ingress.auth].kid"), "{err}");
9401
9402 // The quiet one: a verify key protecting nothing is authenticated and
9403 // anonymous at once. The message names the TOML key, not the flag —
9404 // sending a mirror author to look for `--require-auth` costs them the
9405 // search this validation exists to save.
9406 let err = mirror(&format!(
9407 "[[ingress]]\nprovider = \"passway\"\n{}",
9408 FULL_AUTH.replace("require_prefixes = [\"/\"]", "require_prefixes = []")
9409 ))
9410 .ingress_edges()
9411 .expect_err("a door protecting no prefix must be refused")
9412 .to_string();
9413 assert!(err.contains("[ingress.auth].require_prefixes"), "{err}");
9414 assert!(!err.contains("--require-auth"), "wrong vocabulary: {err}");
9415 }
9416
9417 /// Auth on a non-passway edge is refused rather than ignored. Only passway
9418 /// renders `PASSWAY_AUTH_*`; silently dropping it hands the operator a door
9419 /// they believe is protected and is not — the exact outcome this whole
9420 /// vocabulary exists to prevent.
9421 #[test]
9422 fn auth_on_a_cloudflare_tunnel_edge_is_refused_rather_than_ignored() {
9423 let err = mirror(&format!(
9424 "[[ingress]]\nprovider = \"cloudflare-tunnel\"\nmachines = [\"cf-01\"]\n{FULL_AUTH}"
9425 ))
9426 .ingress_edges()
9427 .expect_err("only a passway edge can render bearer auth")
9428 .to_string();
9429 assert!(err.contains("passway"), "{err}");
9430 assert!(err.contains("[ingress.auth]"), "{err}");
9431 }
9432
9433 #[test]
9434 fn cloud_config_load_derives_passway_machines_only_for_passway_envs() {
9435 let tmp = tempfile::TempDir::new().unwrap();
9436 let root = tmp.path();
9437 let svc = root.join(".yah").join("services").join("dev-yah");
9438 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
9439 std::fs::write(
9440 svc.join("service.toml"),
9441 "schema_version = 1\nname = \"dev-yah\"\n[address]\nkind = \"front-door\"\ndomain = \"yah.dev\"\n",
9442 )
9443 .unwrap();
9444 std::fs::write(
9445 svc.join("mirrors/prod.toml"),
9446 "schema_version = 1\nshape = \"single-machine\"\n\
9447 ingress = \"passway\"\ningress_machines = [\"us-east-001\", \"us-west-001\"]\n\n\
9448 [providers.static]\nkind = \"miniflare-native\"\nport = 8080\n",
9449 )
9450 .unwrap();
9451 std::fs::write(
9452 svc.join("mirrors/local.toml"),
9453 "schema_version = 1\nshape = \"local\"\n\n\
9454 [providers.static]\nkind = \"miniflare-native\"\nport = 8080\n",
9455 )
9456 .unwrap();
9457
9458 let cfg = CloudConfig::load(root).unwrap();
9459 let svc = cfg.service("dev-yah").unwrap();
9460 assert_eq!(
9461 svc.passway_machines.get("prod"),
9462 Some(&vec!["us-east-001".to_string(), "us-west-001".to_string()])
9463 );
9464 assert!(
9465 !svc.passway_machines.contains_key("local"),
9466 "an env with no front door must be absent, not empty: {:?}",
9467 svc.passway_machines
9468 );
9469 }
9470
9471 #[test]
9472 fn cloud_config_load_coexists_legacy_and_new_trees() {
9473 // Both trees present — both fields populated independently.
9474 let tmp = tempfile::TempDir::new().unwrap();
9475 let root = tmp.path();
9476 make_new_tree_with_dev_yah(root);
9477
9478 let cloud_dir = make_legacy_cloud_dir(root);
9479 std::fs::create_dir_all(cloud_dir.join("mirrors")).unwrap();
9480 std::fs::write(
9481 cloud_dir.join("mirrors/noisetable.toml"),
9482 "camp = \"noisetable\"\nregions = [\"pdx\"]\nworkloads = []\n",
9483 )
9484 .unwrap();
9485
9486 let cfg = CloudConfig::load(root).unwrap();
9487 assert_eq!(cfg.providers.len(), 3);
9488 assert!(cfg.service("dev-yah").is_some());
9489 assert_eq!(cfg.legacy_mirrors.len(), 1);
9490 assert!(cfg.legacy_mirror("noisetable").is_some());
9491 }
9492
9493 #[test]
9494 fn web_workload_round_trips() {
9495 // app/yah/web/workload.toml is parsed as a WorkloadSpec via the
9496 // workload-spec crate. The minimum-viable manifest here exercises
9497 // schema_version + kind + build fields.
9498 //
9499 // The on-disk file uses the abbreviated v1 form (kind + build); the
9500 // full WorkloadSpec is verbose, so this test asserts the new
9501 // mesofact-static abbreviated form parses as raw TOML (B3 will plumb
9502 // it through WorkloadSpec proper).
9503 // `routes` above [build] — it is a top-level field, and TOML would
9504 // scope it into that table if written below the header (R658-B1).
9505 let src = r#"
9506schema_version = 1
9507kind = "mesofact-static"
9508
9509routes = "./routes.ts"
9510
9511[build]
9512command = "bun run build"
9513out_dir = "dist"
9514"#;
9515 let v: toml::Value = toml::from_str(src).unwrap();
9516 assert_eq!(
9517 v.get("schema_version").and_then(|x| x.as_integer()),
9518 Some(1)
9519 );
9520 assert_eq!(
9521 v.get("kind").and_then(|x| x.as_str()),
9522 Some("mesofact-static")
9523 );
9524 let build = v
9525 .get("build")
9526 .and_then(|x| x.as_table())
9527 .expect("build table");
9528 assert_eq!(
9529 build.get("command").and_then(|x| x.as_str()),
9530 Some("bun run build")
9531 );
9532 assert_eq!(build.get("out_dir").and_then(|x| x.as_str()), Some("dist"));
9533 }
9534
9535 // ─── Canonical CRUD: ServiceConfig/MirrorConfig save + delete (R323-F1) ──
9536
9537 #[test]
9538 fn service_config_save_creates_canonical_toml_and_round_trips() {
9539 let tmp = tempfile::TempDir::new().unwrap();
9540 let root = tmp.path();
9541
9542 let svc = ServiceConfig {
9543 schema_version: 1,
9544 name: "dev-yah".into(),
9545 address: ServiceAddress::front_door("yah.dev"),
9546 description: None,
9547 db: DbCatalog::default(),
9548 components: vec![ServiceComponent {
9549 mount: None,
9550 id: "site".into(),
9551 kind: "mesofact-static".into(),
9552 path: "app/yah/web".into(),
9553 role: "static".into(),
9554 publishes: Some("static".into()),
9555 wave: 0,
9556 git: None,
9557 deploy: Default::default(),
9558 }],
9559 };
9560 svc.save(root).unwrap();
9561
9562 // Landed at the canonical path.
9563 let path = crate::paths::service_toml(root, "dev-yah");
9564 assert!(
9565 path.exists(),
9566 "service.toml should exist at {}",
9567 path.display()
9568 );
9569
9570 // Reloads through the full CloudConfig loader (no mirrors yet).
9571 let cfg = CloudConfig::load(root).unwrap();
9572 let loaded = cfg.service("dev-yah").expect("dev-yah service");
9573 assert_eq!(loaded.service.domain(), Some("yah.dev"));
9574 assert_eq!(loaded.service.components.len(), 1);
9575 assert_eq!(
9576 loaded.service.components[0].publishes.as_deref(),
9577 Some("static")
9578 );
9579 assert!(loaded.mirrors.is_empty());
9580 }
9581
9582 #[test]
9583 fn a_declared_health_path_survives_the_loader() {
9584 // The field is only worth having if it reaches the consumer — the
9585 // desktop front-door probe reads it off the loaded service, so a
9586 // round-trip that drops it would leave the probe on `/` with the
9587 // config still reading correctly.
9588 let tmp = tempfile::TempDir::new().unwrap();
9589 let root = tmp.path();
9590
9591 ServiceConfig {
9592 schema_version: 1,
9593 name: "api".into(),
9594 address: ServiceAddress::front_door_at("api.noisetable.com", "/api/v1/status"),
9595 description: None,
9596 components: vec![],
9597 db: DbCatalog::default(),
9598 }
9599 .save(root)
9600 .unwrap();
9601
9602 let cfg = CloudConfig::load(root).unwrap();
9603 assert_eq!(
9604 cfg.service("api").unwrap().service.health_path(),
9605 Some("/api/v1/status")
9606 );
9607 }
9608
9609 /// R926. Same reasoning as the health_path round-trip above, and the
9610 /// same failure mode: `description` is `skip_serializing_if =
9611 /// "Option::is_none"`, so a serde attribute that silently dropped it
9612 /// would leave every service listed without one while the file on
9613 /// disk still read correctly.
9614 #[test]
9615 fn a_declared_description_survives_the_loader() {
9616 let tmp = tempfile::TempDir::new().unwrap();
9617 let root = tmp.path();
9618
9619 ServiceConfig {
9620 schema_version: 1,
9621 name: "api".into(),
9622 address: ServiceAddress::front_door("api.noisetable.com"),
9623 description: Some("Account and RPC origin for the noisetable app.".into()),
9624 components: vec![],
9625 db: DbCatalog::default(),
9626 }
9627 .save(root)
9628 .unwrap();
9629
9630 let cfg = CloudConfig::load(root).unwrap();
9631 assert_eq!(
9632 cfg.service("api").unwrap().service.description.as_deref(),
9633 Some("Account and RPC origin for the noisetable app.")
9634 );
9635 }
9636
9637 /// The nine services that predate R926 carry no `description`, so the
9638 /// field has to be genuinely optional rather than optional-with-a-
9639 /// default — a required field here would fail the whole camp's config
9640 /// load, not just the service missing it.
9641 #[test]
9642 fn a_service_without_a_description_still_loads() {
9643 let tmp = tempfile::TempDir::new().unwrap();
9644 let root = tmp.path();
9645 let dir = root.join(".yah/services/api");
9646 std::fs::create_dir_all(&dir).unwrap();
9647 std::fs::write(
9648 dir.join("service.toml"),
9649 "schema_version = 1\nname = \"api\"\n[address]\nkind = \"front-door\"\ndomain = \"api.noisetable.com\"\n",
9650 )
9651 .unwrap();
9652
9653 let cfg = CloudConfig::load(root).unwrap();
9654 assert_eq!(cfg.service("api").unwrap().service.description, None);
9655 }
9656
9657 /// R926-F1. The push relay has no HTTP surface at all: it binds an
9658 /// iroh endpoint and serves one ALPN, so a `domain` for it could only
9659 /// ever be a placeholder, and a placeholder cannot be probed. This is
9660 /// the shape that replaced that dead end.
9661 #[test]
9662 fn a_node_addressed_service_loads_and_reports_no_domain() {
9663 let tmp = tempfile::TempDir::new().unwrap();
9664 let root = tmp.path();
9665 let dir = root.join(".yah/services/push-relay");
9666 std::fs::create_dir_all(&dir).unwrap();
9667 let node_id = "a".repeat(64);
9668 std::fs::write(
9669 dir.join("service.toml"),
9670 format!(
9671 "schema_version = 1\nname = \"push-relay\"\n\
9672 description = \"Mobile push fanout.\"\n\
9673 [address]\nkind = \"node\"\n\
9674 node_id = \"{node_id}\"\nalpn = \"yah/push-relay/1\"\n"
9675 ),
9676 )
9677 .unwrap();
9678
9679 let cfg = CloudConfig::load(root).unwrap();
9680 let svc = &cfg.service("push-relay").unwrap().service;
9681 assert_eq!(svc.domain(), None, "a node has no domain, not a fake one");
9682 assert_eq!(svc.health_path(), None);
9683 assert_eq!(
9684 svc.address,
9685 ServiceAddress::node(&node_id, "yah/push-relay/1")
9686 );
9687 // The label is what a status table prints. It must never be blank
9688 // and must never be a domain-shaped lie.
9689 assert_eq!(svc.address.label(), "node:aaaaaaaa/yah/push-relay/1");
9690 }
9691
9692 /// A caller that genuinely needs a domain gets a sentence naming the
9693 /// service and what it wanted one for — not an `unwrap` panic and not
9694 /// an empty string silently probing the wrong origin.
9695 #[test]
9696 fn require_domain_on_a_node_service_names_the_service_and_the_caller() {
9697 let svc = ServiceConfig {
9698 schema_version: 1,
9699 name: "push-relay".into(),
9700 description: None,
9701 address: ServiceAddress::node("b".repeat(64), "yah/push-relay/1"),
9702 components: vec![],
9703 db: DbCatalog::default(),
9704 };
9705 let err = svc
9706 .require_domain("a DNS record")
9707 .expect_err("a node service has no domain");
9708 let msg = err.to_string();
9709 assert!(msg.contains("push-relay"), "{msg}");
9710 assert!(msg.contains("a DNS record"), "{msg}");
9711 assert!(msg.contains("node:bbbbbbbb"), "{msg}");
9712 }
9713
9714 /// A truncated paste is the realistic way a `node_id` goes wrong, and
9715 /// it would otherwise surface as a dial timeout naming neither the
9716 /// file nor the field.
9717 #[test]
9718 fn a_truncated_node_id_is_refused_at_load() {
9719 let tmp = tempfile::TempDir::new().unwrap();
9720 let root = tmp.path();
9721 let dir = root.join(".yah/services/push-relay");
9722 std::fs::create_dir_all(&dir).unwrap();
9723 std::fs::write(
9724 dir.join("service.toml"),
9725 "schema_version = 1\nname = \"push-relay\"\n\
9726 [address]\nkind = \"node\"\n\
9727 node_id = \"abc123\"\nalpn = \"yah/push-relay/1\"\n",
9728 )
9729 .unwrap();
9730
9731 let err = CloudConfig::load(root).expect_err("a short node_id must not load");
9732 let msg = format!("{err:#}");
9733 assert!(msg.contains("node_id"), "{msg}");
9734 assert!(msg.contains("64 hex"), "{msg}");
9735 }
9736
9737 /// A NodeId with no ALPN names a process, not a service — there would
9738 /// be nothing to dial.
9739 #[test]
9740 fn a_node_address_without_an_alpn_is_refused_at_load() {
9741 let tmp = tempfile::TempDir::new().unwrap();
9742 let root = tmp.path();
9743 let dir = root.join(".yah/services/push-relay");
9744 std::fs::create_dir_all(&dir).unwrap();
9745 std::fs::write(
9746 dir.join("service.toml"),
9747 format!(
9748 "schema_version = 1\nname = \"push-relay\"\n\
9749 [address]\nkind = \"node\"\n\
9750 node_id = \"{}\"\nalpn = \"\"\n",
9751 "c".repeat(64)
9752 ),
9753 )
9754 .unwrap();
9755
9756 let err = CloudConfig::load(root).expect_err("an empty alpn must not load");
9757 assert!(format!("{err:#}").contains("alpn"), "{err:#}");
9758 }
9759
9760 /// `save` writes TOML that `load` accepts, for both address kinds. The
9761 /// ordering trap is real: an `[address]` table emitted before a scalar
9762 /// field would produce a file `toml` cannot parse back.
9763 #[test]
9764 fn both_address_kinds_survive_a_save_load_round_trip() {
9765 for address in [
9766 ServiceAddress::front_door_at("api.example", "/healthz"),
9767 ServiceAddress::node("d".repeat(64), "yah/push-relay/1"),
9768 ] {
9769 let tmp = tempfile::TempDir::new().unwrap();
9770 let root = tmp.path();
9771 let svc = ServiceConfig {
9772 schema_version: 1,
9773 name: "round-trip".into(),
9774 description: Some("described".into()),
9775 address: address.clone(),
9776 components: vec![],
9777 db: DbCatalog::default(),
9778 };
9779 svc.save(root).unwrap();
9780 let cfg = CloudConfig::load(root).unwrap();
9781 let back = &cfg.service("round-trip").unwrap().service;
9782 assert_eq!(back.address, address);
9783 assert_eq!(back.description.as_deref(), Some("described"));
9784 }
9785 }
9786
9787 /// R926. `.yah/services/headscale/service.toml` is the first service in
9788 /// the tree with NO components and NO `mirrors/` directory — it is
9789 /// registered to be described and probed, not deployed (the mesh leader
9790 /// places the appliance, not `yah cloud apply`). That shape has to load,
9791 /// because the alternative is that adding an observability-only service
9792 /// fails the config load for the whole camp.
9793 ///
9794 /// Also pins the query string: the coordination probe is `/key?v=138`,
9795 /// and a validator that got stricter about what follows the leading `/`
9796 /// would silently un-register the one service whose false-green took the
9797 /// mesh down for 37 hours.
9798 #[test]
9799 fn an_observability_only_service_loads_without_components_or_mirrors() {
9800 let tmp = tempfile::TempDir::new().unwrap();
9801 let root = tmp.path();
9802 let dir = root.join(".yah/services/headscale");
9803 std::fs::create_dir_all(&dir).unwrap();
9804 std::fs::write(
9805 dir.join("service.toml"),
9806 "schema_version = 1\n\
9807 name = \"headscale\"\n\
9808 description = \"Mesh coordination server.\"\n\
9809 [address]\n\
9810 kind = \"front-door\"\n\
9811 domain = \"cloud.mesh.yah.dev\"\n\
9812 health_path = \"/key?v=138\"\n",
9813 )
9814 .unwrap();
9815
9816 let cfg = CloudConfig::load(root).unwrap();
9817 let svc = cfg.service("headscale").unwrap();
9818 assert_eq!(svc.service.health_path(), Some("/key?v=138"));
9819 assert_eq!(
9820 svc.service.description.as_deref(),
9821 Some("Mesh coordination server.")
9822 );
9823 assert!(svc.service.components.is_empty());
9824 assert!(svc.mirrors.is_empty());
9825 }
9826
9827 #[test]
9828 fn a_relative_health_path_is_refused_at_load() {
9829 // Not a style rule. A relative path joins onto the origin differently
9830 // depending on which URL builder gets it, so the probe would ask a
9831 // question the file does not read as asking — and it would paint a
9832 // confident dot either way.
9833 let tmp = tempfile::TempDir::new().unwrap();
9834 let root = tmp.path();
9835
9836 ServiceConfig {
9837 schema_version: 1,
9838 name: "api".into(),
9839 address: ServiceAddress::front_door_at("api.noisetable.com", "api/v1/status"),
9840 description: None,
9841 components: vec![],
9842 db: DbCatalog::default(),
9843 }
9844 .save(root)
9845 .unwrap();
9846
9847 let err = CloudConfig::load(root).expect_err("a relative health_path must not load");
9848 let msg = format!("{err:#}");
9849 assert!(
9850 msg.contains("health_path") && msg.contains("/api/v1/status"),
9851 "the error must name the field and the fix, got: {msg}"
9852 );
9853 }
9854
9855 #[test]
9856 fn service_config_save_overwrites_in_place() {
9857 let tmp = tempfile::TempDir::new().unwrap();
9858 let root = tmp.path();
9859
9860 let mut svc = ServiceConfig {
9861 schema_version: 1,
9862 name: "dev-yah".into(),
9863 address: ServiceAddress::front_door("yah.dev"),
9864 description: None,
9865 components: vec![],
9866 db: DbCatalog::default(),
9867 };
9868 svc.save(root).unwrap();
9869 svc.address = ServiceAddress::front_door("yah.example");
9870 svc.save(root).unwrap();
9871
9872 let cfg = CloudConfig::load(root).unwrap();
9873 assert_eq!(
9874 cfg.service("dev-yah").unwrap().service.domain(),
9875 Some("yah.example")
9876 );
9877 }
9878
9879 #[test]
9880 fn mirror_config_save_round_trips_reference_and_inline_slots() {
9881 let tmp = tempfile::TempDir::new().unwrap();
9882 let root = tmp.path();
9883
9884 // A service must exist so the loader walks the mirrors/ dir.
9885 ServiceConfig {
9886 schema_version: 1,
9887 name: "dev-yah".into(),
9888 address: ServiceAddress::front_door("yah.dev"),
9889 description: None,
9890 components: vec![],
9891 db: DbCatalog::default(),
9892 }
9893 .save(root)
9894 .unwrap();
9895
9896 // The cloudflare provider the reference slot points at must resolve,
9897 // or CloudConfig::load's cross-ref check rejects the tree.
9898 let providers = crate::paths::providers_dir(root);
9899 std::fs::create_dir_all(&providers).unwrap();
9900 std::fs::write(
9901 providers.join("cloudflare.toml"),
9902 "schema_version = 1\nid = \"cloudflare\"\nkind = \"cloudflare\"\n",
9903 )
9904 .unwrap();
9905
9906 let mut providers_map = BTreeMap::new();
9907 providers_map.insert(
9908 "static".to_string(),
9909 MirrorProviderSlot::Reference {
9910 provider_id: "cloudflare".into(),
9911 fields: {
9912 let mut f = BTreeMap::new();
9913 f.insert("bucket".to_string(), toml::Value::String("yah-dev".into()));
9914 f
9915 },
9916 },
9917 );
9918 providers_map.insert(
9919 "compute".to_string(),
9920 MirrorProviderSlot::Inline {
9921 kind: Provider::MiniflareNative,
9922 fields: {
9923 let mut f = BTreeMap::new();
9924 f.insert("port".to_string(), toml::Value::Integer(4321));
9925 f
9926 },
9927 },
9928 );
9929 let mirror = MirrorConfig {
9930 schema_version: 1,
9931 shape: MirrorShape::SingleMachine,
9932 providers: providers_map,
9933 ingress: Default::default(),
9934 ingress_machines: Vec::new(),
9935 drivers: Default::default(),
9936 asset_aliases: Default::default(),
9937 build: Default::default(),
9938 };
9939 // Save with canonical name; legacy "cloud" is normalised to "prod" on load.
9940 mirror.save(root, "dev-yah", "prod").unwrap();
9941
9942 let path = crate::paths::service_mirror_toml(root, "dev-yah", "prod");
9943 assert!(
9944 path.exists(),
9945 "mirror toml should exist at {}",
9946 path.display()
9947 );
9948
9949 let cfg = CloudConfig::load(root).unwrap();
9950 let loaded = &cfg.service("dev-yah").unwrap().mirrors["prod"];
9951 assert_eq!(loaded.shape, MirrorShape::SingleMachine);
9952 assert_eq!(loaded.providers["static"].provider_id(), Some("cloudflare"));
9953 assert_eq!(
9954 loaded.providers["compute"].inline_kind(),
9955 Some(Provider::MiniflareNative)
9956 );
9957 }
9958
9959 #[test]
9960 fn service_delete_removes_dir_and_mirrors() {
9961 let tmp = tempfile::TempDir::new().unwrap();
9962 let root = tmp.path();
9963
9964 let svc = ServiceConfig {
9965 schema_version: 1,
9966 name: "dev-yah".into(),
9967 address: ServiceAddress::front_door("yah.dev"),
9968 description: None,
9969 components: vec![],
9970 db: DbCatalog::default(),
9971 };
9972 svc.save(root).unwrap();
9973 MirrorConfig {
9974 schema_version: 1,
9975 shape: MirrorShape::Local,
9976 providers: BTreeMap::new(),
9977 ingress: Default::default(),
9978 ingress_machines: Vec::new(),
9979 drivers: Default::default(),
9980 asset_aliases: Default::default(),
9981 build: Default::default(),
9982 }
9983 .save(root, "dev-yah", "local")
9984 .unwrap();
9985
9986 assert!(
9987 ServiceConfig::delete(root, "dev-yah").unwrap(),
9988 "first delete reports true"
9989 );
9990 assert!(!crate::paths::service_dir(root, "dev-yah").exists());
9991 // Idempotent: deleting again is a no-op that reports false.
9992 assert!(!ServiceConfig::delete(root, "dev-yah").unwrap());
9993
9994 let cfg = CloudConfig::load(root).unwrap();
9995 assert!(cfg.service("dev-yah").is_none());
9996 }
9997
9998 #[test]
9999 fn mirror_delete_leaves_other_mirrors_and_service_intact() {
10000 let tmp = tempfile::TempDir::new().unwrap();
10001 let root = tmp.path();
10002
10003 ServiceConfig {
10004 schema_version: 1,
10005 name: "dev-yah".into(),
10006 address: ServiceAddress::front_door("yah.dev"),
10007 description: None,
10008 components: vec![],
10009 db: DbCatalog::default(),
10010 }
10011 .save(root)
10012 .unwrap();
10013 for env in ["prod", "local"] {
10014 MirrorConfig {
10015 schema_version: 1,
10016 shape: MirrorShape::Local,
10017 providers: BTreeMap::new(),
10018 ingress: Default::default(),
10019 ingress_machines: Vec::new(),
10020 drivers: Default::default(),
10021 asset_aliases: Default::default(),
10022 build: Default::default(),
10023 }
10024 .save(root, "dev-yah", env)
10025 .unwrap();
10026 }
10027
10028 assert!(MirrorConfig::delete(root, "dev-yah", "prod").unwrap());
10029 assert!(!MirrorConfig::delete(root, "dev-yah", "prod").unwrap());
10030
10031 let cfg = CloudConfig::load(root).unwrap();
10032 let svc = cfg
10033 .service("dev-yah")
10034 .expect("service survives mirror delete");
10035 // "prod" is canonical and deletes directly; "local" normalises to "dev" on load.
10036 assert!(!svc.mirrors.contains_key("prod"));
10037 assert!(svc.mirrors.contains_key("dev"));
10038 }
10039
10040 // ─── DomainConfig (R347-F2) ────────────────────────────────────────────
10041
10042 fn write_marketing_service(root: &Path) {
10043 let svc = ServiceConfig {
10044 schema_version: 1,
10045 name: "yah-marketing".into(),
10046 address: ServiceAddress::front_door("yah.dev"),
10047 description: None,
10048 db: DbCatalog::default(),
10049 components: vec![ServiceComponent {
10050 mount: None,
10051 id: "site".into(),
10052 kind: "mesofact-static".into(),
10053 path: "app/yah/web".into(),
10054 role: "static".into(),
10055 publishes: None,
10056 wave: 0,
10057 git: None,
10058 deploy: Default::default(),
10059 }],
10060 };
10061 svc.save(root).unwrap();
10062 }
10063
10064 #[test]
10065 fn round_trip_domain_with_each_route_mode() {
10066 let dom = DomainConfig {
10067 schema_version: 1,
10068 name: "yah-dev".into(),
10069 domain: "yah.dev".into(),
10070 front_door: FrontDoor::Worker,
10071 cdn_bucket: "yah-dev".into(),
10072 worker_bundle_path: Some(".yah/workers/yah-dev/".into()),
10073 routes: vec![
10074 DomainRoute {
10075 headers: Default::default(),
10076 path: "/".into(),
10077 mode: RouteMode::Static {
10078 component: "yah-marketing/site".into(),
10079 },
10080 },
10081 DomainRoute {
10082 headers: Default::default(),
10083 path: "/dashboard/api/*".into(),
10084 mode: RouteMode::Backend {
10085 component: "yah-dashboard/api".into(),
10086 origin: "https://api.dashboard.yah.dev".into(),
10087 origin_path: None,
10088 },
10089 },
10090 DomainRoute {
10091 headers: Default::default(),
10092 path: "/old".into(),
10093 mode: RouteMode::Redirect {
10094 target: "https://yah.dev/blog".into(),
10095 status: 308,
10096 },
10097 },
10098 ],
10099 };
10100 let s = toml::to_string(&dom).unwrap();
10101 let back: DomainConfig = toml::from_str(&s).unwrap();
10102 assert_eq!(back.name, "yah-dev");
10103 assert_eq!(back.routes.len(), 3);
10104 assert!(matches!(back.routes[0].mode, RouteMode::Static { .. }));
10105 assert!(matches!(back.routes[1].mode, RouteMode::Backend { .. }));
10106 assert!(matches!(back.routes[2].mode, RouteMode::Redirect { .. }));
10107 }
10108
10109 /// A one-route `cdn.noisetable.com`-shaped manifest whose static route body
10110 /// is `body`.
10111 fn static_route_manifest(body: &str) -> String {
10112 format!(
10113 "schema_version = 1\nname = \"cdn-noisetable-com\"\ndomain = \"cdn.noisetable.com\"\n\
10114 front_door = \"worker\"\ncdn_bucket = \"noisetable-marketing\"\n\n\
10115 [[routes]]\npath = \"/engine/*\"\nmode = \"static\"\n{body}\n"
10116 )
10117 }
10118
10119 /// R560-F13 — a static route names a `component` OR a `bucket`, never both
10120 /// and never neither, and a bucket has to be a real R2 bucket name.
10121 #[test]
10122 fn a_static_route_names_exactly_one_of_component_or_bucket() {
10123 let dom: DomainConfig =
10124 toml::from_str(&static_route_manifest("bucket = \"noisetable-releases\"")).unwrap();
10125 assert!(matches!(
10126 &dom.routes[0].mode,
10127 RouteMode::StaticBucket { bucket } if bucket == "noisetable-releases"
10128 ));
10129 dom.validate_front_door().unwrap();
10130
10131 let dom: DomainConfig =
10132 toml::from_str(&static_route_manifest("component = \"svc/site\"")).unwrap();
10133 assert!(matches!(
10134 &dom.routes[0].mode,
10135 RouteMode::Static { component } if component == "svc/site"
10136 ));
10137
10138 for (body, needle) in [
10139 (
10140 "component = \"svc/site\"\nbucket = \"noisetable-releases\"",
10141 "both",
10142 ),
10143 ("", "neither"),
10144 ("bucket = \"Noisetable_Releases\"", "not an R2 bucket name"),
10145 ] {
10146 let err = toml::from_str::<DomainConfig>(&static_route_manifest(body))
10147 .unwrap_err()
10148 .to_string();
10149 assert!(err.contains(needle), "{body:?}: {err}");
10150 }
10151 }
10152
10153 /// The Rust variant is `StaticBucket`; the manifest spelling stays
10154 /// `mode = "static"` + `bucket`, through a save as well as a load.
10155 #[test]
10156 fn a_bucket_route_round_trips_as_mode_static_with_a_bucket_key() {
10157 let dom: DomainConfig =
10158 toml::from_str(&static_route_manifest("bucket = \"noisetable-releases\"")).unwrap();
10159 let s = toml::to_string(&dom).unwrap();
10160 assert!(s.contains("mode = \"static\""), "{s}");
10161 assert!(s.contains("bucket = \"noisetable-releases\""), "{s}");
10162 assert!(!s.contains("component"), "{s}");
10163 let back: DomainConfig = toml::from_str(&s).unwrap();
10164 assert!(matches!(back.routes[0].mode, RouteMode::StaticBucket { .. }));
10165 }
10166
10167 /// Passway has no R2 read path, so a bucket route on it is refused at load
10168 /// naming the route, not compiled into an entry that door cannot serve.
10169 #[test]
10170 fn passway_refuses_a_bucket_route() {
10171 let mut dom: DomainConfig =
10172 toml::from_str(&static_route_manifest("bucket = \"noisetable-releases\"")).unwrap();
10173 dom.front_door = FrontDoor::Passway;
10174 let err = dom.validate_front_door().unwrap_err().to_string();
10175 assert!(err.contains("passway") && err.contains("/engine/*"), "{err}");
10176 }
10177
10178 #[test]
10179 fn redirect_status_defaults_to_308() {
10180 let src = r#"
10181schema_version = 1
10182name = "yah-dev"
10183domain = "yah.dev"
10184front_door = "worker"
10185cdn_bucket = "yah-dev"
10186
10187[[routes]]
10188path = "/old"
10189mode = "redirect"
10190target = "https://yah.dev/blog"
10191"#;
10192 let dom: DomainConfig = toml::from_str(src).unwrap();
10193 let RouteMode::Redirect { status, .. } = &dom.routes[0].mode else {
10194 panic!("expected redirect");
10195 };
10196 assert_eq!(*status, 308);
10197 }
10198
10199 #[test]
10200 fn missing_domains_dir_is_empty() {
10201 let tmp = tempfile::TempDir::new().unwrap();
10202 // R844-B7: `.yah/` must exist or this is a wrong-root error rather
10203 // than an empty tree. The absent directory under test is `domains/`.
10204 std::fs::create_dir_all(tmp.path().join(".yah")).unwrap();
10205 let cfg = CloudConfig::load(tmp.path()).unwrap();
10206 assert!(cfg.domains.is_empty());
10207 }
10208
10209 #[test]
10210 fn save_reload_roundtrip() {
10211 let tmp = tempfile::TempDir::new().unwrap();
10212 let root = tmp.path();
10213 write_marketing_service(root);
10214
10215 let dom = DomainConfig {
10216 schema_version: 1,
10217 name: "yah-dev".into(),
10218 domain: "yah.dev".into(),
10219 front_door: FrontDoor::Worker,
10220 cdn_bucket: "yah-dev".into(),
10221 worker_bundle_path: None,
10222 routes: vec![DomainRoute {
10223 headers: Default::default(),
10224 path: "/".into(),
10225 mode: RouteMode::Static {
10226 component: "yah-marketing/site".into(),
10227 },
10228 }],
10229 };
10230 dom.save(root).unwrap();
10231
10232 let cfg = CloudConfig::load(root).unwrap();
10233 let loaded = cfg.domain("yah-dev").expect("yah-dev domain");
10234 assert_eq!(loaded.domain, "yah.dev");
10235 assert_eq!(loaded.routes.len(), 1);
10236 }
10237
10238 #[test]
10239 fn delete_returns_false_when_absent() {
10240 let tmp = tempfile::TempDir::new().unwrap();
10241 assert!(!DomainConfig::delete(tmp.path(), "no-such-domain").unwrap());
10242 }
10243
10244 #[test]
10245 fn delete_returns_true_first_time() {
10246 let tmp = tempfile::TempDir::new().unwrap();
10247 let root = tmp.path();
10248 let dom = DomainConfig {
10249 schema_version: 1,
10250 name: "yah-dev".into(),
10251 domain: "yah.dev".into(),
10252 front_door: FrontDoor::BucketDirect,
10253 cdn_bucket: "yah-dev".into(),
10254 worker_bundle_path: None,
10255 routes: vec![],
10256 };
10257 dom.save(root).unwrap();
10258 assert!(DomainConfig::delete(root, "yah-dev").unwrap());
10259 assert!(!DomainConfig::delete(root, "yah-dev").unwrap());
10260 }
10261
10262 // ---- R594-F12: front-door discriminator ------------------------------
10263
10264 /// Write a raw domain manifest so the tests exercise the deserialize +
10265 /// validate path, not a hand-built struct that skipped serde.
10266 fn write_domain_toml(root: &Path, stem: &str, body: &str) {
10267 let dir = root.join(".yah").join("domains");
10268 std::fs::create_dir_all(&dir).unwrap();
10269 std::fs::write(dir.join(format!("{stem}.toml")), body).unwrap();
10270 }
10271
10272 #[test]
10273 fn front_door_is_required() {
10274 let tmp = tempfile::TempDir::new().unwrap();
10275 let root = tmp.path();
10276 write_marketing_service(root);
10277 write_domain_toml(
10278 root,
10279 "yah-dev",
10280 r#"
10281schema_version = 1
10282name = "yah-dev"
10283domain = "yah.dev"
10284cdn_bucket = "yah-dev"
10285[[routes]]
10286path = "/*"
10287mode = "static"
10288component = "yah-marketing/site"
10289"#,
10290 );
10291 let err = CloudConfig::load(root).unwrap_err().to_string();
10292 // serde's own missing-field message; the point is that omitting the
10293 // discriminator is not a silently-defaulted state.
10294 assert!(err.contains("yah-dev.toml"), "{err}");
10295 }
10296
10297 #[test]
10298 fn bucket_direct_with_routes_is_rejected() {
10299 let tmp = tempfile::TempDir::new().unwrap();
10300 let root = tmp.path();
10301 write_marketing_service(root);
10302 write_domain_toml(
10303 root,
10304 "cdn-yah-dev",
10305 r#"
10306schema_version = 1
10307name = "cdn-yah-dev"
10308domain = "cdn.yah.dev"
10309front_door = "bucket-direct"
10310cdn_bucket = "yah-dev"
10311[[routes]]
10312path = "/docs/*"
10313mode = "static"
10314component = "yah-marketing/site"
10315"#,
10316 );
10317 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10318 assert!(err.contains("front_door"), "{err}");
10319 assert!(err.contains("/docs/*"), "{err}");
10320 }
10321
10322 #[test]
10323 fn bucket_direct_with_worker_bundle_path_is_rejected() {
10324 let tmp = tempfile::TempDir::new().unwrap();
10325 let root = tmp.path();
10326 write_domain_toml(
10327 root,
10328 "cdn-yah-dev",
10329 r#"
10330schema_version = 1
10331name = "cdn-yah-dev"
10332domain = "cdn.yah.dev"
10333front_door = "bucket-direct"
10334cdn_bucket = "yah-dev"
10335worker_bundle_path = ".yah/workers/cdn-yah-dev/"
10336"#,
10337 );
10338 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10339 assert!(err.contains("worker_bundle_path"), "{err}");
10340 }
10341
10342 // ── R746: per-route response headers + component mounts ──────────────────
10343
10344 /// A two-component service: `site` at the root, `app` mounted at `/app`
10345 /// with isolation headers on its route. This is the noisetable.com shape
10346 /// the primitive was built for.
10347 fn write_two_component_service(root: &Path) {
10348 let svc = ServiceConfig {
10349 schema_version: 1,
10350 name: "yah-marketing".into(),
10351 address: ServiceAddress::front_door("yah.dev"),
10352 description: None,
10353 db: DbCatalog::default(),
10354 components: vec![
10355 ServiceComponent {
10356 mount: None,
10357 id: "site".into(),
10358 kind: "mesofact-static".into(),
10359 path: "app/yah/web".into(),
10360 role: "static".into(),
10361 publishes: None,
10362 wave: 0,
10363 git: None,
10364 deploy: Default::default(),
10365 },
10366 ServiceComponent {
10367 mount: Some("/app".into()),
10368 id: "app".into(),
10369 kind: "mesofact-static".into(),
10370 path: "app/browser".into(),
10371 role: "static".into(),
10372 publishes: None,
10373 wave: 0,
10374 git: None,
10375 deploy: Default::default(),
10376 },
10377 ],
10378 };
10379 svc.save(root).unwrap();
10380 }
10381
10382 const MOUNTED_DOMAIN: &str = r#"
10383schema_version = 1
10384name = "yah-dev"
10385domain = "yah.dev"
10386front_door = "worker"
10387cdn_bucket = "yah-dev"
10388
10389[[routes]]
10390path = "/app/*"
10391mode = "static"
10392component = "yah-marketing/app"
10393headers = { "Cross-Origin-Opener-Policy" = "same-origin", "Cross-Origin-Embedder-Policy" = "require-corp" }
10394
10395[[routes]]
10396path = "/*"
10397mode = "static"
10398component = "yah-marketing/site"
10399"#;
10400
10401 #[test]
10402 fn a_mounted_component_routed_at_its_mount_loads() {
10403 let tmp = tempfile::TempDir::new().unwrap();
10404 let root = tmp.path();
10405 write_two_component_service(root);
10406 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
10407 let cfg = CloudConfig::load(root).unwrap();
10408 let dom = cfg.domain("yah-dev").unwrap();
10409 assert_eq!(dom.routes.len(), 2);
10410 assert_eq!(
10411 dom.routes[0].headers.get("Cross-Origin-Opener-Policy").map(String::as_str),
10412 Some("same-origin")
10413 );
10414 assert!(dom.routes[1].headers.is_empty());
10415 }
10416
10417 /// The header table reaches the Worker in MANIFEST order with headerless
10418 /// routes dropped. Order is the whole contract — the front door applies the
10419 /// first match, so `/app/*` before `/*` is what isolates the app without
10420 /// isolating the marketing site.
10421 #[test]
10422 fn route_headers_json_preserves_order_and_drops_headerless_routes() {
10423 let tmp = tempfile::TempDir::new().unwrap();
10424 let root = tmp.path();
10425 write_two_component_service(root);
10426 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
10427 let cfg = CloudConfig::load(root).unwrap();
10428 let json = cfg.domain("yah-dev").unwrap().route_headers_json();
10429
10430 let parsed: serde_json::Value = serde_json::from_str(&json).unwrap();
10431 let rules = parsed.as_array().unwrap();
10432 assert_eq!(rules.len(), 1, "the headerless catch-all is dropped: {json}");
10433 assert_eq!(rules[0]["path"], "/app/*");
10434 assert_eq!(rules[0]["headers"]["Cross-Origin-Embedder-Policy"], "require-corp");
10435 }
10436
10437 #[test]
10438 fn route_headers_json_is_an_empty_array_when_nothing_declares_headers() {
10439 let tmp = tempfile::TempDir::new().unwrap();
10440 let root = tmp.path();
10441 write_marketing_service(root);
10442 write_domain_toml(
10443 root,
10444 "yah-dev",
10445 r#"
10446schema_version = 1
10447name = "yah-dev"
10448domain = "yah.dev"
10449front_door = "worker"
10450cdn_bucket = "yah-dev"
10451
10452[[routes]]
10453path = "/*"
10454mode = "static"
10455component = "yah-marketing/site"
10456"#,
10457 );
10458 let cfg = CloudConfig::load(root).unwrap();
10459 assert_eq!(cfg.domain("yah-dev").unwrap().route_headers_json(), "[]");
10460 }
10461
10462 /// The reconciler's own entry point: given a workspace root and a service
10463 /// name, produce the binding value. `"[]"` when nothing routes the service.
10464 #[test]
10465 fn route_headers_for_service_reads_the_workspace_domains() {
10466 let tmp = tempfile::TempDir::new().unwrap();
10467 let root = tmp.path();
10468 write_two_component_service(root);
10469 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
10470 assert!(route_headers_for_service(root, "yah-marketing")
10471 .unwrap()
10472 .contains("require-corp"));
10473 assert_eq!(route_headers_for_service(root, "some-other-svc").unwrap(), "[]");
10474 }
10475
10476 // ---- R749-T5: a broken table fails the DEPLOY, not the edge -----------
10477
10478 /// The manifest's `headers` map is hand-written TOML, so a header name with
10479 /// spaces in it is one keystroke away — and it survives serialization into
10480 /// a structurally-valid table that neither front door can apply. Fail at
10481 /// load, naming the domain, the route and the header, instead of shipping a
10482 /// binding the Worker throws on and an origin that refuses to boot.
10483 #[test]
10484 fn a_route_header_name_that_is_not_a_header_name_fails_the_load() {
10485 let tmp = tempfile::TempDir::new().unwrap();
10486 let root = tmp.path();
10487 write_marketing_service(root);
10488 write_domain_toml(
10489 root,
10490 "yah-dev",
10491 r#"
10492schema_version = 1
10493name = "yah-dev"
10494domain = "yah.dev"
10495front_door = "worker"
10496cdn_bucket = "yah-dev"
10497
10498[[routes]]
10499path = "/*"
10500mode = "static"
10501component = "yah-marketing/site"
10502headers = { "Cross Origin Opener Policy" = "same-origin" }
10503"#,
10504 );
10505 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10506 assert!(err.contains("yah-dev"), "{err}");
10507 assert!(err.contains("/*"), "{err}");
10508 assert!(err.contains("Cross Origin Opener Policy"), "{err}");
10509 assert!(err.contains("not a valid HTTP header name"), "{err}");
10510 }
10511
10512 /// A newline in a value is header injection if it ever reached the wire, so
10513 /// both doors reject it and so does this.
10514 #[test]
10515 fn a_route_header_value_that_is_not_a_header_value_fails_the_load() {
10516 let tmp = tempfile::TempDir::new().unwrap();
10517 let root = tmp.path();
10518 write_marketing_service(root);
10519 write_domain_toml(
10520 root,
10521 "yah-dev",
10522 r#"
10523schema_version = 1
10524name = "yah-dev"
10525domain = "yah.dev"
10526front_door = "worker"
10527cdn_bucket = "yah-dev"
10528
10529[[routes]]
10530path = "/*"
10531mode = "static"
10532component = "yah-marketing/site"
10533headers = { "X-Frame-Options" = "DENY\nSet-Cookie: pwned=1" }
10534"#,
10535 );
10536 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10537 assert!(err.contains("X-Frame-Options"), "{err}");
10538 assert!(err.contains("not a valid HTTP header value"), "{err}");
10539 }
10540
10541 /// The invariant this gate exists to hold: everything `route_headers_json`
10542 /// emits is applicable. A headerless route contributes no rule, so its path
10543 /// is not the table's business — only rules that ship are checked.
10544 #[test]
10545 fn a_headerless_route_is_not_subject_to_the_route_header_gate() {
10546 let tmp = tempfile::TempDir::new().unwrap();
10547 let root = tmp.path();
10548 write_two_component_service(root);
10549 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
10550 let cfg = CloudConfig::load(root).unwrap();
10551 cfg.domain("yah-dev")
10552 .unwrap()
10553 .validate_route_headers()
10554 .unwrap();
10555 }
10556
10557 /// A `bucket-direct` domain has no front door to set headers on, so it must
10558 /// not be picked up as a service's header source.
10559 #[test]
10560 fn route_headers_ignores_domains_that_are_not_route_driven() {
10561 let doms: BTreeMap<String, DomainConfig> = [(
10562 "cdn".to_string(),
10563 DomainConfig {
10564 schema_version: 1,
10565 name: "cdn".into(),
10566 domain: "cdn.yah.dev".into(),
10567 front_door: FrontDoor::BucketDirect,
10568 cdn_bucket: "yah-dev".into(),
10569 worker_bundle_path: None,
10570 routes: vec![],
10571 },
10572 )]
10573 .into_iter()
10574 .collect();
10575 assert!(domain_serving_service(&doms, "yah-marketing").is_none());
10576 }
10577
10578 #[test]
10579 fn a_mount_that_disagrees_with_its_route_path_is_rejected() {
10580 let tmp = tempfile::TempDir::new().unwrap();
10581 let root = tmp.path();
10582 write_two_component_service(root);
10583 write_domain_toml(
10584 root,
10585 "yah-dev",
10586 r#"
10587schema_version = 1
10588name = "yah-dev"
10589domain = "yah.dev"
10590front_door = "worker"
10591cdn_bucket = "yah-dev"
10592
10593[[routes]]
10594path = "/studio/*"
10595mode = "static"
10596component = "yah-marketing/app"
10597"#,
10598 );
10599 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10600 assert!(err.contains("mount = \"/app\""), "{err}");
10601 assert!(err.contains("/studio/*"), "{err}");
10602 }
10603
10604 /// The other direction: routing an unmounted component under a sub-path
10605 /// points requests at a prefix nothing published to.
10606 #[test]
10607 fn routing_an_unmounted_component_under_a_subpath_is_rejected() {
10608 let tmp = tempfile::TempDir::new().unwrap();
10609 let root = tmp.path();
10610 write_marketing_service(root);
10611 write_domain_toml(
10612 root,
10613 "yah-dev",
10614 r#"
10615schema_version = 1
10616name = "yah-dev"
10617domain = "yah.dev"
10618front_door = "worker"
10619cdn_bucket = "yah-dev"
10620
10621[[routes]]
10622path = "/docs/*"
10623mode = "static"
10624component = "yah-marketing/site"
10625"#,
10626 );
10627 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10628 assert!(err.contains("no `mount`"), "{err}");
10629 assert!(err.contains("/docs"), "{err}");
10630 }
10631
10632 #[test]
10633 fn mount_and_route_prefix_normalization_agree() {
10634 for m in ["/app", "app", "app/", "/app/"] {
10635 assert_eq!(normalize_mount(m), "app", "mount {m:?}");
10636 }
10637 assert_eq!(normalize_mount("/"), "");
10638 assert_eq!(route_path_prefix("/*"), "");
10639 assert_eq!(route_path_prefix("/app/*"), "app");
10640 assert_eq!(route_path_prefix("/app"), "app");
10641 assert_eq!(route_path_prefix("/"), "");
10642 }
10643
10644 // ── R870-B11: a mount is owned by exactly one bundle-tier component ────
10645
10646 /// Two bundle-tier components at the same explicit mount would stage into
10647 /// the same `app/dist/<mount>/` prefix inside one assembled bundle and
10648 /// silently clobber each other — reject at load, before that happens.
10649 #[test]
10650 fn two_bundle_components_at_the_same_mount_are_rejected() {
10651 let tmp = tempfile::TempDir::new().unwrap();
10652 let root = tmp.path();
10653 let svc = ServiceConfig {
10654 schema_version: 1,
10655 name: "noisetable-marketing".into(),
10656 address: ServiceAddress::front_door("noisetable.com"),
10657 description: None,
10658 db: DbCatalog::default(),
10659 components: vec![
10660 ServiceComponent {
10661 mount: Some("/app".into()),
10662 id: "app".into(),
10663 kind: "mesofact-static".into(),
10664 path: "app/browser".into(),
10665 role: "static".into(),
10666 publishes: None,
10667 wave: 0,
10668 git: None,
10669 deploy: Default::default(),
10670 },
10671 ServiceComponent {
10672 mount: Some("app/".into()),
10673 id: "app2".into(),
10674 kind: "mesofact-spa".into(),
10675 path: "app/other".into(),
10676 role: "static".into(),
10677 publishes: None,
10678 wave: 0,
10679 git: None,
10680 deploy: Default::default(),
10681 },
10682 ],
10683 };
10684 svc.save(root).unwrap();
10685 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10686 assert!(err.contains("\"app\""), "{err}");
10687 assert!(err.contains("\"app2\""), "{err}");
10688 assert!(err.contains("mount = \"/app\""), "{err}");
10689 }
10690
10691 /// The unmounted case: two bundle-tier components both leaving `mount`
10692 /// unset both claim the service root, which collides exactly the same
10693 /// way — this is the noisetable shape the ticket was filed against, if
10694 /// `app`'s mount had been forgotten instead of declared.
10695 #[test]
10696 fn two_bundle_components_with_no_mount_are_rejected() {
10697 let tmp = tempfile::TempDir::new().unwrap();
10698 let root = tmp.path();
10699 let svc = ServiceConfig {
10700 schema_version: 1,
10701 name: "noisetable-marketing".into(),
10702 address: ServiceAddress::front_door("noisetable.com"),
10703 description: None,
10704 db: DbCatalog::default(),
10705 components: vec![
10706 ServiceComponent {
10707 mount: None,
10708 id: "site".into(),
10709 kind: "mesofact-spa".into(),
10710 path: "web/landing".into(),
10711 role: "static".into(),
10712 publishes: None,
10713 wave: 0,
10714 git: None,
10715 deploy: Default::default(),
10716 },
10717 ServiceComponent {
10718 mount: None,
10719 id: "app".into(),
10720 kind: "mesofact-static".into(),
10721 path: "app/browser".into(),
10722 role: "static".into(),
10723 publishes: None,
10724 wave: 0,
10725 git: None,
10726 deploy: Default::default(),
10727 },
10728 ],
10729 };
10730 svc.save(root).unwrap();
10731 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10732 assert!(err.contains("\"site\""), "{err}");
10733 assert!(err.contains("\"app\""), "{err}");
10734 assert!(err.contains("the service root"), "{err}");
10735 }
10736
10737 /// R870-F23 widened the same loop to the workload tier. Two components in
10738 /// DIFFERENT tiers at one mount is the same clobber read from the routing
10739 /// side: the inner door's table names one upstream for that prefix, so a
10740 /// request reaches either the bundle or the workload and nothing says
10741 /// which.
10742 #[test]
10743 fn a_bundle_component_and_a_workload_component_at_one_mount_are_rejected() {
10744 let tmp = tempfile::TempDir::new().unwrap();
10745 let root = tmp.path();
10746 let svc = ServiceConfig {
10747 schema_version: 1,
10748 name: "noisetable".into(),
10749 address: ServiceAddress::front_door("noisetable.com"),
10750 description: None,
10751 db: DbCatalog::default(),
10752 components: vec![
10753 ServiceComponent {
10754 mount: Some("app".into()),
10755 id: "app-bundle".into(),
10756 kind: "mesofact-spa".into(),
10757 path: "app/browser".into(),
10758 role: "static".into(),
10759 publishes: None,
10760 wave: 0,
10761 git: None,
10762 deploy: DeployTier::Bundle,
10763 },
10764 ServiceComponent {
10765 mount: Some("/app/".into()),
10766 id: "app-service".into(),
10767 kind: "container".into(),
10768 path: "app/server".into(),
10769 role: "compute".into(),
10770 publishes: None,
10771 wave: 0,
10772 git: None,
10773 deploy: DeployTier::Workload,
10774 },
10775 ],
10776 };
10777 svc.save(root).unwrap();
10778 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10779 assert!(err.contains("\"app-bundle\""), "{err}");
10780 assert!(err.contains("\"app-service\""), "{err}");
10781 assert!(err.contains("deploys as its own workload"), "{err}");
10782 }
10783
10784 /// And two WORKLOAD-tier components at one mount, which the pre-R870-F23
10785 /// loop skipped entirely (it filtered on `kind`, and a workload-tier
10786 /// component need not be a mesofact kind at all).
10787 #[test]
10788 fn two_workload_components_at_one_mount_are_rejected() {
10789 let tmp = tempfile::TempDir::new().unwrap();
10790 let root = tmp.path();
10791 let svc = ServiceConfig {
10792 schema_version: 1,
10793 name: "noisetable".into(),
10794 address: ServiceAddress::front_door("noisetable.com"),
10795 description: None,
10796 db: DbCatalog::default(),
10797 components: vec![
10798 ServiceComponent {
10799 mount: None,
10800 id: "api".into(),
10801 kind: "container".into(),
10802 path: "svc/api".into(),
10803 role: "compute".into(),
10804 publishes: None,
10805 wave: 0,
10806 git: None,
10807 deploy: DeployTier::Workload,
10808 },
10809 ServiceComponent {
10810 mount: Some("/".into()),
10811 id: "api2".into(),
10812 kind: "container".into(),
10813 path: "svc/api2".into(),
10814 role: "compute".into(),
10815 publishes: None,
10816 wave: 0,
10817 git: None,
10818 deploy: DeployTier::Workload,
10819 },
10820 ],
10821 };
10822 svc.save(root).unwrap();
10823 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10824 assert!(err.contains("inner-door route table"), "{err}");
10825 assert!(err.contains("the service root"), "{err}");
10826 }
10827
10828 /// The legitimate shape (distinct mounts) is untouched — regression guard
10829 /// so the new check does not become the next silent-overwrite bug.
10830 #[test]
10831 fn bundle_components_at_distinct_mounts_still_load() {
10832 let tmp = tempfile::TempDir::new().unwrap();
10833 let root = tmp.path();
10834 write_two_component_service(root);
10835 assert!(CloudConfig::load(root).is_ok());
10836 }
10837
10838 #[test]
10839 fn worker_with_no_routes_is_rejected() {
10840 let tmp = tempfile::TempDir::new().unwrap();
10841 let root = tmp.path();
10842 write_domain_toml(
10843 root,
10844 "yah-dev",
10845 r#"
10846schema_version = 1
10847name = "yah-dev"
10848domain = "yah.dev"
10849front_door = "worker"
10850cdn_bucket = "yah-dev"
10851"#,
10852 );
10853 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10854 assert!(err.contains("front_door = \"worker\""), "{err}");
10855 assert!(err.contains("404"), "{err}");
10856 }
10857
10858 #[test]
10859 fn passway_with_no_routes_is_rejected_too() {
10860 let tmp = tempfile::TempDir::new().unwrap();
10861 let root = tmp.path();
10862 write_domain_toml(
10863 root,
10864 "yah-dev",
10865 r#"
10866schema_version = 1
10867name = "yah-dev"
10868domain = "yah.dev"
10869front_door = "passway"
10870cdn_bucket = "yah-dev"
10871"#,
10872 );
10873 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10874 assert!(err.contains("front_door = \"passway\""), "{err}");
10875 }
10876
10877 #[test]
10878 fn bucket_direct_without_routes_loads() {
10879 let tmp = tempfile::TempDir::new().unwrap();
10880 let root = tmp.path();
10881 // Exactly the shape .yah/domains/cdn-yah-dev.toml ships (W175: a pure
10882 // asset tier deliberately has no Worker behaviours).
10883 write_domain_toml(
10884 root,
10885 "cdn-yah-dev",
10886 r#"
10887schema_version = 1
10888name = "cdn-yah-dev"
10889domain = "cdn.yah.dev"
10890front_door = "bucket-direct"
10891cdn_bucket = "yah-dev"
10892"#,
10893 );
10894 let cfg = CloudConfig::load(root).unwrap();
10895 let dom = cfg.domain("cdn-yah-dev").expect("cdn-yah-dev domain");
10896 assert_eq!(dom.front_door, FrontDoor::BucketDirect);
10897 assert!(!dom.front_door.is_route_driven());
10898 }
10899
10900 #[test]
10901 fn front_door_round_trips_through_save() {
10902 let tmp = tempfile::TempDir::new().unwrap();
10903 let root = tmp.path();
10904 write_marketing_service(root);
10905 let dom = DomainConfig {
10906 schema_version: 1,
10907 name: "yah-dev".into(),
10908 domain: "yah.dev".into(),
10909 front_door: FrontDoor::Passway,
10910 cdn_bucket: "yah-dev".into(),
10911 worker_bundle_path: None,
10912 routes: vec![DomainRoute {
10913 headers: Default::default(),
10914 path: "/*".into(),
10915 mode: RouteMode::Static {
10916 component: "yah-marketing/site".into(),
10917 },
10918 }],
10919 };
10920 dom.save(root).unwrap();
10921 let cfg = CloudConfig::load(root).unwrap();
10922 assert_eq!(
10923 cfg.domain("yah-dev").unwrap().front_door,
10924 FrontDoor::Passway
10925 );
10926 }
10927
10928 // The four manifests this repo actually ships are asserted in
10929 // `tests/live_workspace_smoke.rs` — that's the only place with a
10930 // depth-agnostic path to the live `.yah/` tree and a skip path for the
10931 // standalone mirror checkout.
10932
10933 #[test]
10934 fn cross_ref_bails_on_missing_service() {
10935 let tmp = tempfile::TempDir::new().unwrap();
10936 let root = tmp.path();
10937 // No services declared at all — component ref must fail to resolve.
10938 let dom = DomainConfig {
10939 schema_version: 1,
10940 name: "yah-dev".into(),
10941 domain: "yah.dev".into(),
10942 front_door: FrontDoor::Worker,
10943 cdn_bucket: "yah-dev".into(),
10944 worker_bundle_path: None,
10945 routes: vec![DomainRoute {
10946 headers: Default::default(),
10947 path: "/".into(),
10948 mode: RouteMode::Static {
10949 component: "yah-marketing/site".into(),
10950 },
10951 }],
10952 };
10953 dom.save(root).unwrap();
10954
10955 let err = CloudConfig::load(root).unwrap_err();
10956 let msg = format!("{err:#}");
10957 assert!(msg.contains("no such service"), "got: {msg}");
10958 assert!(msg.contains("yah-marketing"), "got: {msg}");
10959 }
10960
10961 #[test]
10962 fn cross_ref_bails_on_missing_component() {
10963 let tmp = tempfile::TempDir::new().unwrap();
10964 let root = tmp.path();
10965 write_marketing_service(root); // has component id "site", not "elsewhere"
10966
10967 let dom = DomainConfig {
10968 schema_version: 1,
10969 name: "yah-dev".into(),
10970 domain: "yah.dev".into(),
10971 front_door: FrontDoor::Worker,
10972 cdn_bucket: "yah-dev".into(),
10973 worker_bundle_path: None,
10974 routes: vec![DomainRoute {
10975 headers: Default::default(),
10976 path: "/".into(),
10977 mode: RouteMode::Static {
10978 component: "yah-marketing/elsewhere".into(),
10979 },
10980 }],
10981 };
10982 dom.save(root).unwrap();
10983
10984 let err = CloudConfig::load(root).unwrap_err();
10985 let msg = format!("{err:#}");
10986 assert!(msg.contains("no component with id"), "got: {msg}");
10987 assert!(msg.contains("elsewhere"), "got: {msg}");
10988 }
10989
10990 #[test]
10991 fn cross_ref_bails_on_malformed_ref() {
10992 let tmp = tempfile::TempDir::new().unwrap();
10993 let root = tmp.path();
10994 write_marketing_service(root);
10995
10996 let dom = DomainConfig {
10997 schema_version: 1,
10998 name: "yah-dev".into(),
10999 domain: "yah.dev".into(),
11000 front_door: FrontDoor::Worker,
11001 cdn_bucket: "yah-dev".into(),
11002 worker_bundle_path: None,
11003 routes: vec![DomainRoute {
11004 headers: Default::default(),
11005 path: "/".into(),
11006 mode: RouteMode::Static {
11007 component: "no-slash-here".into(),
11008 },
11009 }],
11010 };
11011 dom.save(root).unwrap();
11012
11013 let err = CloudConfig::load(root).unwrap_err();
11014 let msg = format!("{err:#}");
11015 assert!(msg.contains("expected"), "got: {msg}");
11016 }
11017
11018 #[test]
11019 fn redirect_routes_skip_component_validation() {
11020 let tmp = tempfile::TempDir::new().unwrap();
11021 let root = tmp.path();
11022 // No services at all — redirect must still load cleanly because it
11023 // references nothing.
11024 let dom = DomainConfig {
11025 schema_version: 1,
11026 name: "yah-dev".into(),
11027 domain: "yah.dev".into(),
11028 front_door: FrontDoor::Worker,
11029 cdn_bucket: "yah-dev".into(),
11030 worker_bundle_path: None,
11031 routes: vec![DomainRoute {
11032 headers: Default::default(),
11033 path: "/old".into(),
11034 mode: RouteMode::Redirect {
11035 target: "https://yah.dev/blog".into(),
11036 status: 308,
11037 },
11038 }],
11039 };
11040 dom.save(root).unwrap();
11041
11042 let cfg = CloudConfig::load(root).unwrap();
11043 assert!(cfg.domain("yah-dev").is_some());
11044 }
11045
11046 #[test]
11047 fn name_must_match_file_stem() {
11048 let tmp = tempfile::TempDir::new().unwrap();
11049 let root = tmp.path();
11050 // Hand-write a file whose stem disagrees with its `name`.
11051 let dir = root.join(".yah").join("domains");
11052 std::fs::create_dir_all(&dir).unwrap();
11053 std::fs::write(
11054 dir.join("yah-dev.toml"),
11055 r#"schema_version = 1
11056name = "different-name"
11057domain = "yah.dev"
11058front_door = "bucket-direct"
11059cdn_bucket = "yah-dev"
11060"#,
11061 )
11062 .unwrap();
11063
11064 let err = CloudConfig::load(root).unwrap_err();
11065 let msg = format!("{err:#}");
11066 assert!(msg.contains("must match the file stem"), "got: {msg}");
11067 }
11068
11069 #[test]
11070 fn net_alias_tier_subdomain_manifest_loads_and_cross_refs() {
11071 // R561-F2: a per-tenant subdomain manifest on the net.yah.dev wildcard
11072 // alias tier is just a DomainConfig whose `domain` is `<name>.net.yah.dev`
11073 // and whose static route cross-refs the tenant's service component.
11074 // This is exactly the shape .yah/domains/scrabcake-net-yah-dev.toml ships.
11075 let tmp = tempfile::TempDir::new().unwrap();
11076 let root = tmp.path();
11077 write_marketing_service(root); // service "yah-marketing", component "site"
11078
11079 let dom = DomainConfig {
11080 schema_version: 1,
11081 name: "tenant-net-yah-dev".into(),
11082 domain: "tenant.net.yah.dev".into(),
11083 front_door: FrontDoor::Worker,
11084 cdn_bucket: "net-yah-dev".into(), // shared per-tier bucket
11085 worker_bundle_path: None,
11086 routes: vec![DomainRoute {
11087 headers: Default::default(),
11088 path: "/*".into(),
11089 mode: RouteMode::Static {
11090 component: "yah-marketing/site".into(),
11091 },
11092 }],
11093 };
11094 dom.save(root).unwrap();
11095
11096 let cfg = CloudConfig::load(root).unwrap();
11097 let dom = cfg
11098 .domain("tenant-net-yah-dev")
11099 .expect("net-tier subdomain manifest should load");
11100 assert_eq!(dom.domain, "tenant.net.yah.dev");
11101 assert_eq!(dom.cdn_bucket, "net-yah-dev");
11102 }
11103
11104 // ─── R572-F3: NodeAllocatable + taints ──────────────────────────────────
11105
11106 #[test]
11107 fn machine_allocatable_round_trips() {
11108 let toml_src = r#"
11109name = "us-west-001"
11110provider = "static"
11111mesh_tags = ["tag:cloud-runner"]
11112[allocatable]
11113memory_mb = 3800
11114cpu_millis = 2000
11115"#;
11116 let m: MachineConfig = toml::from_str(toml_src).unwrap();
11117 let a = m.allocatable.as_ref().expect("allocatable should parse");
11118 assert_eq!(a.memory_mb, 3800);
11119 assert_eq!(a.cpu_millis, 2000);
11120
11121 let s = toml::to_string(&m).unwrap();
11122 let back: MachineConfig = toml::from_str(&s).unwrap();
11123 let a2 = back.allocatable.as_ref().unwrap();
11124 assert_eq!(a2.memory_mb, 3800);
11125 assert_eq!(a2.cpu_millis, 2000);
11126 }
11127
11128 #[test]
11129 fn machine_taints_round_trips() {
11130 let toml_src = r#"
11131name = "us-south-001"
11132provider = "static"
11133mesh_tags = ["tag:cloud-runner"]
11134taints = ["no-appliance"]
11135"#;
11136 let m: MachineConfig = toml::from_str(toml_src).unwrap();
11137 assert_eq!(m.taints, vec!["no-appliance"]);
11138
11139 let s = toml::to_string(&m).unwrap();
11140 let back: MachineConfig = toml::from_str(&s).unwrap();
11141 assert_eq!(back.taints, vec!["no-appliance"]);
11142 }
11143
11144 #[test]
11145 fn machine_allocatable_absent_is_none() {
11146 let toml_src = "name = \"node\"\nprovider = \"static\"\nmesh_tags = []\n";
11147 let m: MachineConfig = toml::from_str(toml_src).unwrap();
11148 assert!(m.allocatable.is_none());
11149 assert!(m.taints.is_empty());
11150 }
11151
11152 #[test]
11153 fn machine_allocatable_skipped_when_none() {
11154 let m = make_machine("node", vec![]);
11155 let s = toml::to_string(&m).unwrap();
11156 assert!(
11157 !s.contains("allocatable"),
11158 "None allocatable must be omitted: {s}"
11159 );
11160 assert!(!s.contains("taints"), "empty taints must be omitted: {s}");
11161 }
11162
11163 #[test]
11164 fn machine_multiple_taints_round_trip() {
11165 let toml_src = r#"
11166name = "quarantined"
11167provider = "static"
11168mesh_tags = ["tag:build-worker"]
11169taints = ["no-server", "no-appliance", "no-job"]
11170"#;
11171 let m: MachineConfig = toml::from_str(toml_src).unwrap();
11172 assert_eq!(m.taints.len(), 3);
11173 assert!(m.taints.contains(&"no-server".to_string()));
11174 assert!(m.taints.contains(&"no-appliance".to_string()));
11175 assert!(m.taints.contains(&"no-job".to_string()));
11176 // R742-T4: every key here is one the scheduler reads. This fixture
11177 // used to carry `no-voter`, which none of them is.
11178 assert!(m.inert_taints().is_empty());
11179 }
11180
11181 // ─── R742-T4 (W305): inert-taint classification ─────────────────────────
11182
11183 #[test]
11184 fn every_archetype_repel_key_is_live() {
11185 for arch in LifecycleArchetype::ALL {
11186 let key = format!("no-{}", arch.taint_key());
11187 assert_eq!(
11188 taint_effect(&key),
11189 TaintEffect::Repels(arch),
11190 "{key} must repel {arch:?}"
11191 );
11192 }
11193 }
11194
11195 #[test]
11196 fn public_ip_is_an_affinity_key_not_an_inert_one() {
11197 assert_eq!(
11198 taint_effect(workload_spec::PUBLIC_IP_TAINT),
11199 TaintEffect::Attracts
11200 );
11201 }
11202
11203 #[test]
11204 fn a_free_form_taint_is_inert_and_says_so() {
11205 // W305's headline example: `taints = ["qa"]` parsed clean and did
11206 // nothing. Environment is not expressible as a taint.
11207 assert_eq!(taint_effect("qa"), TaintEffect::Inert);
11208 // And the one that actually cost fleet state: `no-voter` reads as an
11209 // exclusion and excludes nothing — "voter" is not an archetype.
11210 assert_eq!(taint_effect("no-voter"), TaintEffect::Inert);
11211 // A near-miss on a real key is inert too, not silently forgiven.
11212 assert_eq!(taint_effect("no-servers"), TaintEffect::Inert);
11213
11214 let m = make_machine_with_capacity(
11215 "dev-pi",
11216 8192,
11217 4000,
11218 vec!["no-appliance", "no-voter", "qa"],
11219 );
11220 assert_eq!(m.inert_taints(), vec!["no-voter", "qa"]);
11221 }
11222
11223 // ─── R876-B7: repel-by-default + declarable toleration ──────────────────
11224
11225 /// The headline inversion. A bare `RequiredSpec` — which is exactly what
11226 /// deserializing a mirror's `required = { regions, mesh_tags }` produces,
11227 /// since no TOML in the tree writes `tolerates` — is now repelled by a
11228 /// repelling taint. Before B7 it matched, because repulsion was conditional
11229 /// on a `#[serde(skip)]` field that this path could never fill.
11230 #[test]
11231 fn an_undeclared_spec_is_repelled_by_a_repelling_taint() {
11232 let tainted = make_machine_with_capacity("n", 8192, 4000, vec!["no-server"]);
11233 assert!(!RequiredSpec::default().matches(&tainted));
11234
11235 // And it is the DESERIALIZED shape that matters, not a hand-built one:
11236 // this is the mirror path reproduced exactly.
11237 let from_toml: RequiredSpec =
11238 toml::from_str("regions = [\"us-east\"]\n").expect("a mirror-shaped required parses");
11239 assert!(from_toml.tolerates.is_empty());
11240 let mut in_region = tainted.clone();
11241 in_region.region = Some("us-east".to_string());
11242 assert!(
11243 !from_toml.matches(&in_region),
11244 "a mirror-declared placement must now read machine.taints"
11245 );
11246 }
11247
11248 /// The opt-back-in half, and the one an operator writes by hand.
11249 #[test]
11250 fn an_explicit_toleration_admits_the_tainted_machine_again() {
11251 let tainted = make_machine_with_capacity("n", 8192, 4000, vec!["no-server"]);
11252 let spec = RequiredSpec {
11253 tolerates: vec!["no-server".to_string()],
11254 ..Default::default()
11255 };
11256 assert!(spec.matches(&tainted));
11257
11258 // Per-key, not a blanket pass: tolerating one repelling key says nothing
11259 // about another.
11260 let both = make_machine_with_capacity("n", 8192, 4000, vec!["no-server", "no-appliance"]);
11261 assert!(!spec.matches(&both));
11262
11263 // And it deserializes — the whole point of replacing a `#[serde(skip)]`
11264 // field is that a mirror can now declare this.
11265 let from_toml: RequiredSpec = toml::from_str("tolerates = [\"no-server\"]\n")
11266 .expect("a slot can declare a toleration");
11267 assert!(from_toml.matches(&tainted));
11268 }
11269
11270 /// An untainted machine is unaffected, which is what makes the migration
11271 /// bounded: six of the nine fleet machines carry no repelling taint at all.
11272 #[test]
11273 fn an_untainted_machine_matches_exactly_as_before() {
11274 let clean = make_machine_with_capacity("n", 8192, 4000, vec![]);
11275 assert!(RequiredSpec::default().matches(&clean));
11276 assert!(RequiredSpec {
11277 tolerates: vec!["no-server".to_string()],
11278 ..Default::default()
11279 }
11280 .matches(&clean));
11281 }
11282
11283 /// THE MIGRATION'S LOAD-BEARING FACT. `public-ip` is on three fleet nodes
11284 /// including us-east-001, the only origin serving the yah.dev apex. It is an
11285 /// *affinity* key, so repel-by-default must not touch it — reading every
11286 /// taint as repulsion would evict the apex on the next apply.
11287 #[test]
11288 fn an_affinity_taint_does_not_repel() {
11289 let public = make_machine_with_capacity("us-east-001", 8192, 4000, vec!["public-ip"]);
11290 assert!(
11291 RequiredSpec::default().matches(&public),
11292 "public-ip attracts; it must never be read as repulsion"
11293 );
11294 }
11295
11296 /// `select_matching` filters on the same predicate, so a tainted machine
11297 /// leaves the candidate set rather than being silently placed onto.
11298 #[test]
11299 fn select_matching_drops_a_tainted_candidate_and_keeps_the_rest() {
11300 let drained = make_machine_with_capacity("drained", 8192, 4000, vec!["no-server"]);
11301 let healthy = make_machine_with_capacity("healthy", 8192, 4000, vec![]);
11302 let pool = [&drained, &healthy];
11303
11304 let picked = select_matching(&pool, &RequiredSpec::default(), 1, "test pool", "empty")
11305 .expect("one candidate remains");
11306 assert_eq!(
11307 picked.iter().map(|m| m.name.as_str()).collect::<Vec<_>>(),
11308 vec!["healthy"],
11309 "tainting the first candidate moves the placement to the second"
11310 );
11311
11312 // Asking for both is a shortfall, not a half-placement.
11313 let err = select_matching(&pool, &RequiredSpec::default(), 2, "test pool", "empty")
11314 .expect_err("only one of two matches");
11315 assert!(format!("{err:#}").contains("only 1 of 2 machines match"));
11316 }
11317
11318 /// The `admit_workload` path must be behaviourally unchanged: its spec is
11319 /// built by `admission_spec`, which now emits the complementary tolerations.
11320 #[test]
11321 fn admission_preserves_archetype_scoped_repulsion_across_the_inversion() {
11322 let ws = minimal_spec("srv", 1); // a Server
11323 let req = admission_spec(&ws, &[]).unwrap();
11324 assert_eq!(ws.effective_archetype(), LifecycleArchetype::Server);
11325
11326 let no_server = make_machine_with_capacity("n", 8192, 4000, vec!["no-server"]);
11327 let no_appliance = make_machine_with_capacity("n", 8192, 4000, vec!["no-appliance"]);
11328 assert!(!req.matches(&no_server), "its own class still repels it");
11329 assert!(
11330 req.matches(&no_appliance),
11331 "another class's taint still does not — this is the pre-B7 answer"
11332 );
11333 }
11334
11335 #[test]
11336 fn describe_names_the_toleration_so_a_refusal_is_readable() {
11337 let spec = RequiredSpec {
11338 regions: vec!["us-west".to_string()],
11339 tolerates: vec!["no-appliance".to_string()],
11340 ..Default::default()
11341 };
11342 assert_eq!(
11343 spec.describe(),
11344 "required.regions=[us-west] + required.tolerates=[no-appliance]"
11345 );
11346 }
11347
11348 /// A toleration widens; it must not make an underspecified slot look
11349 /// specified, or the deploy side stops refusing one.
11350 #[test]
11351 fn a_toleration_alone_is_still_an_unconstrained_spec() {
11352 assert!(RequiredSpec {
11353 tolerates: vec!["no-server".to_string()],
11354 ..Default::default()
11355 }
11356 .is_unconstrained());
11357 }
11358
11359 #[test]
11360 fn an_inert_taint_changes_no_placement_decision() {
11361 // The reason this is a lint and not a behaviour change: the guard's
11362 // whole premise is that these keys are invisible to `matches`.
11363 let clean = make_machine_with_capacity("n", 8192, 4000, vec![]);
11364 let noisy = make_machine_with_capacity("n", 8192, 4000, vec!["no-voter", "qa"]);
11365 // R876-B7: still true under repel-by-default, and for a sharper reason —
11366 // `matches` now walks `machine.taints` itself, so an unclassifiable key
11367 // is skipped by `taint_effect` rather than merely never looked up.
11368 for arch in LifecycleArchetype::ALL {
11369 let req = RequiredSpec {
11370 tolerates: tolerations_excluding(&[arch]),
11371 ..Default::default()
11372 };
11373 assert_eq!(req.matches(&clean), req.matches(&noisy));
11374 assert!(req.matches(&noisy), "neither key repels");
11375 }
11376 }
11377
11378 /// The toleration set [`admission_spec`] derives for a group of `archetypes`
11379 /// — every repelling key that is not the group's own class.
11380 fn tolerations_excluding(archetypes: &[LifecycleArchetype]) -> Vec<String> {
11381 LifecycleArchetype::ALL
11382 .into_iter()
11383 .filter(|a| !archetypes.contains(a))
11384 .map(|a| format!("no-{}", a.taint_key()))
11385 .collect()
11386 }
11387
11388 #[test]
11389 fn live_taint_keys_lists_the_whole_legal_vocabulary() {
11390 assert_eq!(
11391 live_taint_keys(),
11392 vec!["no-appliance", "no-job", "no-server", "public-ip"]
11393 );
11394 }
11395
11396 // ─── R742-F1 (W305): sovereign groups ───────────────────────────────────
11397
11398 /// A machine in `group`, with the role left unwritten — which is the state
11399 /// of every machine TOML that predates R605-F12 and resolves to `voter`.
11400 fn in_group(name: &str, group: Option<&str>) -> MachineConfig {
11401 MachineConfig {
11402 sovereign_group: group.map(String::from),
11403 ..make_machine(name, vec![])
11404 }
11405 }
11406
11407 /// A machine in `group` with its quorum eligibility stated (R605-F12).
11408 fn in_group_as(name: &str, group: &str, role: SovereignRole) -> MachineConfig {
11409 MachineConfig {
11410 sovereign_group: Some(group.to_string()),
11411 sovereign_role: Some(role),
11412 ..make_machine(name, vec![])
11413 }
11414 }
11415
11416 #[test]
11417 fn a_join_within_one_sovereign_group_is_permitted() {
11418 assert_eq!(
11419 judge_join(
11420 &in_group("us-west-013", Some("dev")),
11421 &in_group("us-west-011", Some("dev")),
11422 ),
11423 JoinVerdict::Permit
11424 );
11425 }
11426
11427 /// The case the field exists for: before it, the only thing standing
11428 /// between a dev Pi and the prod quorum was a comment in a TOML.
11429 #[test]
11430 fn a_cross_group_join_is_refused_naming_both_groups() {
11431 let verdict = judge_join(
11432 &in_group("us-west-011", Some("dev")),
11433 &in_group("us-west-001", Some("prod")),
11434 );
11435 let JoinVerdict::Refuse(msg) = verdict else {
11436 panic!("a dev node joining prod must be refused: {verdict:?}");
11437 };
11438 // A refusal that does not name what it saw is one the operator has to
11439 // go and reconstruct, so it gets worked around instead of fixed.
11440 assert!(msg.contains("us-west-011") && msg.contains("us-west-001"), "{msg}");
11441 assert!(msg.contains("dev") && msg.contains("prod"), "{msg}");
11442 }
11443
11444 /// `None` is a declaration ("standalone, in no group"), not a gap — so
11445 /// growing prod with an unstamped box is a cross-group join too, and the
11446 /// refusal has to say which file makes it legal.
11447 #[test]
11448 fn an_undeclared_node_cannot_join_a_declared_group() {
11449 let verdict = judge_join(
11450 &in_group("us-west-002", None),
11451 &in_group("us-west-001", Some("prod")),
11452 );
11453 let JoinVerdict::Refuse(msg) = verdict else {
11454 panic!("an unstamped node joining prod must be refused: {verdict:?}");
11455 };
11456 assert!(
11457 msg.contains(".yah/infra/machines/us-west-002.toml"),
11458 "the refusal must name the file to stamp: {msg}"
11459 );
11460 }
11461
11462 #[test]
11463 fn a_declared_node_cannot_join_a_standalone_target() {
11464 // us-west-003 is `mode: standalone` on purpose; it is not a group of
11465 // one waiting to be grown.
11466 let verdict = judge_join(
11467 &in_group("us-west-001", Some("prod")),
11468 &in_group("us-west-003", None),
11469 );
11470 assert!(matches!(verdict, JoinVerdict::Refuse(msg) if msg.contains("us-west-003")));
11471 }
11472
11473 #[test]
11474 fn two_undeclared_nodes_cannot_form_an_undeclared_group() {
11475 let verdict = judge_join(
11476 &in_group("us-west-002", None),
11477 &in_group("us-west-015", None),
11478 );
11479 assert!(
11480 matches!(&verdict, JoinVerdict::Refuse(msg) if msg.contains("us-west-002")
11481 && msg.contains("us-west-015")),
11482 "forming a group nobody declared must be refused, naming both: {verdict:?}"
11483 );
11484 }
11485
11486 // ─── R605-F12: the voting axis ──────────────────────────────────────────
11487
11488 /// The whole ticket in one assertion. us-west-003 is a member of prod —
11489 /// same secrets, same upgrade cadence, same destruction — and must never
11490 /// hold a prod raft seat. Before the role axis, the only thing refusing it
11491 /// was its *absent* group stamp, so writing down the truth above would have
11492 /// removed the guard.
11493 #[test]
11494 fn a_non_voting_member_is_refused_into_its_own_group() {
11495 let verdict = judge_join(
11496 &in_group_as("us-west-003", "prod", SovereignRole::NonVoter),
11497 &in_group_as("us-west-001", "prod", SovereignRole::Voter),
11498 );
11499 let JoinVerdict::Refuse(msg) = verdict else {
11500 panic!("a non-voting prod member must not join the prod quorum: {verdict:?}");
11501 };
11502 assert!(msg.contains("us-west-003") && msg.contains("NON-VOTING"), "{msg}");
11503 // The refusal must not blame the group: both sides say "prod", and a
11504 // cross-group message here would read as a bug in the check itself.
11505 assert!(!msg.contains("cross-group"), "{msg}");
11506 assert!(
11507 msg.contains(".yah/infra/machines/us-west-003.toml"),
11508 "the refusal must name the file that decides it: {msg}"
11509 );
11510 }
11511
11512 /// Read from the other end: a box declared non-voting has no quorum seat to
11513 /// be grown, so it cannot be a join target either.
11514 #[test]
11515 fn a_non_voting_target_has_no_quorum_to_grow() {
11516 let verdict = judge_join(
11517 &in_group_as("us-west-001", "prod", SovereignRole::Voter),
11518 &in_group_as("us-west-003", "prod", SovereignRole::NonVoter),
11519 );
11520 assert!(
11521 matches!(&verdict, JoinVerdict::Refuse(msg) if msg.contains("the target")
11522 && msg.contains("us-west-003")),
11523 "{verdict:?}"
11524 );
11525 }
11526
11527 /// A non-voter joining a *standalone* target is refused for two reasons at
11528 /// once, and the message must pick the one whose fix would actually work.
11529 /// Naming the role here would send the operator to flip `sovereign_role`
11530 /// and come back to the same refusal.
11531 #[test]
11532 fn a_refusal_names_the_group_when_fixing_the_role_would_not_help() {
11533 let verdict = judge_join(
11534 &in_group_as("us-west-003", "prod", SovereignRole::NonVoter),
11535 &in_group("us-west-002", None),
11536 );
11537 let JoinVerdict::Refuse(msg) = verdict else {
11538 panic!("a standalone target has no group to join: {verdict:?}");
11539 };
11540 assert!(
11541 msg.contains(".yah/infra/machines/us-west-002.toml"),
11542 "the refusal must point at the target's missing group stamp: {msg}"
11543 );
11544 assert!(!msg.contains("NON-VOTING"), "{msg}");
11545 }
11546
11547 /// The back-compat seam, pinned: the six nodes stamped before R605-F12
11548 /// write no role, and an absent role means what declaring a group has
11549 /// always meant. If this flips, the live prod and dev quorums stop being
11550 /// growable on a config the operator never edited.
11551 #[test]
11552 fn an_unwritten_role_still_joins_its_group() {
11553 let joiner = in_group("us-west-013", Some("dev"));
11554 assert_eq!(joiner.sovereign_role, None);
11555 assert_eq!(
11556 judge_join(&joiner, &in_group("us-west-011", Some("dev"))),
11557 JoinVerdict::Permit
11558 );
11559 assert_eq!(
11560 judge_join(
11561 &joiner,
11562 &in_group_as("us-west-011", "dev", SovereignRole::Voter)
11563 ),
11564 JoinVerdict::Permit
11565 );
11566 }
11567
11568 /// A non-voting member is still a *member*, and the two claims must not be
11569 /// conflated: `sovereign_membership()` reports the group either way, so a
11570 /// consumer asking "is this box in prod's blast radius" gets yes.
11571 #[test]
11572 fn a_non_voter_is_still_in_the_group_it_names() {
11573 let m = in_group_as("us-west-003", "prod", SovereignRole::NonVoter);
11574 assert_eq!(m.sovereign_membership().group, Some("prod"));
11575 assert!(!m.sovereign_membership().role.is_voter());
11576
11577 // …and the group-membership query the fleet reads is unaffected by it.
11578 let cfg = make_empty_cfg(vec![
11579 m,
11580 in_group_as("us-west-001", "prod", SovereignRole::Voter),
11581 ]);
11582 assert_eq!(
11583 cfg.machines_in_group("prod")
11584 .iter()
11585 .map(|m| m.name.as_str())
11586 .collect::<Vec<_>>(),
11587 vec!["us-west-003", "us-west-001"]
11588 );
11589 }
11590
11591 /// The role travels through TOML in one spelling, and an absent one stays
11592 /// absent on the way back out — otherwise every machine file would grow a
11593 /// `sovereign_role = "voter"` line the operator never wrote, and the
11594 /// unroled-member lint would have nothing left to find.
11595 #[test]
11596 fn sovereign_role_round_trips_and_is_omitted_when_unwritten() {
11597 let m: MachineConfig = toml::from_str(
11598 r#"
11599name = "us-west-003"
11600provider = "static"
11601region = "us-west"
11602arch = "x86_64"
11603mesh_tags = []
11604sovereign_group = "prod"
11605sovereign_role = "non-voter"
11606"#,
11607 )
11608 .unwrap();
11609 assert_eq!(m.sovereign_role, Some(SovereignRole::NonVoter));
11610 assert!(toml::to_string(&m)
11611 .unwrap()
11612 .contains(r#"sovereign_role = "non-voter""#));
11613
11614 let unwritten = MachineConfig {
11615 sovereign_role: None,
11616 ..m
11617 };
11618 assert!(!toml::to_string(&unwritten)
11619 .unwrap()
11620 .contains("sovereign_role"));
11621 }
11622
11623 /// The invariant the ticket is most explicit about: a sovereign group is a
11624 /// blast radius, not a filter. If this ever fails, `matches` has grown an
11625 /// axis it must not have and dev-mode workloads have silently become
11626 /// unschedulable on the dev group.
11627 #[test]
11628 fn sovereign_group_is_not_a_placement_input() {
11629 let standalone = in_group("n", None);
11630 let grouped = in_group("n", Some("dev"));
11631 let other = in_group("n", Some("prod"));
11632
11633 for spec in [
11634 RequiredSpec::default(),
11635 RequiredSpec {
11636 regions: vec!["us-west".into()],
11637 ..Default::default()
11638 },
11639 RequiredSpec {
11640 tolerates: tolerations_excluding(&[LifecycleArchetype::Appliance]),
11641 ..Default::default()
11642 },
11643 ] {
11644 let baseline = spec.matches(&standalone);
11645 assert_eq!(spec.matches(&grouped), baseline);
11646 assert_eq!(spec.matches(&other), baseline);
11647 }
11648 }
11649
11650 // ─── R742-F3 (W305): group → machine set, and group-scoped admission ────
11651
11652 /// The primitive `migrate --to` needs and `rollout plan` still lacks
11653 /// (W314 gap 1): a group exists only as the set of machines naming it, so
11654 /// membership has to be derived rather than declared anywhere.
11655 #[test]
11656 fn machines_in_group_derives_membership_from_the_declarations() {
11657 let cfg = make_empty_cfg(vec![
11658 in_group("us-west-001", Some("prod")),
11659 in_group("us-west-011", Some("dev")),
11660 in_group("us-west-013", Some("dev")),
11661 in_group("us-west-002", None),
11662 ]);
11663
11664 let dev: Vec<&str> = cfg
11665 .machines_in_group("dev")
11666 .iter()
11667 .map(|m| m.name.as_str())
11668 .collect();
11669 assert_eq!(dev, vec!["us-west-011", "us-west-013"]);
11670 assert_eq!(cfg.machines_in_group("prod").len(), 1);
11671
11672 // Standalone is "in no group", not "in a group called none" — so an
11673 // unstamped box is never swept into a migration target.
11674 assert!(cfg.machines_in_group("").is_empty());
11675 assert!(cfg.machines_in_group("staging").is_empty());
11676 }
11677
11678 #[test]
11679 fn declared_sovereign_groups_is_the_vocabulary_a_bad_target_is_named_against() {
11680 let cfg = make_empty_cfg(vec![
11681 in_group("a", Some("prod")),
11682 in_group("b", Some("dev")),
11683 in_group("c", Some("prod")),
11684 in_group("d", None),
11685 ]);
11686 // Sorted + deduped, and standalone contributes nothing.
11687 assert_eq!(cfg.declared_sovereign_groups(), vec!["dev", "prod"]);
11688 assert!(make_empty_cfg(vec![in_group("a", None)])
11689 .declared_sovereign_groups()
11690 .is_empty());
11691 }
11692
11693 /// Group-scoped admission must be the SAME predicate as unscoped
11694 /// admission, only over fewer candidates. If it ever diverges, `migrate`
11695 /// becomes a way to place a workload somewhere `yah cloud apply` would
11696 /// refuse — which is exactly the silent routing-around W305 exists to stop.
11697 #[test]
11698 fn admit_workload_in_group_narrows_candidates_without_changing_the_predicate() {
11699 let mut prod = in_group("us-west-001", Some("prod"));
11700 prod.mesh_tags = vec!["tag:cloud-runner".into()];
11701 let mut dev_repels = in_group("us-west-011", Some("dev"));
11702 dev_repels.taints = vec!["no-appliance".into()];
11703 let mut dev_ok = in_group("us-west-013", Some("dev"));
11704 dev_ok.mesh_tags = vec!["tag:cloud-runner".into()];
11705
11706 let cfg = make_empty_cfg(vec![prod, dev_repels, dev_ok]);
11707
11708 let mut ws = ws_with_selector(None);
11709 ws.archetype = Some(LifecycleArchetype::Appliance);
11710
11711 // Unscoped picks the first match in declaration order.
11712 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "us-west-001");
11713 // Scoped skips the repelling dev node and lands on the other one —
11714 // the taint is honoured, not bypassed.
11715 assert_eq!(
11716 cfg.admit_workload_in_group(&ws, "dev").unwrap().name,
11717 "us-west-013"
11718 );
11719 }
11720
11721 #[test]
11722 fn admit_workload_in_group_distinguishes_an_empty_group_from_a_repelling_one() {
11723 let mut dev = in_group("us-west-011", Some("dev"));
11724 dev.taints = vec!["no-appliance".into()];
11725 let cfg = make_empty_cfg(vec![in_group("us-west-001", Some("prod")), dev]);
11726
11727 let mut ws = ws_with_selector(None);
11728 ws.archetype = Some(LifecycleArchetype::Appliance);
11729
11730 // A group nobody declares names the legal vocabulary, because a typo
11731 // is the realistic cause and "no candidates" would send the operator
11732 // hunting for a placement problem that does not exist.
11733 let missing = cfg.admit_workload_in_group(&ws, "stagng").unwrap_err().to_string();
11734 assert!(missing.contains("no machine declares"), "{missing}");
11735 assert!(missing.contains("dev") && missing.contains("prod"), "{missing}");
11736
11737 // A group that exists but refuses names the machines it tried.
11738 let repelled = cfg.admit_workload_in_group(&ws, "dev").unwrap_err().to_string();
11739 assert!(repelled.contains("us-west-011"), "{repelled}");
11740 }
11741
11742 #[test]
11743 fn sovereign_group_round_trips_and_is_omitted_when_standalone() {
11744 let src = r#"
11745name = "us-west-011"
11746provider = "static"
11747mesh_tags = []
11748sovereign_group = "dev"
11749"#;
11750 let m: MachineConfig = toml::from_str(src).unwrap();
11751 assert_eq!(m.sovereign_group.as_deref(), Some("dev"));
11752 assert!(toml::to_string(&m).unwrap().contains("sovereign_group"));
11753
11754 // A machine that predates the field parses as standalone and does not
11755 // grow the key back on write.
11756 let legacy: MachineConfig =
11757 toml::from_str("name = \"us-west-002\"\nprovider = \"static\"\nmesh_tags = []\n")
11758 .unwrap();
11759 assert_eq!(legacy.sovereign_group, None);
11760 assert!(!toml::to_string(&legacy).unwrap().contains("sovereign_group"));
11761 }
11762
11763 // ─── R572-F5: capacity floor + absolute (untolerable) taints ────────────
11764
11765 fn make_machine_with_capacity(
11766 name: &str,
11767 memory_mb: u32,
11768 cpu_millis: u32,
11769 taints: Vec<&str>,
11770 ) -> MachineConfig {
11771 MachineConfig {
11772 allocatable: Some(NodeAllocatable {
11773 memory_mb,
11774 cpu_millis,
11775 }),
11776 taints: taints.into_iter().map(String::from).collect(),
11777 ..make_machine(name, vec![])
11778 }
11779 }
11780
11781 fn server_spec(memory_mb: u32, cpu_millis: u32) -> WorkloadSpec {
11782 use workload_spec::{ImageRef, LifecycleArchetype, TierTag};
11783 let mut ws = WorkloadSpec::for_forge(
11784 "f5-test",
11785 ImageRef {
11786 registry: "localhost".into(),
11787 repository: "test".into(),
11788 tag: "latest".into(),
11789 digest: workload_spec::testing::test_digest(),
11790 },
11791 TierTag("infra".into()),
11792 vec![],
11793 );
11794 ws.archetype = Some(LifecycleArchetype::Server);
11795 ws.resources.memory_mb = memory_mb;
11796 ws.resources.cpu_millis = cpu_millis;
11797 // These are SERVER specs that borrow `for_forge` as a constructor
11798 // shortcut, so drop the forge memory request it stamps on — otherwise
11799 // every spec here silently requests the forge default instead of the
11800 // `memory_mb` the caller passed, and the capacity-floor tests below
11801 // stop testing their own argument. A server workload declares no
11802 // request, which is the documented fall-back-to-`resources.memory_mb`
11803 // path (`WorkloadSpec::memory_request_mb`).
11804 ws.resources.memory_request_mb = None;
11805 ws
11806 }
11807
11808 fn appliance_spec_ws(memory_mb: u32, cpu_millis: u32) -> WorkloadSpec {
11809 use workload_spec::LifecycleArchetype;
11810 let mut ws = server_spec(memory_mb, cpu_millis);
11811 ws.archetype = Some(LifecycleArchetype::Appliance);
11812 ws
11813 }
11814
11815 #[test]
11816 fn capacity_floor_rejects_undersized_node() {
11817 let cfg = make_empty_cfg(vec![make_machine_with_capacity("small", 256, 500, vec![])]);
11818 let ws = server_spec(512, 1000); // demands more than available
11819 assert!(cfg.admit_workload(&ws).is_err());
11820 }
11821
11822 #[test]
11823 fn capacity_floor_accepts_exact_fit() {
11824 let cfg = make_empty_cfg(vec![make_machine_with_capacity("exact", 512, 1000, vec![])]);
11825 let ws = server_spec(512, 1000);
11826 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "exact");
11827 }
11828
11829 #[test]
11830 fn capacity_floor_passes_when_allocatable_absent() {
11831 // A machine with no allocatable block skips the capacity check (no data).
11832 let cfg = make_empty_cfg(vec![make_machine("no-alloc", vec![])]);
11833 let ws = server_spec(99999, 99999); // would exceed any real node
11834 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "no-alloc");
11835 }
11836
11837 #[test]
11838 fn taint_repulsion_blocks_appliance_on_no_appliance_node() {
11839 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
11840 "south",
11841 1024,
11842 2000,
11843 vec!["no-appliance"],
11844 )]);
11845 let ws = appliance_spec_ws(256, 500);
11846 assert!(
11847 cfg.admit_workload(&ws).is_err(),
11848 "appliance must be repelled by no-appliance taint"
11849 );
11850 }
11851
11852 #[test]
11853 fn taint_repulsion_allows_server_on_no_appliance_node() {
11854 // "no-appliance" only repels Appliance workloads; servers are unaffected.
11855 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
11856 "south",
11857 1024,
11858 2000,
11859 vec!["no-appliance"],
11860 )]);
11861 let ws = server_spec(256, 500);
11862 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "south");
11863 }
11864
11865 #[test]
11866 fn taint_repulsion_job_not_blocked_by_no_server() {
11867 use workload_spec::LifecycleArchetype;
11868 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
11869 "build-box",
11870 8192,
11871 4000,
11872 vec!["no-server", "no-appliance"],
11873 )]);
11874 let mut ws = server_spec(256, 500);
11875 ws.archetype = Some(LifecycleArchetype::Job);
11876 // Job only repelled by "no-job"; "no-server" and "no-appliance" don't affect it.
11877 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "build-box");
11878 }
11879
11880 #[test]
11881 fn requires_taint_affinity_blocks_placement_without_it() {
11882 use workload_spec::{LifecycleArchetype, PUBLIC_IP_TAINT, REQUIRES_TAINT_ANNOTATION};
11883 // Simulate the passway ingress appliance: requires "public-ip" taint.
11884 let mut ws = appliance_spec_ws(256, 512);
11885 ws.archetype = Some(LifecycleArchetype::Appliance);
11886 ws.annotations
11887 .insert(REQUIRES_TAINT_ANNOTATION.into(), PUBLIC_IP_TAINT.into());
11888
11889 // Node without the taint: rejected.
11890 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
11891 "no-pip",
11892 2048,
11893 2000,
11894 vec![],
11895 )]);
11896 assert!(cfg.admit_workload(&ws).is_err());
11897
11898 // Node with the taint: accepted.
11899 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
11900 "pub-node",
11901 2048,
11902 2000,
11903 vec!["public-ip"],
11904 )]);
11905 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "pub-node");
11906 }
11907
11908 #[test]
11909 fn w244_fleet_scenario_appliance_rejected_from_south_and_west002() {
11910 // Full W244 fleet table scenario:
11911 // us-west-001/east-001: no taints, large capacity → appliance lands here
11912 // us-south-001: no-appliance taint → appliance rejected
11913 // us-west-002: no-server, no-appliance → appliance rejected
11914 let cfg = make_empty_cfg(vec![
11915 make_machine_with_capacity("us-south-001", 512, 1000, vec!["no-appliance"]),
11916 make_machine_with_capacity(
11917 "us-west-002",
11918 16384,
11919 8000,
11920 vec!["no-server", "no-appliance"],
11921 ),
11922 make_machine_with_capacity("us-west-001", 4096, 4000, vec![]),
11923 ]);
11924 let ws = appliance_spec_ws(256, 500);
11925 // Skips south (no-appliance) and west-002 (no-appliance), lands on west-001.
11926 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "us-west-001");
11927 }
11928
11929 #[test]
11930 fn w244_fleet_scenario_job_lands_on_west002_first() {
11931 use workload_spec::LifecycleArchetype;
11932 // Jobs should prefer (or at least land on) the job-only box.
11933 let cfg = make_empty_cfg(vec![
11934 make_machine_with_capacity("us-west-001", 4096, 4000, vec![]),
11935 make_machine_with_capacity(
11936 "us-west-002",
11937 16384,
11938 8000,
11939 vec!["no-server", "no-appliance"],
11940 ),
11941 ]);
11942 let mut ws = server_spec(256, 500);
11943 ws.archetype = Some(LifecycleArchetype::Job);
11944 // No fleet node declares `no-job`, so a Job is repelled by nothing;
11945 // west-001 comes first in declaration order (greedy, no preference),
11946 // which is the expected tie-break. Note this is *absence of a repel
11947 // key*, not toleration — no workload can tolerate a taint (W305).
11948 let picked = cfg.admit_workload(&ws).unwrap();
11949 // Both are eligible.
11950 assert!(
11951 picked.name == "us-west-001" || picked.name == "us-west-002",
11952 "job must land on an eligible node, got {}",
11953 picked.name
11954 );
11955 }
11956
11957 #[test]
11958 fn r569_f4_macos_node_taints_keep_cloud_critical_off_but_admit_build_jobs() {
11959 use workload_spec::LifecycleArchetype;
11960 // R569-F4: the headless M2 (us-west-015) joins the fleet as a
11961 // build-worker but must never take cloud-critical load. It carries the
11962 // same repel set as the x86 build-worker (`no-server, no-appliance` —
11963 // see .yah/infra/machines/us-west-015.toml). This pins that intent:
11964 // with a plain cloud node available beside the Mac, every
11965 // cloud-critical archetype lands on the cloud node and never the Mac;
11966 // build Jobs (the Mac's actual purpose) remain eligible on it.
11967 //
11968 // R742-T4: `no-voter` used to sit in this set and in the TOML. It was
11969 // never read here — there is no "voter" workload archetype — and
11970 // R569-F3's learner-only join is what actually keeps the box out of
11971 // quorum. It is now rejected by `yah cloud validate` as inert.
11972 let mac_taints = vec!["no-server", "no-appliance"];
11973 let fleet = || {
11974 make_empty_cfg(vec![
11975 make_machine_with_capacity("us-west-015", 24576, 8000, mac_taints.clone()),
11976 make_machine_with_capacity("us-west-001", 4096, 4000, vec![]),
11977 ])
11978 };
11979
11980 // A cloud-critical Server workload is repelled from the Mac and lands
11981 // on the untainted cloud node.
11982 let cfg = fleet();
11983 assert_eq!(
11984 cfg.admit_workload(&server_spec(256, 500)).unwrap().name,
11985 "us-west-001",
11986 "a Server workload must never land on the no-server Mac node"
11987 );
11988
11989 // Same for an Appliance (pinned/stateful cloud-critical) workload.
11990 let cfg = fleet();
11991 assert_eq!(
11992 cfg.admit_workload(&appliance_spec_ws(256, 500))
11993 .unwrap()
11994 .name,
11995 "us-west-001",
11996 "an Appliance workload must never land on the no-appliance Mac node"
11997 );
11998
11999 // Sharpest repulsion proof: with ONLY the Mac in the fleet, a
12000 // cloud-critical Server workload is rejected outright — the taint keeps
12001 // it off even when that means nowhere to run.
12002 let mac_only = make_empty_cfg(vec![make_machine_with_capacity(
12003 "us-west-015",
12004 24576,
12005 8000,
12006 mac_taints.clone(),
12007 )]);
12008 assert!(
12009 mac_only.admit_workload(&server_spec(256, 500)).is_err(),
12010 "a Server workload must be repelled from a Mac-only fleet, not admitted"
12011 );
12012
12013 // But the Mac's real job — build/forge workloads — IS admitted on it:
12014 // it tolerates every fleet taint (there is no `no-job`).
12015 let mut job = server_spec(256, 500);
12016 job.archetype = Some(LifecycleArchetype::Job);
12017 assert_eq!(
12018 mac_only.admit_workload(&job).unwrap().name,
12019 "us-west-015",
12020 "a build Job must still be admitted on the Mac build-worker"
12021 );
12022 }
12023
12024 // ─── R615-F1: linked infra sources (`.yah/infra/sources.toml`) ─────────
12025
12026 #[test]
12027 fn sources_load_is_empty_when_the_file_is_absent() {
12028 // "Every camp without linked infra has none" — which today is every
12029 // camp — must not be an error.
12030 let tmp = tempfile::TempDir::new().unwrap();
12031 let cfg = SourcesConfig::load(tmp.path()).unwrap();
12032 assert_eq!(cfg, SourcesConfig::default());
12033 assert!(cfg.source.is_empty());
12034 assert_eq!(cfg.schema_version, 1);
12035 }
12036
12037 #[test]
12038 fn sources_parses_a_path_kind_exactly_like_w274s_example() {
12039 let tmp = tempfile::TempDir::new().unwrap();
12040 std::fs::write(
12041 tmp.path().join("sources.toml"),
12042 r#"
12043schema_version = 1
12044
12045[[source]]
12046owner = "yah"
12047kind = "path"
12048path = "../yah"
12049mode = "read-only"
12050"#,
12051 )
12052 .unwrap();
12053 let cfg = SourcesConfig::load(tmp.path()).unwrap();
12054 assert_eq!(cfg.source.len(), 1);
12055 let s = &cfg.source[0];
12056 assert_eq!(s.owner, "yah");
12057 assert_eq!(s.mode, SourceMode::ReadOnly);
12058 assert!(s.select.is_empty());
12059 match &s.kind {
12060 InfraSourceKind::Path { path } => assert_eq!(path, "../yah"),
12061 other => panic!("expected Path, got {other:?}"),
12062 }
12063 }
12064
12065 #[test]
12066 fn sources_parses_a_git_kind_reusing_gitsource_verbatim() {
12067 let tmp = tempfile::TempDir::new().unwrap();
12068 std::fs::write(
12069 tmp.path().join("sources.toml"),
12070 r#"
12071schema_version = 1
12072
12073[[source]]
12074owner = "yah"
12075kind = "git"
12076repo = "git@github.com:yah-ai/infra.git"
12077ref = "main"
12078subdir = "infra"
12079select = ["tag:cloud-runner"]
12080mode = "read-only"
12081"#,
12082 )
12083 .unwrap();
12084 let cfg = SourcesConfig::load(tmp.path()).unwrap();
12085 assert_eq!(cfg.source.len(), 1);
12086 let s = &cfg.source[0];
12087 assert_eq!(s.select, vec!["tag:cloud-runner".to_string()]);
12088 match &s.kind {
12089 InfraSourceKind::Git(git) => {
12090 assert_eq!(git.repo, "git@github.com:yah-ai/infra.git");
12091 assert_eq!(git.r#ref, "main");
12092 assert_eq!(git.subdir.as_deref(), Some("infra"));
12093 }
12094 other => panic!("expected Git, got {other:?}"),
12095 }
12096 }
12097
12098 #[test]
12099 fn sources_mode_defaults_to_read_only_and_manage_is_explicit() {
12100 let tmp = tempfile::TempDir::new().unwrap();
12101 std::fs::write(
12102 tmp.path().join("sources.toml"),
12103 r#"
12104schema_version = 1
12105
12106[[source]]
12107owner = "a"
12108kind = "path"
12109path = "../a"
12110
12111[[source]]
12112owner = "b"
12113kind = "path"
12114path = "../b"
12115mode = "manage"
12116"#,
12117 )
12118 .unwrap();
12119 let cfg = SourcesConfig::load(tmp.path()).unwrap();
12120 assert_eq!(cfg.source[0].mode, SourceMode::ReadOnly, "omitted mode = read-only");
12121 assert_eq!(cfg.source[1].mode, SourceMode::Manage);
12122 }
12123
12124 #[test]
12125 fn sources_preserves_declaration_order() {
12126 // Overlay order matters (R615-F2) when two sources name the same
12127 // machine — the list must round-trip in file order, not be reordered
12128 // by owner or kind.
12129 let tmp = tempfile::TempDir::new().unwrap();
12130 std::fs::write(
12131 tmp.path().join("sources.toml"),
12132 r#"
12133schema_version = 1
12134
12135[[source]]
12136owner = "second"
12137kind = "path"
12138path = "../second"
12139
12140[[source]]
12141owner = "first"
12142kind = "path"
12143path = "../first"
12144"#,
12145 )
12146 .unwrap();
12147 let cfg = SourcesConfig::load(tmp.path()).unwrap();
12148 let owners: Vec<&str> = cfg.source.iter().map(|s| s.owner.as_str()).collect();
12149 assert_eq!(owners, vec!["second", "first"]);
12150 }
12151
12152 #[test]
12153 fn sources_round_trips_through_serialize() {
12154 let cfg = SourcesConfig {
12155 schema_version: 1,
12156 source: vec![
12157 InfraSource {
12158 owner: "yah".into(),
12159 kind: InfraSourceKind::Path {
12160 path: "../yah".into(),
12161 },
12162 mode: SourceMode::ReadOnly,
12163 select: vec![],
12164 },
12165 InfraSource {
12166 owner: "yah".into(),
12167 kind: InfraSourceKind::Git(GitSource {
12168 repo: "git@github.com:yah-ai/infra.git".into(),
12169 r#ref: "main".into(),
12170 subdir: Some("infra".into()),
12171 }),
12172 mode: SourceMode::Manage,
12173 select: vec!["tag:cloud-runner".into()],
12174 },
12175 ],
12176 };
12177 let toml_str = toml::to_string_pretty(&cfg).unwrap();
12178 let reloaded: SourcesConfig = toml::from_str(&toml_str).unwrap();
12179 assert_eq!(reloaded, cfg, "round-trip through TOML must be lossless:\n{toml_str}");
12180 }
12181
12182 // ─── R615-F2: overlay loader in CloudConfig::load ───────────────────────
12183
12184 fn write_min_machine(dir: &Path, name: &str, extra_toml: &str) {
12185 std::fs::create_dir_all(dir).unwrap();
12186 // `extra_toml` supplies `mesh_tags` when the caller cares about it;
12187 // otherwise default to the empty list. Never hardcode `mesh_tags`
12188 // here as well as in `extra_toml` -- TOML rejects a duplicate key.
12189 let mesh_tags = if extra_toml.contains("mesh_tags") {
12190 String::new()
12191 } else {
12192 "mesh_tags = []\n".to_string()
12193 };
12194 std::fs::write(
12195 dir.join(format!("{name}.toml")),
12196 format!("name = \"{name}\"\nprovider = \"static\"\n{mesh_tags}{extra_toml}"),
12197 )
12198 .unwrap();
12199 }
12200
12201 fn write_min_provider(dir: &Path, id: &str) {
12202 std::fs::create_dir_all(dir).unwrap();
12203 std::fs::write(
12204 dir.join(format!("{id}.toml")),
12205 format!("schema_version = 1\nid = \"{id}\"\nkind = \"static\"\n"),
12206 )
12207 .unwrap();
12208 }
12209
12210 fn write_sources_toml(camp_root: &Path, body: &str) {
12211 let dir = camp_root.join(".yah/infra");
12212 std::fs::create_dir_all(&dir).unwrap();
12213 std::fs::write(dir.join("sources.toml"), body).unwrap();
12214 }
12215
12216 #[test]
12217 fn load_with_no_sources_toml_is_unchanged() {
12218 let tmp = tempfile::TempDir::new().unwrap();
12219 write_min_machine(&tmp.path().join(".yah/infra/machines"), "local-1", "");
12220 let cfg = CloudConfig::load(tmp.path()).unwrap();
12221 assert_eq!(cfg.machines.len(), 1);
12222 assert!(cfg.machine_origins.is_empty());
12223 assert!(cfg.provider_origins.is_empty());
12224 }
12225
12226 #[test]
12227 fn path_source_overlays_machines_and_providers_tagged_with_origin() {
12228 let camp = tempfile::TempDir::new().unwrap();
12229 let other = tempfile::TempDir::new().unwrap();
12230 write_min_machine(&other.path().join(".yah/infra/machines"), "borrowed-1", "");
12231 write_min_provider(&other.path().join(".yah/infra/providers"), "borrowed-provider");
12232 write_sources_toml(
12233 camp.path(),
12234 &format!(
12235 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
12236 other.path().display()
12237 ),
12238 );
12239
12240 let cfg = CloudConfig::load(camp.path()).unwrap();
12241 assert_eq!(cfg.machines.len(), 1);
12242 assert_eq!(cfg.machines[0].name, "borrowed-1");
12243 assert_eq!(cfg.providers.len(), 1);
12244 assert_eq!(cfg.providers[0].id, "borrowed-provider");
12245
12246 let origin = cfg.machine_origins.get("borrowed-1").expect("origin recorded");
12247 assert_eq!(origin.owner, "other");
12248 assert_eq!(origin.mode, SourceMode::ReadOnly);
12249 assert!(origin.source.starts_with("path:"));
12250 assert_eq!(
12251 cfg.provider_origins.get("borrowed-provider").unwrap().owner,
12252 "other"
12253 );
12254 }
12255
12256 #[test]
12257 fn camp_local_wins_on_name_collision_and_carries_no_origin() {
12258 let camp = tempfile::TempDir::new().unwrap();
12259 let other = tempfile::TempDir::new().unwrap();
12260 // Both declare a machine named "shared" -- camp-local's copy must win,
12261 // and it must never gain an origin tag.
12262 write_min_machine(&camp.path().join(".yah/infra/machines"), "shared", "");
12263 write_min_machine(
12264 &other.path().join(".yah/infra/machines"),
12265 "shared",
12266 "nickname = \"the borrowed one\"\n",
12267 );
12268 write_sources_toml(
12269 camp.path(),
12270 &format!(
12271 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
12272 other.path().display()
12273 ),
12274 );
12275
12276 let cfg = CloudConfig::load(camp.path()).unwrap();
12277 assert_eq!(cfg.machines.len(), 1, "the name collides, so exactly one entry");
12278 assert_eq!(cfg.machines[0].nickname, None, "camp-local's copy, not the borrowed one");
12279 assert!(
12280 !cfg.machine_origins.contains_key("shared"),
12281 "camp-local entries never carry an origin tag"
12282 );
12283 }
12284
12285 #[test]
12286 fn an_earlier_source_wins_over_a_later_one_on_collision() {
12287 let camp = tempfile::TempDir::new().unwrap();
12288 let first = tempfile::TempDir::new().unwrap();
12289 let second = tempfile::TempDir::new().unwrap();
12290 write_min_machine(&first.path().join(".yah/infra/machines"), "dup", "");
12291 write_min_machine(&second.path().join(".yah/infra/machines"), "dup", "");
12292 write_sources_toml(
12293 camp.path(),
12294 &format!(
12295 "schema_version = 1\n\n[[source]]\nowner = \"first\"\nkind = \"path\"\npath = \"{}\"\n\n[[source]]\nowner = \"second\"\nkind = \"path\"\npath = \"{}\"\n",
12296 first.path().display(),
12297 second.path().display()
12298 ),
12299 );
12300
12301 let cfg = CloudConfig::load(camp.path()).unwrap();
12302 assert_eq!(cfg.machines.len(), 1);
12303 assert_eq!(cfg.machine_origins.get("dup").unwrap().owner, "first");
12304 }
12305
12306 #[test]
12307 fn select_filters_borrowed_machines_by_name_or_mesh_tag() {
12308 let camp = tempfile::TempDir::new().unwrap();
12309 let other = tempfile::TempDir::new().unwrap();
12310 write_min_machine(&other.path().join(".yah/infra/machines"), "runner-1", "mesh_tags = [\"tag:cloud-runner\"]\n");
12311 write_min_machine(&other.path().join(".yah/infra/machines"), "excluded-1", "");
12312 write_sources_toml(
12313 camp.path(),
12314 &format!(
12315 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\nselect = [\"tag:cloud-runner\"]\n",
12316 other.path().display()
12317 ),
12318 );
12319
12320 let cfg = CloudConfig::load(camp.path()).unwrap();
12321 assert_eq!(cfg.machines.len(), 1);
12322 assert_eq!(cfg.machines[0].name, "runner-1");
12323 }
12324
12325 #[test]
12326 fn one_unparseable_foreign_machine_does_not_sink_the_rest_of_the_directory_or_the_load() {
12327 let camp = tempfile::TempDir::new().unwrap();
12328 let other = tempfile::TempDir::new().unwrap();
12329 let dir = other.path().join(".yah/infra/machines");
12330 write_min_machine(&dir, "good", "");
12331 // Schema-skew gotcha: a foreign machine this binary's MachineConfig
12332 // can't parse at all (not just an unknown field -- MachineConfig has
12333 // no deny_unknown_fields, so this has to fail on a TYPE, not a name).
12334 std::fs::write(dir.join("bad.toml"), "name = 1\nprovider = 2\n").unwrap();
12335 write_sources_toml(
12336 camp.path(),
12337 &format!(
12338 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
12339 other.path().display()
12340 ),
12341 );
12342
12343 // Must not error at all -- camp-local load must never fail because a
12344 // source it doesn't own has one bad file.
12345 let cfg = CloudConfig::load(camp.path()).unwrap();
12346 assert_eq!(cfg.machines.len(), 1, "the good entry still loads");
12347 assert_eq!(cfg.machines[0].name, "good");
12348 }
12349
12350 #[test]
12351 fn an_unsynced_git_source_overlays_nothing_and_is_not_an_error() {
12352 // No `yah infra sync` (R615-T3) has ever run, so the cache dir this
12353 // resolves to doesn't exist. Must be silent, not fatal.
12354 let camp = tempfile::TempDir::new().unwrap();
12355 write_sources_toml(
12356 camp.path(),
12357 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"git\"\nrepo = \"git@github.com:yah-ai/infra.git\"\nref = \"main\"\n",
12358 );
12359 let cfg = CloudConfig::load(camp.path()).unwrap();
12360 assert!(cfg.machines.is_empty());
12361 assert!(cfg.machine_origins.is_empty());
12362 }
12363
12364 #[test]
12365 fn a_synced_git_source_reads_from_the_cache_dir_not_the_repo_path() {
12366 // No `subdir` declared -- the checkout ROOT is the infra root.
12367 let camp = tempfile::TempDir::new().unwrap();
12368 let cache = crate::paths::infra_source_cache_dir(camp.path(), "yah");
12369 write_min_machine(&cache.join("machines"), "synced-1", "");
12370 write_sources_toml(
12371 camp.path(),
12372 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"git\"\nrepo = \"git@github.com:yah-ai/infra.git\"\nref = \"main\"\n",
12373 );
12374 let cfg = CloudConfig::load(camp.path()).unwrap();
12375 assert_eq!(cfg.machines.len(), 1);
12376 assert_eq!(cfg.machines[0].name, "synced-1");
12377 assert!(cfg.machine_origins.get("synced-1").unwrap().source.starts_with("git:"));
12378 }
12379
12380 #[test]
12381 fn a_git_sources_subdir_is_honoured_like_the_component_case() {
12382 // W274's own example declares `subdir = "infra"` for a monorepo whose
12383 // registry lives under a subdirectory of the clone rather than at its
12384 // root -- prove `infra_root` actually reads it, not just `.subdir` on
12385 // GitSource parsing (R615-F1 already covers that half).
12386 let camp = tempfile::TempDir::new().unwrap();
12387 let cache = crate::paths::infra_source_cache_dir(camp.path(), "yah");
12388 write_min_machine(&cache.join("infra").join("machines"), "subdir-1", "");
12389 // Also plant a decoy at the checkout root to prove the root itself is
12390 // NOT read when a subdir is declared.
12391 write_min_machine(&cache.join("machines"), "root-decoy", "");
12392 write_sources_toml(
12393 camp.path(),
12394 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"git\"\nrepo = \"git@github.com:yah-ai/infra.git\"\nref = \"main\"\nsubdir = \"infra\"\n",
12395 );
12396 let cfg = CloudConfig::load(camp.path()).unwrap();
12397 assert_eq!(cfg.machines.len(), 1);
12398 assert_eq!(cfg.machines[0].name, "subdir-1");
12399 }
12400
12401 #[test]
12402 fn load_from_config_dir_never_applies_sources_overlay() {
12403 // R615-F2's explicit decision: multi-root sibling trees don't inherit
12404 // the classic .yah/infra/sources.toml. Prove it rather than assert it
12405 // silently -- a sources.toml sitting at workspace_root/.yah/infra/
12406 // must NOT leak into a load_from_config_dir call even though both
12407 // share the same workspace_root.
12408 let camp = tempfile::TempDir::new().unwrap();
12409 let other = tempfile::TempDir::new().unwrap();
12410 write_min_machine(&other.path().join(".yah/infra/machines"), "borrowed-1", "");
12411 write_sources_toml(
12412 camp.path(),
12413 &format!(
12414 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
12415 other.path().display()
12416 ),
12417 );
12418 let sibling_config_dir = camp.path().join(".noisetable");
12419 std::fs::create_dir_all(&sibling_config_dir).unwrap();
12420
12421 let cfg = CloudConfig::load_from_config_dir(&sibling_config_dir, camp.path()).unwrap();
12422 assert!(cfg.machines.is_empty(), "sources.toml must not apply here");
12423 assert!(cfg.machine_origins.is_empty());
12424 }
12425
12426 // ─── R615-T5: `inherit_machines` retirement — cutover proof ────────────
12427
12428 /// The successor to R615-T5's parity proof. That earlier pair of tests
12429 /// asserted the legacy `[infra].inherit_machines` redirect and an
12430 /// equivalent `kind = "path"` source resolved the same machine set, and
12431 /// that the two coexisted without duplicating rows. Both claims were about
12432 /// a mechanism that no longer exists, so they retired with it — what has
12433 /// to hold *now* is the other half of the same guarantee: a camp that
12434 /// declares only `sources.toml` resolves the shared root exactly as the
12435 /// redirect used to, and a stale `inherit_machines` key left behind in
12436 /// `camp.toml` changes nothing.
12437 ///
12438 /// That stale-key case is not hypothetical: it is precisely the state a
12439 /// camp is in between the code cutover and someone tidying its
12440 /// `camp.toml`, and a silent re-resolution there would double-count the
12441 /// borrowed nodes or hide their origin badge.
12442 #[test]
12443 fn a_stale_inherit_machines_key_does_not_change_what_sources_toml_resolves() {
12444 let shared = tempfile::TempDir::new().unwrap();
12445 write_min_machine(&shared.path().join(".yah/infra/machines"), "shared-node-1", "");
12446 write_min_machine(&shared.path().join(".yah/infra/machines"), "shared-node-2", "");
12447
12448 let sources_toml = format!(
12449 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"path\"\npath = \"{}\"\nmode = \"read-only\"\n",
12450 shared.path().display()
12451 );
12452
12453 // Camp A: migrated cleanly — sources.toml only.
12454 let clean = tempfile::TempDir::new().unwrap();
12455 write_sources_toml(clean.path(), &sources_toml);
12456
12457 // Camp B: mid-migration — same source, plus the retired key still
12458 // sitting in camp.toml pointing at the same root.
12459 let stale = tempfile::TempDir::new().unwrap();
12460 std::fs::create_dir_all(stale.path().join(".yah")).unwrap();
12461 std::fs::write(
12462 stale.path().join(".yah/camp.toml"),
12463 format!(
12464 "[infra]\ninherit_machines = \"{}\"\n",
12465 shared.path().display()
12466 ),
12467 )
12468 .unwrap();
12469 write_sources_toml(stale.path(), &sources_toml);
12470
12471 let via_clean = CloudConfig::load(clean.path()).unwrap();
12472 let via_stale = CloudConfig::load(stale.path()).unwrap();
12473
12474 let names = |cfg: &CloudConfig| {
12475 let mut v: Vec<String> = cfg.machines.iter().map(|m| m.name.clone()).collect();
12476 v.sort();
12477 v
12478 };
12479 assert_eq!(
12480 names(&via_clean),
12481 names(&via_stale),
12482 "a leftover inherit_machines key must be inert — the retired redirect is gone"
12483 );
12484 assert_eq!(names(&via_clean), vec!["shared-node-1", "shared-node-2"]);
12485
12486 // And both are *borrowed*, not camp-local. This is the operator-facing
12487 // win the stopgap could never deliver: under the old redirect these
12488 // resolved with no origin at all, indistinguishable from locally-owned
12489 // nodes.
12490 assert_eq!(via_clean.machine_origins.len(), 2);
12491 assert_eq!(via_stale.machine_origins.len(), 2);
12492 for origin in via_stale.machine_origins.values() {
12493 assert_eq!(origin.owner, "yah");
12494 assert_eq!(origin.mode, SourceMode::ReadOnly);
12495 }
12496 }
12497
12498 // ─── R860-T4 (W338): placement groups ───────────────────────────────────
12499
12500 /// One requirement edge, written the way a spec author writes it.
12501 fn requirement(ident: &str, locality: Locality) -> workload_spec::Requirement {
12502 workload_spec::Requirement {
12503 ident: workload_spec::MeshIdent(ident.into()),
12504 locality,
12505 supply: workload_spec::Supply::Wait,
12506 provides: None,
12507 }
12508 }
12509
12510 /// A `minimal_spec` (256 MiB / 250 millicores, Server by inference) that
12511 /// requires the given edges.
12512 fn spec_requiring(name: &str, requires: Vec<workload_spec::Requirement>) -> WorkloadSpec {
12513 WorkloadSpec {
12514 requires,
12515 ..minimal_spec(name, 1)
12516 }
12517 }
12518
12519 /// The declared inventory an ident is resolved against — `.yah/infra/workloads/`.
12520 fn declared(specs: Vec<WorkloadSpec>) -> Vec<WorkloadConfig> {
12521 specs.into_iter().map(|spec| WorkloadConfig { spec }).collect()
12522 }
12523
12524 fn member_names(group: &[WorkloadSpec]) -> Vec<&str> {
12525 group.iter().map(|s| s.name.as_str()).collect()
12526 }
12527
12528 /// The headline case: `local` means "same node", so the two specs are one
12529 /// placement unit and admission has to reason about both.
12530 #[test]
12531 fn a_local_edge_binds_the_provider_into_the_placement_group() {
12532 let inventory = declared(vec![minimal_spec("headscale-replicator", 1)]);
12533 let requirer = spec_requiring(
12534 "headscale",
12535 vec![requirement("headscale-replicator", Locality::Local)],
12536 );
12537
12538 assert_eq!(
12539 member_names(&placement_group(&requirer, &inventory)),
12540 vec!["headscale", "headscale-replicator"]
12541 );
12542 }
12543
12544 /// The edge that must NOT bind. `prefer-local` "never blocks placement"
12545 /// (W338's locality table), and `anywhere` — which is what every legacy
12546 /// `depends_on` folds into — is an ordinary service dependency. Binding
12547 /// either would silently make every dependency in the tree a co-scheduling
12548 /// constraint and start summing unrelated workloads into the capacity floor.
12549 #[test]
12550 fn prefer_local_and_anywhere_edges_do_not_bind_the_group() {
12551 let inventory = declared(vec![
12552 minimal_spec("headscale-db", 1),
12553 minimal_spec("metrics", 1),
12554 minimal_spec("legacy-dep", 1),
12555 ]);
12556
12557 let requirer = WorkloadSpec {
12558 depends_on: vec![workload_spec::MeshIdent("legacy-dep".into())],
12559 ..spec_requiring(
12560 "headscale",
12561 vec![
12562 requirement("headscale-db", Locality::PreferLocal),
12563 requirement("metrics", Locality::Anywhere),
12564 ],
12565 )
12566 };
12567
12568 assert_eq!(
12569 member_names(&placement_group(&requirer, &inventory)),
12570 vec!["headscale"]
12571 );
12572 }
12573
12574 /// Transitive, and via the inline spec a `supply = "self"` requirement
12575 /// carries rather than via an ident lookup — the sidecar shape W338's
12576 /// worked example is built on.
12577 #[test]
12578 fn the_group_is_the_transitive_closure_and_traverses_inline_provides() {
12579 let inline = workload_spec::Requirement {
12580 supply: workload_spec::Supply::SelfProvision,
12581 provides: Some(Box::new(minimal_spec("headscale-restore", 1))),
12582 ..requirement("headscale-restore", Locality::Local)
12583 };
12584 let middle = WorkloadSpec {
12585 requires: vec![requirement("wal-shipper", Locality::Local)],
12586 ..minimal_spec("headscale-replicator", 1)
12587 };
12588 let inventory = declared(vec![middle, minimal_spec("wal-shipper", 1)]);
12589
12590 let requirer = spec_requiring(
12591 "headscale",
12592 vec![
12593 inline,
12594 requirement("headscale-replicator", Locality::Local),
12595 ],
12596 );
12597
12598 assert_eq!(
12599 member_names(&placement_group(&requirer, &inventory)),
12600 vec![
12601 "headscale",
12602 "headscale-restore",
12603 "headscale-replicator",
12604 "wal-shipper"
12605 ]
12606 );
12607 }
12608
12609 /// `validate::check_requires` bounds `provides` nesting to depth 1 but
12610 /// cannot stop two separately-declared specs from naming each other. Without
12611 /// the visited set this closure never terminates, so admission would hang
12612 /// rather than refuse — the worst failure shape for a deploy gate.
12613 #[test]
12614 fn an_ident_cycle_closes_the_group_instead_of_looping_forever() {
12615 let b = spec_requiring("b", vec![requirement("a", Locality::Local)]);
12616 let a = spec_requiring("a", vec![requirement("b", Locality::Local)]);
12617 let inventory = declared(vec![a.clone(), b]);
12618
12619 assert_eq!(member_names(&placement_group(&a, &inventory)), vec!["a", "b"]);
12620 }
12621
12622 /// An unresolvable ident is skipped, not fatal: admission is a pure function
12623 /// of the declared inventory, and refusing every deploy whose provider is
12624 /// not yet declared would make `requires` unusable before R860-T6 lands.
12625 #[test]
12626 fn an_unresolvable_local_ident_is_skipped_rather_than_refused() {
12627 let requirer = spec_requiring("headscale", vec![requirement("not-declared", Locality::Local)]);
12628 assert_eq!(
12629 member_names(&placement_group(&requirer, &[])),
12630 vec!["headscale"]
12631 );
12632 }
12633
12634 /// W338 §Placement consequences 1: the capacity floor is the group's sum.
12635 /// A node that fits the requirer alone must refuse the group — placing it
12636 /// there would oversubscribe the node the moment the provider follows.
12637 #[test]
12638 fn the_capacity_floor_is_the_sum_of_the_group_not_the_requirer_alone() {
12639 let provider = minimal_spec("headscale-replicator", 1);
12640 let requirer = spec_requiring(
12641 "headscale",
12642 vec![requirement("headscale-replicator", Locality::Local)],
12643 );
12644 // Two `minimal_spec`s: 256 MiB + 250 millicores each.
12645 let inventory = declared(vec![provider]);
12646
12647 let too_small = CloudConfig {
12648 workloads: inventory.clone(),
12649 ..make_empty_cfg(vec![make_machine_with_capacity("small", 300, 4000, vec![])])
12650 };
12651 let err = too_small.admit_workload(&requirer).unwrap_err().to_string();
12652 assert!(
12653 err.contains("memory_mb>=512"),
12654 "the floor must name the group's summed demand, got: {err}"
12655 );
12656
12657 let big_enough = CloudConfig {
12658 workloads: inventory,
12659 ..make_empty_cfg(vec![make_machine_with_capacity("roomy", 512, 4000, vec![])])
12660 };
12661 assert_eq!(
12662 big_enough.admit_workload(&requirer).unwrap().name,
12663 "roomy",
12664 "a node covering the sum must still admit the group"
12665 );
12666 }
12667
12668 /// W338 §Placement consequences 2, and the reason repulsion is computed over
12669 /// a set at all: the requirer is a `Server`, so the pre-R860 axis would have
12670 /// let it onto a `no-appliance` dev Pi and dragged its Appliance provider
12671 /// there with it.
12672 #[test]
12673 fn a_server_requiring_an_appliance_locally_is_repelled_by_no_appliance() {
12674 let appliance = WorkloadSpec {
12675 archetype: Some(LifecycleArchetype::Appliance),
12676 ..minimal_spec("headscale", 1)
12677 };
12678 let requirer = spec_requiring("headscale-ui", vec![requirement("headscale", Locality::Local)]);
12679 assert_eq!(
12680 requirer.effective_archetype(),
12681 LifecycleArchetype::Server,
12682 "precondition: the requirer itself must not be an Appliance"
12683 );
12684
12685 let cfg = CloudConfig {
12686 workloads: declared(vec![appliance]),
12687 ..make_empty_cfg(vec![
12688 make_machine_with_capacity("dev-pi", 8192, 4000, vec!["no-appliance"]),
12689 make_machine_with_capacity("us-west-001", 8192, 4000, vec![]),
12690 ])
12691 };
12692
12693 assert_eq!(
12694 cfg.admit_workload(&requirer).unwrap().name,
12695 "us-west-001",
12696 "the dev Pi repels the group's Appliance member"
12697 );
12698
12699 // And with the Appliance gone from the group, the same requirer is
12700 // admissible on the same Pi — proving the repulsion came from the edge.
12701 let alone = minimal_spec("headscale-ui", 1);
12702 assert_eq!(cfg.admit_workload(&alone).unwrap().name, "dev-pi");
12703 }
12704
12705 /// The set-valued form of the per-workload drain skip the node makes in
12706 /// `drain_workloads`: one Appliance member pins the whole group.
12707 #[test]
12708 fn a_group_containing_an_appliance_is_not_drainable() {
12709 let server = minimal_spec("headscale-ui", 1);
12710 let appliance = WorkloadSpec {
12711 archetype: Some(LifecycleArchetype::Appliance),
12712 ..minimal_spec("headscale", 1)
12713 };
12714
12715 assert!(group_is_drainable(std::slice::from_ref(&server)));
12716 assert!(!group_is_drainable(&[server, appliance]));
12717 }
12718
12719 /// The regression that matters most: nothing in the tree declares
12720 /// `requires` yet, so every existing spec's group is exactly itself and its
12721 /// admission axes must be bit-identical to the pre-R860 derivation.
12722 #[test]
12723 fn a_spec_with_no_local_edges_admits_exactly_as_it_did_before() {
12724 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
12725 let req = admission_spec(&ws, &[]).unwrap();
12726
12727 assert_eq!(req.mesh_tags, vec!["tag:build-worker", "arch:x86"]);
12728 assert_eq!(req.memory_mb, ws.memory_request_mb());
12729 assert_eq!(req.cpu_millis, ws.resources.cpu_millis);
12730 // R876-B7: the axis is now the complement — every repelling key EXCEPT
12731 // this spec's own class, which is the same predicate stated from the
12732 // other side. Asserted against the derivation rather than a literal so
12733 // it stays true if a fourth archetype is added.
12734 assert_eq!(
12735 req.tolerates,
12736 tolerations_excluding(&[ws.effective_archetype()])
12737 );
12738 let own = format!("no-{}", ws.effective_archetype().taint_key());
12739 assert!(
12740 !req.tolerates.contains(&own),
12741 "a spec never tolerates the taint aimed at its own class"
12742 );
12743 }
12744
12745 // ─── R860-T5 (W338 §Placement consequences 3): native-exec capability ────
12746
12747 /// A `minimal_spec` carrying the `yah.exec = native` marker — the only way
12748 /// a workload says "fork+exec me on the host" (`WorkloadSpec::
12749 /// wants_native_exec`). It stays a Container workload on the wire; the
12750 /// marker is the whole difference.
12751 fn native_spec(name: &str) -> WorkloadSpec {
12752 let mut ws = minimal_spec(name, 1);
12753 ws.annotations.insert(
12754 workload_spec::NATIVE_EXEC_ANNOTATION.to_string(),
12755 workload_spec::NATIVE_EXEC_VALUE.to_string(),
12756 );
12757 assert!(ws.wants_native_exec(), "precondition: the marker must read back");
12758 ws
12759 }
12760
12761 /// The R858 failure, now caught at placement instead of at dispatch: a node
12762 /// whose kamaji has no `--native-exec-dir` accepted the election and then
12763 /// refused the deploy, and nothing upstream could see it coming.
12764 #[test]
12765 fn a_node_without_the_native_exec_capability_cannot_host_a_native_workload() {
12766 let native = native_spec("headscale");
12767
12768 let incapable = make_empty_cfg(vec![make_machine("us-south-001", vec![])]);
12769 let err = incapable.admit_workload(&native).unwrap_err().to_string();
12770 assert!(
12771 err.contains(NATIVE_EXEC_MESH_TAG),
12772 "the refusal must name the missing capability, got: {err}"
12773 );
12774
12775 let capable = make_empty_cfg(vec![
12776 make_machine("us-south-001", vec![]),
12777 make_machine("us-west-001", vec![NATIVE_EXEC_MESH_TAG]),
12778 ]);
12779 assert_eq!(
12780 capable.admit_workload(&native).unwrap().name,
12781 "us-west-001",
12782 "a node declaring the capability admits the native workload"
12783 );
12784 }
12785
12786 /// W338's actual sentence: a `supply = "self"` spec "must be placeable where
12787 /// its requirer lands". The requirer here is an ordinary container workload
12788 /// — it is the *provider* reached by a `local` edge that needs the host
12789 /// backend, so the capability has to be required of the group, not of the
12790 /// spec being deployed.
12791 #[test]
12792 fn a_local_edge_to_a_native_provider_makes_the_requirer_need_the_capability() {
12793 let requirer = spec_requiring(
12794 "headscale-ui",
12795 vec![requirement("headscale", Locality::Local)],
12796 );
12797 assert!(
12798 !requirer.wants_native_exec(),
12799 "precondition: the requirer itself is an ordinary container workload"
12800 );
12801
12802 let cfg = CloudConfig {
12803 workloads: declared(vec![native_spec("headscale")]),
12804 ..make_empty_cfg(vec![
12805 make_machine("plain", vec![]),
12806 make_machine("us-west-001", vec![NATIVE_EXEC_MESH_TAG]),
12807 ])
12808 };
12809
12810 assert_eq!(
12811 cfg.admit_workload(&requirer).unwrap().name,
12812 "us-west-001",
12813 "the group's native member pulls the requirer onto a capable node"
12814 );
12815
12816 // Without the edge the same requirer is admissible on the plain node,
12817 // so the constraint provably came from the group and not from the spec.
12818 let alone = minimal_spec("headscale-ui", 1);
12819 assert_eq!(cfg.admit_workload(&alone).unwrap().name, "plain");
12820 }
12821
12822 /// The regression guard: nothing in the tree is native-marked today, so
12823 /// every existing spec's axes must be untouched by this ticket.
12824 #[test]
12825 fn a_group_with_no_native_member_does_not_require_the_capability() {
12826 let inventory = declared(vec![minimal_spec("headscale-replicator", 1)]);
12827 let requirer = spec_requiring(
12828 "headscale",
12829 vec![requirement("headscale-replicator", Locality::Local)],
12830 );
12831
12832 let req = admission_spec(&requirer, &inventory).unwrap();
12833 assert!(
12834 !req.mesh_tags.iter().any(|t| t == NATIVE_EXEC_MESH_TAG),
12835 "no native member ⇒ no capability axis, got: {:?}",
12836 req.mesh_tags
12837 );
12838
12839 // And it still lands on a node that declares nothing at all.
12840 let cfg = CloudConfig {
12841 workloads: inventory,
12842 ..make_empty_cfg(vec![make_machine("plain", vec![])])
12843 };
12844 assert_eq!(cfg.admit_workload(&requirer).unwrap().name, "plain");
12845 }
12846
12847 // ─── R894-F1: trust declares a minimum isolation substrate ───────────────
12848
12849 /// A `minimal_spec` marked untrusted **without** raising its substrate —
12850 /// i.e. the incoherent pairing this ticket exists to refuse.
12851 ///
12852 /// It deliberately does not go through `WorkloadSpec::stamp_untrusted`,
12853 /// which raises the substrate as it stamps: the point of these tests is the
12854 /// admission-side gate, so the spec has to arrive in the state a buggy or
12855 /// hostile producer would leave it in, not the state the correct producer
12856 /// guarantees.
12857 fn untrusted_spec(name: &str, exec: Option<&str>) -> WorkloadSpec {
12858 let mut ws = minimal_spec(name, 1);
12859 ws.annotations.insert(
12860 workload_spec::TRUST_ANNOTATION.to_string(),
12861 workload_spec::TRUST_UNTRUSTED_VALUE.to_string(),
12862 );
12863 if let Some(v) = exec {
12864 ws.annotations.insert(
12865 workload_spec::NATIVE_EXEC_ANNOTATION.to_string(),
12866 v.to_string(),
12867 );
12868 }
12869 assert_eq!(
12870 ws.trust().unwrap(),
12871 workload_spec::TrustLevel::Untrusted,
12872 "precondition: the trust marker must read back"
12873 );
12874 ws
12875 }
12876
12877 /// THE ACCEPTANCE GATE. Both refusals happen inside `admit_workload` — the
12878 /// real dispatch path `MeshYubabaClient::elect_node` calls — against a fleet
12879 /// that contains a microVM-capable node, so the refusal is provably the
12880 /// trust rule and not "nothing admits this".
12881 #[test]
12882 fn admit_workload_refuses_untrusted_code_on_a_shared_kernel() {
12883 let cfg = make_empty_cfg(vec![
12884 make_machine("plain", vec![]),
12885 make_machine("us-west-003", vec![MICROVM_MESH_TAG]),
12886 ]);
12887
12888 // (1) Untrusted + `yah.exec = native`: the widest possible substrate.
12889 let native = untrusted_spec("tenant-camp", Some(workload_spec::NATIVE_EXEC_VALUE));
12890 let err = cfg.admit_workload(&native).unwrap_err().to_string();
12891 assert!(
12892 err.contains("tenant-camp") && err.contains("native") && err.contains("microvm"),
12893 "the refusal must name the workload, what it asked for and the floor, got: {err}"
12894 );
12895
12896 // (2) Untrusted with no substrate marker at all — the container default.
12897 // This is the one a "absent means trusted for ALL origins" spelling
12898 // would have admitted.
12899 let container = untrusted_spec("tenant-camp", None);
12900 assert_eq!(
12901 container.exec_substrate(),
12902 workload_spec::ExecSubstrate::Container,
12903 "precondition: no marker means the container backend"
12904 );
12905 let err = cfg.admit_workload(&container).unwrap_err().to_string();
12906 assert!(
12907 err.contains("container") && err.contains("microvm"),
12908 "an unmarked untrusted spec is refused for the same reason, got: {err}"
12909 );
12910
12911 // (3) The same spec asking for a microVM is admitted, onto the node that
12912 // declares the capability. Same fleet, same workload name — so (1) and
12913 // (2) provably failed on the substrate and nothing else.
12914 let vm = untrusted_spec("tenant-camp", Some(workload_spec::MICROVM_EXEC_VALUE));
12915 assert_eq!(cfg.admit_workload(&vm).unwrap().name, "us-west-003");
12916 }
12917
12918 /// A caller may always request *more* isolation than it needs. A trusted
12919 /// workload asking for a microVM is not "exceeding" anything — the rule is
12920 /// one-directional.
12921 #[test]
12922 fn a_trusted_workload_may_still_request_a_stricter_substrate() {
12923 let mut ws = minimal_spec("forge-build", 1);
12924 ws.annotations.insert(
12925 workload_spec::NATIVE_EXEC_ANNOTATION.to_string(),
12926 workload_spec::MICROVM_EXEC_VALUE.to_string(),
12927 );
12928 assert_eq!(ws.trust().unwrap(), workload_spec::TrustLevel::Trusted);
12929
12930 let cfg = make_empty_cfg(vec![
12931 make_machine("plain", vec![]),
12932 make_machine("us-west-003", vec![MICROVM_MESH_TAG]),
12933 ]);
12934 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "us-west-003");
12935 }
12936
12937 /// An untrusted member reached by a `local` edge is still on the node, so
12938 /// the group is refused even though the spec being deployed is fine — the
12939 /// same group-wide reasoning every other axis in `admission_spec` uses.
12940 #[test]
12941 fn an_untrusted_provider_refuses_the_whole_placement_group() {
12942 let requirer = spec_requiring(
12943 "camp-front",
12944 vec![requirement("tenant-camp", Locality::Local)],
12945 );
12946 assert_eq!(
12947 requirer.trust().unwrap(),
12948 workload_spec::TrustLevel::Trusted,
12949 "precondition: the requirer itself is an ordinary trusted workload"
12950 );
12951
12952 let cfg = CloudConfig {
12953 workloads: declared(vec![untrusted_spec("tenant-camp", None)]),
12954 ..make_empty_cfg(vec![make_machine("us-west-003", vec![MICROVM_MESH_TAG])])
12955 };
12956 let err = cfg.admit_workload(&requirer).unwrap_err().to_string();
12957 assert!(
12958 err.contains("tenant-camp"),
12959 "the refusal must name the offending MEMBER, not the requirer, got: {err}"
12960 );
12961
12962 // The same requirer without the edge admits fine, so the refusal
12963 // provably came from the group.
12964 let alone = minimal_spec("camp-front", 1);
12965 assert_eq!(cfg.admit_workload(&alone).unwrap().name, "us-west-003");
12966 }
12967
12968 /// A typo in the trust value is refused, not resolved to either side.
12969 /// Reading it as trusted would turn the microVM floor off by misspelling.
12970 #[test]
12971 fn an_unreadable_trust_declaration_is_refused_rather_than_defaulted() {
12972 let mut ws = minimal_spec("tenant-camp", 1);
12973 ws.annotations.insert(
12974 workload_spec::TRUST_ANNOTATION.to_string(),
12975 "untrused".to_string(),
12976 );
12977
12978 let cfg = make_empty_cfg(vec![make_machine("plain", vec![])]);
12979 let err = cfg.admit_workload(&ws).unwrap_err().to_string();
12980 assert!(
12981 err.contains("untrused") && err.contains("tenant-camp"),
12982 "the refusal must quote the unreadable value, got: {err}"
12983 );
12984 }
12985
12986 /// The regression guard for the new mesh-tag axis, in the shape R860-T5's
12987 /// own guard uses: nothing in the fleet is microVM-marked today, so every
12988 /// existing spec's axes must be untouched.
12989 #[test]
12990 fn a_group_with_no_microvm_member_does_not_require_the_capability() {
12991 let ws = minimal_spec("headscale-ui", 1);
12992 let req = admission_spec(&ws, &[]).unwrap();
12993 assert!(
12994 !req.mesh_tags.iter().any(|t| t == MICROVM_MESH_TAG),
12995 "no microVM member ⇒ no capability axis, got: {:?}",
12996 req.mesh_tags
12997 );
12998
12999 // And a microVM-marked spec is refused by a node declaring nothing,
13000 // naming the tag — the dispatch-time surprise R858 paid for.
13001 let mut vm = minimal_spec("forge-build", 1);
13002 vm.annotations.insert(
13003 workload_spec::NATIVE_EXEC_ANNOTATION.to_string(),
13004 workload_spec::MICROVM_EXEC_VALUE.to_string(),
13005 );
13006 let cfg = make_empty_cfg(vec![make_machine("plain", vec![])]);
13007 let err = cfg.admit_workload(&vm).unwrap_err().to_string();
13008 assert!(
13009 err.contains(MICROVM_MESH_TAG),
13010 "the refusal must name the missing capability, got: {err}"
13011 );
13012 }
13013
13014 /// The gate is structural, not a convention followed at three call sites:
13015 /// every admission entry point goes through `admission_spec`, so all three
13016 /// refuse the same spec.
13017 #[test]
13018 fn every_admission_entry_point_enforces_the_trust_floor() {
13019 let ws = untrusted_spec("tenant-camp", Some(workload_spec::NATIVE_EXEC_VALUE));
13020 let mut machine = make_machine("us-west-003", vec![MICROVM_MESH_TAG]);
13021 machine.sovereign_group = Some("home".to_string());
13022 let cfg = make_empty_cfg(vec![machine]);
13023
13024 assert!(cfg.admit_workload(&ws).is_err());
13025 assert!(cfg.admit_workload_candidates(&ws).is_err());
13026 assert!(cfg.admit_workload_in_group(&ws, "home").is_err());
13027 }
13028
13029 // ── R892-B1: a declared key must never be dropped in silence ─────────────
13030
13031 /// The real shape, not a hand-minimised one: a key check that passes on a
13032 /// toy spec and trips on a production file is worse than no check.
13033 /// Modelled on `.yah/infra/workloads/yah-cloud-admin.toml`.
13034 const REAL_WORKLOAD_TOML: &str = r#"
13035name = "yah-cloud-admin"
13036tier = "infra"
13037archetype = "server"
13038replicas = 1
13039restart_policy = "always"
13040
13041[image]
13042registry = "cr.yah.dev"
13043repository = "yah-cloud-admin"
13044tag = "20260903-amd64"
13045digest = "sha256:efaa7824ebf1654b226e4bf9cf4f86f1ce39b17900895b1f9d83e4d987f58733"
13046
13047[resources]
13048memory_mb = 256
13049cpu_millis = 250
13050
13051[annotations]
13052"yah.network" = "host"
13053
13054[expose.mesh]
13055identity = "yah-cloud-admin"
13056ports = [4325]
13057allow_from = []
13058
13059[[env]]
13060name = "YAH_CLOUD_ADMIN_ADDR"
13061value = { literal = { value = "100.64.0.1:4325" } }
13062
13063[[secrets]]
13064source = { cluster = { name = "cheers/cloud-admin/verify-key" } }
13065target = { file = { path = "/run/secrets/cheers-verify.key", mode = 0o400 } }
13066
13067[healthcheck]
13068probe = { http_get = { path = "/__mesofact/health", port = 4325, expect_status = 200 } }
13069interval = 15000
13070timeout = 5000
13071initial_delay = 20000
13072failure_threshold = 3
13073
13074[stop_policy]
13075signal = 15
13076grace_period = 10000
13077"#;
13078
13079 fn check_keys(src: &str) -> Result<()> {
13080 let spec: WorkloadSpec = toml::from_str(src).expect("fixture must parse");
13081 refuse_dropped_keys(src, &spec, "fixture.toml")
13082 }
13083
13084 /// Non-vacuity, and the guard against a check that flags everything: a file
13085 /// whose every key the schema knows must load clean.
13086 #[test]
13087 fn a_workload_file_whose_keys_all_survive_the_parse_is_accepted() {
13088 check_keys(REAL_WORKLOAD_TOML).expect("a fully-known file must not be refused");
13089 }
13090
13091 /// The 2026-09-11 outage, reproduced in one line of TOML: R885-T6 deleted
13092 /// `ResourceLimits::ephemeral_storage_mb`, every camp's file still declared
13093 /// it, and serde discarded it without a word.
13094 #[test]
13095 fn the_key_that_caused_the_outage_is_now_refused_by_name() {
13096 let src = REAL_WORKLOAD_TOML.replace(
13097 "cpu_millis = 250",
13098 "cpu_millis = 250\nephemeral_storage_mb = 256",
13099 );
13100 let err = check_keys(&src).expect_err("a dropped key must refuse the file");
13101 let msg = format!("{err:#}");
13102 assert!(
13103 msg.contains("resources.ephemeral_storage_mb"),
13104 "the refusal must name the key by its full path, got: {msg}"
13105 );
13106 }
13107
13108 /// Nested and top-level keys are both reported, so a misspelling anywhere in
13109 /// the file is as loud as a removed field.
13110 #[test]
13111 fn every_unknown_key_is_named_not_just_the_first() {
13112 let src = format!("{REAL_WORKLOAD_TOML}\nreplica_count = 3\n\n[healthcheck_typo]\nx = 1\n");
13113 let err = check_keys(&src).expect_err("unknown keys must refuse the file");
13114 let msg = format!("{err:#}");
13115 assert!(msg.contains("replica_count"), "got: {msg}");
13116 assert!(msg.contains("healthcheck_typo"), "got: {msg}");
13117 }
13118
13119 /// A value the serialiser re-spells (an enum, a duration) is not a dropped
13120 /// key. Only a missing KEY is, which is what keeps this check from firing on
13121 /// every legitimate file in the fleet.
13122 #[test]
13123 fn a_canonicalised_value_is_not_reported_as_dropped() {
13124 let declared = toml::Value::try_from(toml::toml! {
13125 restart = "always"
13126 [nested]
13127 list = [1, 2]
13128 })
13129 .unwrap();
13130 let kept = toml::Value::try_from(toml::toml! {
13131 restart = "Always"
13132 [nested]
13133 list = [9, 9]
13134 })
13135 .unwrap();
13136 let mut out = Vec::new();
13137 collect_dropped_keys(&declared, &kept, "", &mut out);
13138 assert!(out.is_empty(), "values differ, keys do not: {out:?}");
13139 }
13140
13141 /// R896-B5: a key whose field was deleted as inert loads with a warning
13142 /// instead of refusing the file, so deleting the field does not break every
13143 /// camp whose committed TOML still declares it. Driven off the registry, so
13144 /// the next retired key is covered the moment it is listed.
13145 #[test]
13146 fn every_retired_key_is_ignored_not_refused() {
13147 for retired in workload_spec::RETIRED_KEYS {
13148 let mut declared: toml::Value = toml::from_str(REAL_WORKLOAD_TOML).unwrap();
13149 let (parents, leaf) = match retired.path.rsplit_once('.') {
13150 Some((p, l)) => (p.split('.').collect::<Vec<_>>(), l),
13151 None => (vec![], retired.path),
13152 };
13153 let mut table = declared.as_table_mut().unwrap();
13154 for p in parents {
13155 table = table
13156 .entry(p)
13157 .or_insert_with(|| toml::Value::Table(Default::default()))
13158 .as_table_mut()
13159 .unwrap();
13160 }
13161 table.insert(leaf.into(), toml::Value::Integer(1));
13162 let src = toml::to_string(&declared).unwrap();
13163 check_keys(&src)
13164 .unwrap_or_else(|e| panic!("retired `{}` must not refuse: {e:#}", retired.path));
13165 }
13166 }
13167
13168 /// The registry is an exemption from R892-B1, not a hole in it: a retired
13169 /// key sitting beside an unknown one still refuses the file, naming only the
13170 /// unknown key.
13171 #[test]
13172 fn a_retired_key_does_not_excuse_an_unknown_one() {
13173 let src = format!("schema_version = 1\n{REAL_WORKLOAD_TOML}\nreplica_count = 3\n");
13174 let msg = format!("{:#}", check_keys(&src).expect_err("unknown key must still refuse"));
13175 assert!(msg.contains("replica_count"), "got: {msg}");
13176 assert!(!msg.contains("schema_version"), "retired key must not be named: {msg}");
13177 }
13178}