cloud/config.rs
1//! @yah:ticket(R040-F16, "pg-on-mesh service recipe: bind tailscale0 + pg_hba.conf snippet + ufw rules")
2//! @yah:at(2026-05-05T00:32:34Z)
3//! @yah:assignee(agent:claude)
4//! @yah:status(review)
5//! @yah:parent(R040)
6//! @yah:handoff("Companion to R040-F15. Inter-node TCP (Postgres primary↔replica, NATS clusters, anything raw-protocol) lives on the Headscale mesh, not on Hetzner public IPs. Each node has a stable 100.64.x.x mesh IP that survives replacement of the underlying box, so DNS / config / pg_hba never churn when a CPX-11 is rebuilt. WireGuard already encrypts the wire — TLS becomes defense-in-depth, not load-bearing. This ticket carries the concrete pg-shaped recipe so the first stateful service deploy doesn't have to re-derive the pattern; subsequent services (redis, NATS, etc.) cargo-cult from it.")
7//! @yah:next("ServiceConfig gains a `bind_interface: Option<String>` field (e.g. `Some(\"tailscale0\")` for mesh-only services). The cloud-init/podman compose renderer translates this into either `--network host` + `pg listen_addresses = '<mesh-ip>'` OR a podman macvlan/host-binding pattern that achieves the same.")
8//! @yah:next("Generated pg_hba.conf snippet: allow the mesh subnet (100.64.0.0/10) for replication + app users. Postgres binds to the node's tailscale0 mesh IP only — `listen_addresses` is templated from the node's `tailscale ip --4` at first boot.")
9//! @yah:next("Generated ufw rules: `ufw allow in on tailscale0 to any port 5432; ufw deny 5432` — mirrors the existing yah-yubaba 7443 pattern in mirror.yml. Same shape works for any mesh-only port.")
10//! @yah:next("Replica connection string uses primary's mesh IP, NOT its public IP. Stable across box replacement.")
11//! @yah:next("Out of scope: pg_basebackup orchestration, failover, WAL archiving — those belong in noisetable's domain; this ticket only standardizes the binding/firewall/auth shape so noisetable's pg deployment doesn't reinvent it.")
12//!
13//!
14//! @yah:ticket(R323-F9, "Add sync-wave ordering to ServiceComponent (deploy-panel wave order)")
15//! @yah:assignee(agent:claude)
16//! @yah:at(2026-05-26T15:20:25Z)
17//! @yah:status(review)
18//! @yah:phase(P2)
19//! @yah:parent(R323)
20//! @yah:next("ServiceComponent gains a wave/order field (or depends_on between components) so the deploy panel (R323-F4) can group workload rollout rows into sync waves (wave 0 parallel, wait healthy, wave 1, …). Today all components are implicitly wave 0.")
21//! @yah:next("compute_service/compute_cell in reconciler/sync_status.rs surface the wave per workload so F4 doesn't re-derive it.")
22//! @yah:gotcha("Until this lands, F4 should render every workload as wave 0 (no ordering).")
23//! @yah:handoff("Added wave: u32 (serde default=0, skip_serializing_if zero) to ServiceComponent in config.rs. Added is_zero_u32 helper. Fixed the three struct literal call-sites that now need wave: 0 (config.rs test, local_sim.rs x2, mesofact_static.rs). Added wave?: number to the TS ServiceComponent interface with a doc comment. Deploy panel now reads c.wave ?? 0 for each WorkloadRow instead of hardcoded 0. SyncFooter computes maxWave from the components array and renders 'wave 0' (all-zero case) or 'waves 0–N' (multi-wave). All 218 cloud lib tests pass; bun run typecheck clean.")
24//! @yah:verify("cargo test -p cloud --lib # 218 passed")
25//! @yah:verify("cd packages/yah/ui && bun run typecheck # no new errors")
26//! @yah:verify("In service.toml: add wave = 1 to a component, rebuild, open the deploy panel — that workload row shows 'w1' badge; SyncFooter shows 'waves 0–1'")
27//! @yah:verify("Component with no wave field in TOML deserializes as wave=0 (default). Saving a wave=0 component omits the field from the output TOML (skip_serializing_if).")
28//!
29//! @arch:see(.yah/docs/working/W142-pond.md)
30//!
31//! @yah:relay(R615, "Linked infra sources: sources.toml overlay so a camp can borrow another camp's substrate")
32//! @yah:at(2026-07-20T18:18:05Z)
33//! @yah:status(open)
34//! @arch:see(.yah/docs/working/W274-linked-infra-sources.md)
35//!
36//! @yah:ticket(R615-F1, "InfraSource types + SourcesConfig::load(infra_dir) parsing .yah/infra/sources.toml")
37//! @yah:status(review)
38//! @yah:assignee(agent:bundle-anthropic-miravel)
39//! @yah:at(2026-08-08T19:55:57Z)
40//! @yah:phase(P1)
41//! @yah:parent(R615)
42//! @yah:next("Add InfraSourceKind { Path { path }, Git(GitSource) } + InfraSource { owner, kind, mode, select } to cloud/src/config.rs. Reuse the existing GitSource (config.rs:1205, { repo, ref, subdir }) verbatim — do not invent a second git-source shape.")
43//! @yah:next("SourcesConfig::load(infra_dir) reads .yah/infra/sources.toml (schema_version = 1, ordered [[source]] array). Absent file = empty list, never an error — every existing camp has no sources.toml.")
44//! @yah:next("mode is the write-gate: read-only (borrower cannot mutate) vs owner-manages. Model it as an enum, not a bool, so a future read-write-with-approval tier is additive.")
45//! @yah:verify("cargo check -p cloud && cargo test -p cloud")
46//! @arch:see(.yah/docs/working/W274-linked-infra-sources.md)
47//! @yah:tier(Cleric)
48//! @yah:handoff("InfraSourceKind{Path{path},Git(GitSource)} + SourceMode{ReadOnly,Manage} + InfraSource{owner,kind,mode,select} + SourcesConfig{schema_version,source} all landed in oss/yubaba/crates/cloud/src/config.rs (after default_git_ref, ~line 1550). GitSource reused verbatim -- Git(GitSource) wraps the existing R561 type unchanged, no second git-source shape. InfraSourceKind is internally tagged (#[serde(tag=\"kind\", rename_all=\"kebab-case\")]) and flattened into InfraSource so a [[source]] table reads exactly like W274's example: owner/kind/path-or-repo+ref+subdir/mode/select all at one table level. mode: SourceMode defaults ReadOnly via #[serde(default)] on the field (enum, not bool, per the ticket's own instruction -- Manage is the explicit escape hatch). SourcesConfig::load(infra_dir) returns Ok(default()) -- schema_version=1, empty source list -- when sources.toml is absent; only parses+errors when the file exists and is malformed.")
49//! @yah:handoff("Tree anchor 85801e7f. Pathspec: oss/yubaba/crates/cloud/src/config.rs (only file touched). Tests: cargo test -p yah-cloud --lib (from oss/yubaba) 710 passed / 0 failed / 4 ignored, +6 new over the 704 baseline your R707-T6 verification recorded (sources_load_is_empty_when_the_file_is_absent, sources_parses_a_path_kind_exactly_like_w274s_example, sources_parses_a_git_kind_reusing_gitsource_verbatim, sources_mode_defaults_to_read_only_and_manage_is_explicit, sources_preserves_declaration_order, sources_round_trips_through_serialize). cargo check -p cloud also green (implied by the test build).")
50//! @yah:handoff("Tree anchor at handoff: 85801e7f6b76b369c0c8ecd2e5c7874990cd9286 — the shared tree as I left it. Diff against it (`git diff 85801e7f6b76b369c0c8ecd2e5c7874990cd9286..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
51//! @yah:next("R615-F2 picks this straight up: overlay these sources into CloudConfig::load, tagging origin{owner,source} and merging camp-local-wins-on-collision.")
52//! @yah:handoff("Verified pre-existing work: InfraSourceKind{Path,Git(GitSource)} + SourceMode + InfraSource + SourcesConfig all present in oss/yubaba/crates/cloud/src/config.rs at tree anchor 871fde1c, matching the inline @yah:handoff notes already on this ticket. GitSource reused verbatim, no second git-source shape. This session added no new code -- only ran verification and closed the board state, which a prior session left stuck in `open` despite the work being done (code + handoff notes landed, but board.review/handoff was never called).")
53//! @yah:verify("cargo check -p yah-cloud -- clean (2 pre-existing unrelated warnings)")
54//! @yah:verify("cargo test -p yah-cloud --lib -- 723 passed; 0 failed; 4 ignored (from oss/yubaba)")
55//!
56//! @yah:ticket(R615-F2, "Overlay loader: resolve sources in CloudConfig::load, tag origin, camp-local wins on collision")
57//! @yah:status(review)
58//! @yah:assignee(agent:bundle-anthropic-miravel)
59//! @yah:at(2026-08-08T19:56:05Z)
60//! @yah:phase(P1)
61//! @yah:parent(R615)
62//! @yah:next("In CloudConfig::load, after loading camp-local machines/providers/rules, resolve each source to an infra root (git sources read from the .yah/cache/infra/ sync cache — load stays offline), load that root's machines/providers/rules, tag each entry with origin { owner, source }, and overlay UNDER camp-local. Camp-local wins on name collision.")
63//! @yah:next("The machine load site is config.rs:533 (load_dir::<MachineConfig>(paths::machines_dir(...))). Note config.rs:575 load_from_config_dir is a SECOND machine load site that deliberately skips the inherit_machines redirect for multi-root/sibling trees (W206) — decide explicitly whether sources overlay applies there too, and document the answer either way.")
64//! @yah:verify("cargo check -p cloud && cargo test -p cloud")
65//! @yah:verify("A camp with sources.toml [[source]] kind=path to a sibling camp sees that camp's machines in CloudConfig::load, each tagged with the source owner")
66//! @yah:gotcha("Cross-camp MachineConfig schema skew is real: noisetable ships an older machine schema (location/server_type/hosts_mirrors) while yah's use region/arch/[connect]. A borrowed source can carry fields the borrower's binary predates. Overlay load MUST tolerate/skip unparseable foreign entries per-file and warn — never fail the whole load.")
67//! @arch:see(.yah/docs/working/W274-linked-infra-sources.md)
68//! @yah:depends_on(R615-F1)
69//! @yah:tier(Warrior)
70//! @yah:handoff("Overlay landed in CloudConfig::load (oss/yubaba/crates/cloud/src/config.rs). After camp-local machines/providers/legacy-merge finish, SourcesConfig::load(paths::infra_dir(workspace_root)) resolves + overlay_infra_sources() merges each source's machines/providers UNDER what's already there -- camp-local wins any name collision, and among sources themselves the earlier-declared one wins (both proven by dedicated tests). Provenance is NOT a field on MachineConfig/ProviderConfig: added CloudConfig.machine_origins/provider_origins: BTreeMap<String, InfraOrigin> instead, keyed by name/id. Reason recorded in a doc comment on InfraOrigin -- MachineConfig/ProviderConfig are constructed by struct literal in test helpers across several crates (including crates/yah/agent-tools/src/cloud_tools.rs, which is fenced/live-owned this session), so widening either shape would have forced an edit there for zero semantic gain; origin is a property of the LOAD, not the machine.")
71//! @yah:handoff("GOTCHA closed: added load_dir_tolerant<T>() -- a per-file-tolerant sibling of the existing (strict) load_dir -- so one unparseable foreign machine/provider (schema skew) skips-with-a-tracing::warn! and never sinks the rest of that source's directory or this camp's own load. Proven by one_unparseable_foreign_machine_does_not_sink_the_rest_of_the_directory_or_the_load. load_dir itself is untouched -- camp-local files still hard-fail on a bad TOML, which is correct, only borrowed roots get the tolerant path.")
72//! @yah:handoff("Git sources: InfraSource::infra_root() resolves kind=path to <workspace_root>/<path>/.yah/infra (live tree, no I/O beyond building the path) and kind=git to paths::infra_source_cache_dir(workspace_root, owner)/infra -- a NEW path helper in paths.rs, also what R615-T3's `yah infra sync` target directory must be so the two line up. An unsynced git source (cache dir absent) overlays nothing and is explicitly NOT an error (test: an_unsynced_git_source_overlays_nothing_and_is_not_an_error) -- load() stays fully offline as W274 §3 requires.")
73//! @yah:handoff("select filtering implemented for machines only (name exact-match or literal mesh_tags membership -- not a glob engine, matches W274's own example verbatim) via machine_matches_select(); does NOT apply to providers -- documented as a deliberate choice, nothing in W274 or the ticket describes a provider-scoped filter.")
74//! @yah:handoff("EXPLICIT DECISION on the config.rs:575-equivalent gotcha (now load_from_config_dir): sources overlay does NOT apply there. Multi-root sibling config dirs (W206 layout (b)) are a second config root INSIDE the same camp, not a second camp -- .yah/infra/sources.toml is tied to paths::infra_dir(workspace_root) specifically, which has no well-defined meaning for an arbitrary config_dir. Documented in the function's doc comment and proven by load_from_config_dir_never_applies_sources_overlay (a sources.toml at the real workspace root does NOT leak into a load_from_config_dir call against a sibling .noisetable/ dir under that same root).")
75//! @yah:handoff("Tree anchor 85801e7f. Pathspec: oss/yubaba/crates/cloud/src/config.rs, oss/yubaba/crates/cloud/src/paths.rs (added infra_source_cache_dir + 1 test), oss/yubaba/crates/cloud/src/reconciler/mesofact_bundle.rs (CloudConfig test-literal fixed for the 2 new fields), app/yah/cli/src/cloud.rs (3 CloudConfig test-literal sites fixed, same reason). Tests: cargo test -p yah-cloud --lib (from oss/yubaba) 720 passed / 0 failed / 4 ignored, +10 over R615-F1's 710 baseline (9 overlay tests in config.rs + 1 in paths.rs). cargo build -p yah --lib (repo root) green -- confirms nothing downstream (agent-tools, cloud.rs, hub) broke from CloudConfig's two new fields.")
76//! @yah:handoff("Tree anchor at handoff: 85801e7f6b76b369c0c8ecd2e5c7874990cd9286 — the shared tree as I left it. Diff against it (`git diff 85801e7f6b76b369c0c8ecd2e5c7874990cd9286..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
77//! @yah:next("R615-T3 (yah infra sync) is unblocked and has everything it needs: paths::infra_source_cache_dir(workspace_root, owner) is the exact target directory to clone/pull git sources into, already matching what F2's overlay reads from.")
78//! @yah:next("R615-F4 (Infra tab origin badge, not in my assigned lane) can read CloudConfig.machine_origins/provider_origins directly -- no further backend plumbing needed for the badge itself.")
79//! @yah:handoff("Verified pre-existing work: overlay landed in CloudConfig::load (oss/yubaba/crates/cloud/src/config.rs) at tree anchor 871fde1c -- SourcesConfig::load resolves sources, overlay_infra_sources() merges under camp-local with camp-local-wins and earlier-source-wins collision rules, machine_origins/provider_origins BTreeMaps added to CloudConfig, load_dir_tolerant() added for per-file-tolerant foreign schema skew, InfraSource::infra_root() resolves path/git kinds, load_from_config_dir explicitly does NOT get the overlay (documented). Matches this ticket's own inline @yah:handoff notes. This session added no new code -- only ran verification and closed board state that a prior session left stuck in `open` despite the work being done.")
80//! @yah:verify("cargo check -p yah-cloud -- clean (2 pre-existing unrelated warnings)")
81//! @yah:verify("cargo test -p yah-cloud --lib -- 723 passed; 0 failed; 4 ignored (from oss/yubaba), includes overlay tests + load_dir_tolerant test + infra_source_cache_dir test in paths.rs")
82//!
83//! @yah:ticket(R605-F12, "Sovereign groups have no voting axis, so non-voting membership is inexpressible and the raft guard is enforced by an absent field")
84//! @yah:status(review)
85//! @yah:at(2026-08-20T05:15:30Z)
86//! @yah:assignee(agent:bundle-anthropic-ashguard)
87//! @yah:parent(R605)
88//! @arch:see(.yah/docs/working/W325-isolated-x86-build-capacity.md)
89//! @yah:next("OPERATOR INTENT (2026-08-19) that the model cannot currently record: us-west-003 is a NON-VOTING member of the us-west-001-based (prod) sovereign group, and us-west-011 is a DIFFERENT sovereign (dev) from 001/003. The dev/prod split is already declared correctly. The non-voting membership is not — us-west-003.toml declares no sovereign_group at all.")
90//! @yah:next("THE GAP: MachineConfig::sovereign_group is a single Option<String>, so membership is binary, and judge_join (oss/yubaba/crates/cloud/src/config.rs:459) permits a join IFF both sides declare the same non-None group. There is no way to say 'in this blast radius, but not quorum-eligible'.")
91//! @yah:next("WHY THAT IS ACTIVELY BAD, not just missing: today the ONLY thing refusing us-west-003 into the prod raft at the join gate is its ABSENT stamp. Its own file is emphatic it must never hold a raft node id ('a home-internet partition should never be able to stall the raft'), and that guarantee currently rests on a field nobody wrote. Stamping it prod to record the operator's real intent would REMOVE the guard. This is precisely the W305 failure mode that produced R742-T4: `no-voter` sat inert on three nodes asserting something nothing enforced.")
92//! @yah:next("PROPOSED SHAPE (recommended): a second axis, e.g. sovereign_role = voter | non-voter (default voter for back-compat, or make it required), with judge_join permitting a same-group join only for voters. Then us-west-003 stamps prod + non-voter, the intent is machine-readable, and the raft guard stops depending on omission. us-west-004 (R605-T7) would take the same shape.")
93//! @yah:next("TOUCHES TWO COPIES OF THE PREDICATE, do not fix only one: cloud::judge_join renders the camp-side refusal, but the predicate itself lives in workload_spec::sovereign::join_permitted because yubaba's POST /raft/add-learner gate asks the same question and there is deliberately no yubaba -> cloud edge. Also re-read `yubaba serve --sovereign-group`, whose node-side gate is narrower on purpose (an unset flag means 'declared nothing', not 'declared standalone').")
94//! @yah:gotcha("THE CODE AND THE OPERATOR CURRENTLY DISAGREE ABOUT 003, and a reader should know which is which before editing. judge_join's own doc comment asserts 'prod and dev are both stamped, and us-west-002/003/015 are deliberately not raft members' — i.e. R742-F1 modelled 003 as STANDALONE. The operator's model is that it is a NON-VOTING MEMBER of prod. Those are different claims, not a wording difference: standalone means no blast-radius relationship to 001 at all. Do not silently 'correct' either side; this ticket is the reconciliation.")
95//! @yah:gotcha("FLEET STATE AS DECLARED (2026-08-19): prod = us-west-001, us-south-001, us-east-001. dev = us-west-011, us-west-013, us-west-014. NO sovereign_group declared = us-west-002, us-west-003, us-west-015. Verify against the files rather than trusting this list — xtask/tests/fleet_sovereign_groups.rs pins the roster and will need updating in the same change (it also asserts the stamp parses as a TOP-LEVEL key, which matters because 003 has a long comment block before [allocatable] where a stamp would silently become a member of that table).")
96//! @yah:gotcha("SEPARATE AXIS, DO NOT ENTANGLE: mesh membership is not sovereign membership. The standing rule is ONE mesh for the entire fleet regardless of group (operator, 2026-08-19), so us-west-003 and us-west-011 enrolling in headscale is unrelated work with no design question in it — see R605-T10. A voting axis on sovereign_group must not become a reason to keep any node off the mesh.")
97//! @yah:gotcha("SHARED-TREE COLLISION, live 2026-08-20: R772 (Miravel:spade, session:ce6d74a9) is refactoring oss/yubaba/crates/cloud/src/validate.rs at the same time and the file is currently RED - error[E0425] cannot find function load_machines at validate.rs:753, a half-landed extraction of the machine-loading walk that check_inert_taints / check_retired_arch_tags / the new check_unroled_sovereign_members all duplicate. That error is NOT from this ticket. Told them by party.chat and asked them to absorb check_unroled_sovereign_members into load_machines rather than leave one holdout. Do not hand-fight the file.")
98//! @yah:gotcha("R772 ALSO BROKE THREE PRE-EXISTING INGRESS TESTS, again not this ticket: two_services_fronting_one_node_collate_into_one_front_door, a_cross_service_hostname_clash_is_reported_with_both_declarations, one_mirrors_broken_declaration_does_not_hide_the_rest - all failing with 'providers.compute.use = hetzner - no such provider'. Cause is their new CloudConfig::load(workspace_root) at validate.rs:750 inside collate_workspace_ingress; the fronted_mirror fixture declares the slot but never writes infra/providers/hetzner.toml, and CloudConfig::load runs cross_ref_validate. Left alone deliberately - peer-owned.")
99//! @yah:gotcha("TRAP THAT MADE THREE OF MY OWN TESTS PASS FOR THE WRONG REASON: the machine-lint sweeps SKIP unparseable TOMLs by design (a peer's half-written scaffold must not sink the sweep). So a test fixture missing a REQUIRED MachineConfig field - mesh_tags is the one that bites - is silently skipped, the lint finds nothing, and every assert-empty test passes vacuously. Only the one test asserting found.len() == 1 noticed. write_sovereign_machine now always writes mesh_tags = [] and carries a comment saying why. Check this before trusting any new test in cloud::validate.")
100//! @yah:verify("cargo test -p yah-workload-spec --lib sovereign (from oss/yah-base) -- 9 passed, 0 failed. Covers both new refusals (a_non_voting_member_does_not_join_its_own_group, a_non_voting_target_has_no_quorum_to_join), the back-compat pin (the_default_role_is_the_pre_r605_f12_meaning), and the one-spelling round-trip across TOML/CLI/JSON.")
101//! @yah:verify("cargo test -p yubaba --lib sovereign (from oss/yubaba) -- 13 passed, 0 failed. Includes a_non_voting_joiner_is_refused_by_role_not_by_group, a_non_voting_target_refuses_every_joiner, a_group_without_a_role_key_is_a_voter_not_a_refusal (the deployed-fleet back-compat seam), a_peer_reports_its_role_in_the_toml_spelling.")
102//! @yah:verify("cargo test -p yubaba --test raft_sovereign_group (from oss/yubaba) -- 11 passed, 0 failed, up from 8. Three new end-to-end against real single-node rafts: a_non_voting_member_of_the_same_group_is_refused, a_non_voting_leader_refuses_to_grow_its_quorum, a_node_publishes_its_role_and_the_leader_reads_it_there (which also proves the request body cannot vote a non-voter in - the leader dials the joiner).")
103//! @yah:verify("cargo test -p xtask --test fleet_sovereign_groups (from repo root) -- 2 passed, 0 failed. THE DECISIVE ONE: parses the real .yah/infra/machines/*.toml through the actual MachineConfig deserializer. Confirms us-west-003 = prod + non-voter on disk, all six pre-existing voters now stamped sovereign_role = voter explicitly, and neither key swallowed by a table header.")
104//! @yah:verify("cargo test -p yah-cloud --lib (from oss/yubaba) -- 891 passed, 3 failed, where all 3 failures were R772's ingress-collate tests and none were mine. A clean re-run is BLOCKED, not failing: R555's in-flight AdmissionGrant.secrets field breaks velveteen-exec, and yah-cloud is not a root workspace member so its dev-deps can only resolve from the oss/yubaba workspace. Re-run once R555 lands.")
105//! @yah:handoff("LANDED, operator chose the second-axis shape (Call 1 = A, 2026-08-20). sovereign_role = voter | non-voter now sits beside sovereign_group, and ONE predicate judges both: workload_spec::sovereign::join_permitted(Membership, Membership) where Membership { group: Option<&str>, role: SovereignRole }. Permitted iff same non-None group AND both sides Voter. Both copies of the predicate call it - cloud::judge_join (camp-side) and yubaba::sovereign_group::judge (node-side) - so the rule itself cannot drift; only the prose differs, which was already the R742-F1 split.")
106//! @yah:handoff("WHY THE ROLE IS CHECKED ON BOTH SIDES, since only the joiner half was asked for: a join grows a quorum and it takes two nodes. Refusing a non-voting JOINER is the us-west-003 case. Refusing a non-voting TARGET is the same assertion read from the other end - a box declared non-voting that is serving add-learner is already holding a raft seat its own declaration forbids, and permitting there would paper over the contradiction. Both refusals name the role rather than the group when the groups match, because a message reading 'cross-group join refused: prod and prod' reads as a bug in the check.")
107//! @yah:handoff("THE DEFAULT IS THE LOAD-BEARING DECISION AND IT IS DELIBERATELY PERMISSIVE. An absent sovereign_role resolves to Voter (MachineConfig::sovereign_membership, the ONE place the Option is resolved). Reason: before this field, declaring a group WAS declaring quorum eligibility, so absence has to keep meaning that or the change silently retires six live voters. The permissiveness is bounded at the other end by cloud::validate::check_unroled_sovereign_members, which makes `yah cloud validate` FAIL on a group stamp with no role beside it - so the default can be reached by choice but not by silence. MachineConfig::sovereign_role stays Option<SovereignRole> (not a defaulted plain field) precisely so that lint can tell 'chose voter' from 'never considered it'.")
108//! @yah:handoff("NODE-SIDE BACK-COMPAT SEAM, pinned by a test because it is a decision and not an oversight: a peer answering GET /raft/status with a sovereign_group but NO sovereign_role key - every yubaba built between R742-F1 and R605-F12, which today is the entire prod raft - is read as Voter, not refused. Refusing would freeze a stamped cluster's growth until every member was rolled, strictly worse than what the role guards against, and it is the same degrade-toward-prior-behaviour stance the module already took for the group. Residue, named rather than hidden in read_group's doc: a box whose machine.toml says non-voter but whose daemon predates the flag answers 'voter' and the node gate admits it. judge_join refuses it camp-side, which is where operator-driven joins go. Window closes per-group as its nodes carry the flag.")
109//! @yah:handoff("FILES: workload-spec/src/sovereign.rs (SovereignRole + Membership + role-aware join_permitted, +227). cloud/src/config.rs (sovereign_role field, sovereign_membership(), judge_join same-group role branch, SovereignRole re-exported from cloud::config). cloud/src/validate.rs (check_unroled_sovereign_members + UnroledSovereignMember). app/yah/cli/src/cloud.rs (lint wired: ERROR in `yah cloud validate`, WARNING in the apply preflight - same split as inert-taint/retired-arch-tag, because an unwritten role changes no placement decision and the machine may be declared in a tree this camp does not own). yubaba/src/{sovereign_group,lib,main}.rs (--sovereign-role flag, ServerState.sovereign_role, /raft/status publishes it always-never-null, gate both directions). yubaba-test-harness/src/solo_node.rs (solo_node_with_sovereign_role). .yah/infra/machines/*.toml (7 files). xtask/tests/fleet_sovereign_groups.rs + fleet_build_placement.rs. W325 section 3d.")
110//! @yah:handoff("ONE BEHAVIOUR CHANGE WORTH A SECOND OPINION: a node started with --sovereign-role non-voter AND a --raft-node-id now refuses EVERY add-learner. I judged that correct - it is a contradiction the operator should see loudly - but the symptom is 'joins mysteriously stop working' rather than a startup refusal. main.rs warns loudly at boot when that pair is present; I did NOT make it fatal, because refusing to start could brick a node mid-roll. Reconsider if it bites.")
111//! @yah:handoff("NOT DONE, and it is a HARD GATE: .yah/schema/machine.toml.schema.json has NOT been regenerated, so sovereign_role is absent from it and schema-drift-guard (scripts/check-schema-drift.sh, a step in .yah/qed/check.toml, run by CI on every push) WILL FAIL. Fix is `cargo run -p xtask -- emit-schemas` from the repo root - it was queued behind ~7 concurrent peer cargo builds for the whole session. Nothing else is required to make this pushable.")
112//! @yah:handoff("ALSO NOT RE-CONFIRMED: `cargo test -p yah-cloud --lib` needs a clean run. Its last real run was 891 passed / 3 failed with all three failures belonging to R772's ingress-collate work and none to this ticket. The re-run is BLOCKED not failing - R555's in-flight AdmissionGrant.secrets field breaks velveteen-exec, and yah-cloud is not a root workspace member so its dev-deps only resolve from the oss/yubaba workspace where that break lives. Re-run from oss/yubaba once R555 lands.")
113//! @yah:verify("cargo run -p xtask -- emit-schemas (from repo root) -- wrote 8 files, exit 0 after an 18m24s build queued behind ~7 concurrent peer cargo jobs. .yah/schema/machine.toml.schema.json now carries the sovereign_role property (anyOf SovereignRole | null, with the full doc comment) and the SovereignRole definition as a oneOf over the two string enums voter / non-voter. The schema-drift-guard gate for THIS ticket is closed.")
114//! @yah:gotcha("emit-schemas IS ALL-OR-NOTHING AND WILL PICK UP A PEER'S UNCOMMITTED WORK. Running it to close this ticket's machine-schema drift also regenerated .yah/schema/secret.toml.schema.json (+34) from R555-F5's in-flight SecretAccess::Recipes / RecipeMatch source. That output is CORRECT for the tree as it stands and was not hand-edited, but it means the schema diff in the working tree is not purely R605-F12's: machine.toml.schema.json (+32) is this ticket, secret.toml.schema.json (+34) is R555. Told Ashguard:spade by party.chat so they carry it with their commit rather than regenerating on top. Anyone splitting these commits needs to split the schema diff too.")
115//! @yah:handoff("ALL GATES CLOSED as of 2026-08-20. Both items listed as outstanding in the earlier handoff notes are done: emit-schemas ran (machine.toml.schema.json carries sovereign_role + the SovereignRole voter/non-voter enum, drift guard satisfied), and cargo test -p yah-cloud --lib is 896 passed / 0 failed once R555 and R772 settled. 45 tests green across workload-spec (9), yubaba lib (13), yubaba raft integration (11), yah-cloud lib (10 of this ticket's, within 896), xtask fleet (2). Ready for review. NOTE for whoever commits: the working tree's schema diff is not purely this ticket - .yah/schema/machine.toml.schema.json (+32) is R605-F12, .yah/schema/secret.toml.schema.json (+34) is R555-F5, both correct generated output from one emit-schemas run. Ashguard:spade has agreed to carry theirs.")
116//! @yah:verify("cargo test -p yah-cloud --lib (from oss/yubaba) -- 896 passed, 0 FAILED, 4 ignored. The blocked check from earlier is now clean: R555 landed the velveteen-exec and TransformRecipe.secrets fixes, R772's ingress-collate work settled (they replaced the CloudConfig::load in collate_workspace_ingress with a narrower machines-only loader, so cross_ref_validate can no longer fail the collate over an unrelated provider typo). All 45 R605-F12 tests across the four crates are green simultaneously on one tree.")
117//! @yah:verify("Confirmed by NAME rather than by total, since a passing count proves nothing about which tests ran: cargo test -p yah-cloud --lib -- role voter voting lists all ten of this ticket's cloud tests green - a_non_voting_member_is_refused_into_its_own_group, a_non_voting_target_has_no_quorum_to_grow, a_refusal_names_the_group_when_fixing_the_role_would_not_help, an_unwritten_role_still_joins_its_group, a_non_voter_is_still_in_the_group_it_names, sovereign_role_round_trips_and_is_omitted_when_unwritten, a_group_with_no_role_is_reported_with_the_declaring_file, either_stated_role_is_clean, a_machine_in_no_group_is_not_asked_for_a_role, unroled_findings_are_ordered_by_file_so_output_is_stable.")
118//!
119//! @yah:ticket(R876-B7, "Node taints are structurally inert for mirror-declared placements: you cannot drain a node, and it fails silently")
120//! @yah:at(2026-09-09T09:05:55Z)
121//! @yah:status(review)
122//! @yah:assignee(agent:bundle-anthropic-ashguard)
123//! @yah:parent(R876)
124//! @yah:severity(high)
125//! @yah:next("SECOND HALF, and it is what makes the relay's headline question answerable: a working taint must produce a MOVE, not a refusal. Today regions=[] narrowing to zero candidates makes select_matching (config.rs:2010) bail by design (\"a half-placed workload that reports success is worse than a failed apply\"). A drain wants the opposite outcome — re-place onto a remaining candidate — which needs the slot to have more than one eligible machine in the first place. Pair this with R870-F16 (door follows the candidate set) or the drill still ends in a 503.")
126//! @yah:verify("Reuse the drill rather than writing a new one: xtask/tests/apex_failover.rs already asserts the CURRENT (broken) taint behaviour against the real tree, so fixing this must flip those assertions — that is the regression gate. Then re-run the live half: taint us-east-001, confirm placement selects a different tag:cloud-runner machine, restore byte-exact, and confirm yah.dev stays 200 throughout.")
127//! @yah:gotcha("IT FAILS SILENTLY, WHICH IS THE SHARP EDGE. \"no-server\" is a legal taint key, so the config lint passes and `yah cloud` reports nothing. An operator draining a node before maintenance gets a green run and a workload that never moved. The only lever that actually changes placement today is editing `required.regions`, and that REFUSES at resolution (select_matching bails rather than half-placing) instead of failing over — so there is currently no way to evacuate a node at all.")
128//! @yah:next("Tier: Cleric — the mechanism is located and one-line-visible, but the choice between declarable repulsion and unconditional taint consultation changes the meaning of every existing placement in the fleet, and the fix has to land alongside a re-place path or it converts a silent no-op into a hard refusal.")
129//! @yah:gotcha("MEASURED, NOT INFERRED — R876-S2's drill, 2026-09-09. `taints = [\"public-ip\", \"no-server\"]` was written onto the REAL .yah/infra/machines/us-east-001.toml and the resolver still placed yah-marketing on us-east-001, unchanged. Restored byte-exact (diff empty, sha256 back to 17dd15e2..., git clean against blob d66ab6d8); yah.dev stayed 200 throughout and no mutating apply was run.")
130//! @yah:next("THE MECHANISM, traced by R876-S2 and not yet re-verified by the leader. Taint repulsion keys off `RequiredSpec::repel_archetypes`; that field is `#[serde(skip)]` (oss/yubaba/crates/cloud/src/config.rs:4067), so a slot declared in a mirror's `required = {...}` ALWAYS deserializes with it empty. `matches` (config.rs:4182) consequently never reads `machine.taints` at all. Confirm both line anchors before editing — the shared tree moves.")
131//! @yah:handoff("SEMANTICS LANDED — repel-by-default + declarable toleration. `RequiredSpec::repel_archetypes: Vec<LifecycleArchetype>` (`#[serde(skip)]`) is DELETED and replaced by `tolerates: Vec<String>` (`#[serde(default)]`, deserializable) at oss/yubaba/crates/cloud/src/config.rs:4319. `matches` (config.rs:4397) no longer iterates a field of `self`: it walks `machine.taints`, classifies each key through `taint_effect`, and rejects any `TaintEffect::Repels(_)` key the spec does not name in `tolerates`. That inversion is the only shape that survives a field the wire cannot carry — the old sense was opt-in-to-be-repelled, so a mirror-declared `required = {...}` always deserialized with an empty archetype set and `machine.taints` was never read at all. Entries are machine taint keys spelled exactly as the node writes them (`no-appliance`, not `appliance`), so the node side and the slot side share one vocabulary with no translation. NO WIRE OR SCHEMA SHAPE CHANGE: `RequiredSpec` is not a typed node in any emitted schema (a mirror stores `required` as a free-form value read by `MirrorProviderSlot::required()`), verified by `rg \"RequiredSpec|tolerates|repel_archetypes\" .yah/schema/*.json` — the only hits are prose inside a doc-comment description.")
132//! @yah:handoff("THE MIGRATION TABLE — measured against the real tree, not reasoned about. FLEET TAINTS, all nine machines (`grep -rE \"^\\s*taints\\s*=\" .yah/infra/machines/*.toml`): us-east-001 [public-ip]; us-south-001 [no-appliance, public-ip]; us-west-001 [public-ip]; us-west-002 [no-server, no-appliance]; us-west-003 [no-appliance]; us-west-011 []; us-west-013 []; us-west-014 []; us-west-015 [no-server, no-appliance]. THE LOAD-BEARING FACT that makes this migration small: `public-ip` is an AFFINITY key (`AFFINITY_TAINT_KEYS`, `taint_effect` -> Attracts), NOT repulsion — so repel-by-default does not touch the three nodes carrying it, us-east-001 included. Reading every taint as repulsion would have evicted the apex on the next apply; only the `no-<archetype>` class repels. Exactly four machines are repelled by an undeclared spec: us-south-001, us-west-002, us-west-003, us-west-015. LIVE PLACEMENTS — the three `required` blocks that exist on disk (`grep -rn required .yah/services/*/mirrors/*.toml`): (1) yah-marketing providers.bundle, cloud.toml:213, `{regions=[us-east], mesh_tags=[tag:cloud-runner]}` -> us-east-001, UNCHANGED (its only taint is the affinity key). (2) yah-cloud providers.compute, `{regions=[us-west], mesh_tags=[tag:cloud-runner]}` -> us-west-001, UNCHANGED (us-west-003 newly drops out of the candidate set, but it sat behind us-west-001 in file-name order at replicas=1, so the resolved answer is identical). (3) yah-cloud-admin providers.compute, same constraint -> us-west-001, UNCHANGED. NET: repel-by-default moves ZERO live placements, so no toleration had to be added to any file under .yah/services/ or .yah/infra/ and none was. No file under .yah/infra/machines/ or .yah/services/ was written by this ticket at all.")
133//! @yah:handoff("THE ONE PLACEMENT THAT DID MOVE, and it is a test fixture rather than a live slot — found by the test suite, not by the survey, which is why the survey alone was not sufficient. `xtask/tests/mirror_ingress.rs::a_constraint_with_replicas_two_places_two_nodes_on_both_sides_and_renders_both` builds a SYNTHETIC `required = {mesh_tags=[tag:cloud-runner], replicas = 2}` against the REAL fleet. Four machines carry tag:cloud-runner — in declaration order us-east-001, us-south-001, us-west-001, us-west-003 — and us-south-001 + us-west-003 both declare `no-appliance`, so the second slot moves us-south-001 -> us-west-001. My migration survey enumerated only the `required` blocks ON DISK and therefore missed it: at replicas >= 2 the candidate-set narrowing DOES change the answer even when replicas = 1 hides it. Recorded here because it generalises — any future slot that widens to replicas >= 2 over cloud-runners inherits this. Fixed at the site that caught it (mirror_ingress.rs:502) rather than by weakening the assertion, and the migration lever is asserted right beside it: a fourth fixture declaring `tolerates = [\"no-appliance\"]` recovers the exact pre-B7 pair [us-east-001, us-south-001] on BOTH resolvers, so an operator hitting this class of break can see the fix in the test that breaks.")
134//! @yah:verify("BASELINE MEASURED BEFORE EDITING, then re-measured after. `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` = 1129 passed / 0 failed / 4 ignored, exit 0 (the run completed and printed its result line before my first Edit; a deferred W298 skew advisory later named config.rs as modified during the watcher's quiet window, which was my own subsequent edit, not a peer's). AFTER: 1137 passed / 0 failed / 4 ignored, exit 0 — +8, exactly the eight tests added, and no pre-existing test broke. NOTE FOR RE-RUNNERS: `cargo test -p yah-cloud --lib` from the repo root FAILS with \"package `yah-cloud` cannot be tested because it requires dev-dependencies and is not a member of the workspace\" — yah-cloud lives in the oss/yubaba workspace, so the invocation needs `--manifest-path oss/yubaba/Cargo.toml`. Also `cargo check --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --all-targets` exit 0 and `-p yubaba --all-targets` exit 0 (yubaba consumes cloud, so it is where the field removal would have surfaced). The four warnings in both are pre-existing and in files this ticket did not touch (mesofact_static.rs unused imports, app_manifest.rs dead field, pond_door.rs unused fn, reconciler/mod.rs non-snake-case).")
135//! @yah:verify("EIGHT NEW UNIT TESTS in config.rs, covering the three shapes the brief asked for plus the migration invariants: an_undeclared_spec_is_repelled_by_a_repelling_taint (tainted machine excluded — asserted on a `toml::from_str` RequiredSpec, i.e. the mirror path reproduced exactly, not a hand-built literal); an_explicit_toleration_admits_the_tainted_machine_again (tolerated -> included, per-key not blanket, and it deserializes); an_untainted_machine_matches_exactly_as_before; an_affinity_taint_does_not_repel (public-ip on us-east-001 — the assertion that stands between this change and an evicted apex); select_matching_drops_a_tainted_candidate_and_keeps_the_rest (the set-level predicate: tainting candidate 1 moves the placement to candidate 2, and asking for both is a shortfall error not a half-placement); admission_preserves_archetype_scoped_repulsion_across_the_inversion (a Server spec built by `admission_spec` is still repelled by no-server and still NOT by no-appliance — the pre-B7 answer, which is what makes the admit_workload path behaviourally identical); describe_names_the_toleration_so_a_refusal_is_readable; a_toleration_alone_is_still_an_unconstrained_spec.")
136//! @yah:verify("REGRESSION GATE FLIPPED, not deleted. `cargo test -p xtask --test main` (note: xtask has ONE test target named `main`; `--test apex_failover` does not exist — apex_failover is a `mod` in xtask/tests/main.rs). Result 65 passed / 1 failed. xtask/tests/apex_failover.rs: the drill's finding-1 test was inverted and renamed every_repelling_taint_at_once_leaves_the_apex_bundle_exactly_where_it_was -> ..._now_makes_the_apex_node_ineligible; it now asserts that ONE repelling key is enough (checked before the all-three case so a regression handling only the union is still caught), that all three refuse, and that restoring us-east-001's real taint list [\"public-ip\"] puts the placement straight back. The module header was rewritten to say the hole is closed. ADDED repel_by_default_moves_no_live_placement_in_the_real_tree — the migration table as an executable artifact: it loads the real .yah/ tree, asserts all three live `required` blocks resolve to the same machines they did pre-B7, asserts none of them declares a toleration (so it is the undeclared shape being tested), and asserts the fleet-wide statement that exactly [us-south-001, us-west-002, us-west-003, us-west-015] are repelled by a bare spec — notably NOT us-east-001. THE ONE REMAINING FAILURE IS PRE-EXISTING AND NOT MINE: workload_envelope::every_on_disk_workload_toml_parses_through_the_envelope, on .yah/infra/state/sources/scrabcake/site/site/workload.toml (`unknown field routes`). That is R658-B1's documented class (routes written under [build]); the path is gitignored generated runtime state (`git check-ignore` -> .yah/.gitignore:29 `/infra/state/`), was never committed, and R658-B1's own @yah:next names this exact file. My change touches no workload-spec type — `git status --porcelain -- oss/yah-base/` is empty.")
137//! @yah:handoff("SCOPE BOUNDARY HELD, deliberately. yah-marketing's candidate set was NOT widened: `.yah/services/yah-marketing/mirrors/cloud.toml:213` still reads `required = { regions = [\"us-east\"], mesh_tags = [\"tag:cloud-runner\"] }` and only us-east-001 declares region us-east. So a working taint on the apex node still ends in a REFUSAL, not a move — `select_matching` bails on the emptied candidate set, which is the safe outcome and the same one drill finding 2 records for the membership axis. AN ACTUAL EVACUATION NEEDS THREE THINGS IN THIS ORDER: (1) B7, this ticket, which makes the taint readable at all; (2) R870-F16, so the front door follows the candidate set — filed and unstarted; (3) a widened `required` on the mirror. Doing (3) before (2) buys a workload that relocates and a yah.dev that 503s, which is why it was not done here. Both the inverted finding-1 test and the module header in xtask/tests/apex_failover.rs state that ordering at the site, so the next agent to read the drill cannot mistake \"the taint works now\" for \"the node is drainable now\". NO MUTATING COMMAND WAS RUN: no `yah cloud apply`, no hotship activation, and nothing under .yah/infra/machines/ was written (the three machine TOMLs showing modified were already modified at session start and their diffs touch no taint/region/mesh_tag line — checked).")
138//! @yah:handoff("GENERATED ARTIFACTS REGENERATED, and one of them is a peer's. `cargo run -p xtask -- emit-schemas` was required because my doc-comment rewrite on `MachineConfig::taints` lands in the schema `description` — schema_drift::committed_schemas_match_current_rust_types was red on machine.toml.schema.json. The regen also swept in mirror.toml.schema.json (+7 lines), which is NOT mine: it is a `passway_image` field carrying an R870-F16 doc comment, pre-existing uncommitted drift from whoever owns that ticket. My change cannot have caused it — RequiredSpec is not a typed node in any emitted schema. Regenerated per CLAUDE.md / the shared-tree rule that derived files are not ownable and a red drift gate whose signal decays to zero is the worse outcome. @Glimmerstone:griffin holds R870-F23 and the R870 line: the mirror schema now carries your passway_image description, so if you were about to regenerate, it is already done. Both schema files are the only two under .yah/schema/ that changed.")
139//! @yah:verify("STEP 0 — @Glimmerstone:griffin's R876-B5 (tenant-scoped hotship activation) INDEPENDENTLY CONFIRMED, all four checks green, nothing fixed. (1) `bash -n scripts/hotship.sh` clean. (2) `./scripts/hotship.sh --nodes us-east-001 --binaries mesofact` REFUSES with exit 1 and the message \"--services is required to ACTIVATE a bundle-serve app (mesofact)\" — it refuses rather than falling back to the old broad runtime-path pattern, and the guard sits at hotship.sh:507 ahead of the version stamp and every remote call. (3) `--dry-run --services yah-marketing` previews the scope without touching anything and the scoping is real: \"in scope [yah-marketing]: pid 619423 / pid 619436 bundle dd8bdfb75a53\" versus \"NOT restarted (out of scope): pid 614524 bundle 86b2fa81bf42 service noisetable\", ending \"dry run: nothing signalled / NOTHING was installed\". (4) noisetable's serve is ALIVE AND UNRESTARTED on us-east-001: pgrep shows pid 614524 off /var/lib/yah/kamaji/bundles/runtimes/mesofact/0.8.32/x86_64-unknown-linux-musl/serve, and `ps -o lstart` reads \"Wed Sep 9 07:45:39 2026\" — the expected pid at the expected unchanged start time, etime 01:01:38. `curl -sS -o /dev/null -w %{http_code} https://yah.dev/` = 200. No real hotship activation was run.")
140//! @yah:verify("BUILDS. `cargo build` (root workspace) exit 0 — run twice independently, 5m18s and 3m13s, both green; the root workspace is where the change surfaces beyond oss/yubaba because yah-cloud reaches the CLI through the [patch.crates-io] bridge. `cargo check --manifest-path oss/yubaba/Cargo.toml -p yubaba --all-targets` exit 0. Clean re-measure of `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` after all annotation writes: 1137 passed / 0 failed / 4 ignored, exit 0 — identical to the first post-change measurement, so the earlier W298 skew advisory naming config.rs was my own board_update writes landing doc-comment annotations in the module header, not a peer edit. A later advisory on the root build named app/yah/cli/src/cloud.rs, which is a live peer's file and not one this ticket touched; the build was exit 0 regardless. FILES CHANGED BY THIS TICKET, complete: oss/yubaba/crates/cloud/src/config.rs, xtask/tests/apex_failover.rs, xtask/tests/mirror_ingress.rs, .yah/schema/machine.toml.schema.json, .yah/schema/mirror.toml.schema.json. Nothing under .yah/infra/ or .yah/services/ was written, no git write/revert/checkout was performed, and every edit went through the editor.")
141//! @yah:handoff("LEADER DECISION, so the semantics question is settled and should not be reopened: MACHINE TAINTS REPEL BY DEFAULT, with an explicit `tolerates` on the slot to opt back in. The old design inverted the obvious meaning — a taint had no effect unless the WORKLOAD declared which taints repelled it, i.e. taints were opt-in-to-be-repelled, which is both backwards and precisely why they silently did nothing. `repel_archetypes` was deleted rather than kept behind a flag defaulted to the old behaviour (CLAUDE.md, \"break it, don't tape it\").")
142//! @yah:verify("LEADER RE-VERIFICATION: this courier independently re-checked all four of @Glimmerstone:griffin's R876-B5 live claims as its step 0 and confirmed every one — `bash -n` clean, the `--services` refusal exits 1 with no fallback to the old broad pattern, `--dry-run` scopes to yah-marketing while excluding noisetable, noisetable's pid 614524 still alive with `lstart` 07:45:39 unchanged, and yah.dev 200. Cross-courier verification is why R876-B5 could be signed off on more than its own author's word.")
143//! @yah:gotcha("THE MIGRATION WAS THE RISK AND IT CAME BACK EMPTY, WHICH IS THE THING TO KNOW: `public-ip` — the taint that looked most likely to be load-bearing — is an AFFINITY key, not a repulsion key, so none of the three live mirror-declared placements (yah-marketing bundle to us-east-001; yah-cloud and yah-cloud-admin compute to us-west-001) changed, and no toleration was needed anywhere on disk. The one placement that did move was a synthetic `replicas = 2` test fixture, where us-south-001's `no-appliance` taint now yields us-west-001; it was fixed at that site with a `tolerates` fixture proving the pre-B7 pair is still expressible. Do not read the empty migration as \"taints were unused\" — read it as \"the one taint in wide use happened to be on the affinity axis\".")
144//!
145//! @yah:ticket(R870-F23, "Render and supervise the inner door: the service.toml + domain-manifest join that feeds passway's PathRouter config")
146//! @yah:status(review)
147//! @yah:phase(P2)
148//! @yah:at(2026-09-11T00:25:14Z)
149//! @yah:assignee(agent:bundle-anthropic-ashguard)
150//! @yah:parent(R870)
151//! @yah:next("THE CONSUMER SIDE IS DONE AND ITS FORMAT IS FIXED (R870-T18, in review). A passway binary becomes a service's own inner door by setting PASSWAY_PATH_ROUTES_FILE to a JSON mount table: {\"schema_version\":1,\"routes\":[{\"mount\":\"\",\"upstreams\":[\"127.0.0.1:8081\"]},{\"mount\":\"/app\",\"upstreams\":[\"127.0.0.1:8082\"],\"headers\":{\"cross-origin-opener-policy\":\"same-origin\"}}]}. Parser + validation: oss/passway/crates/passway/src/path_routes_file.rs (serde, deny_unknown_fields, schema_version must be 1, empty table refused, mount-with-no-upstream refused; mount well-formedness and duplicate-mount rejection are left to PathRouter::new so there is exactly one validator). Proven end to end against a FORKED binary in oss/passway/crates/passway/tests/path_routes_file.rs. This ticket is the producer: write that file.")
152//! @yah:next("WHY THIS IS A SEPARATE TICKET AND NOT HALF OF R870-T18. T18's own escape clause names the criterion — \"a different crate, a different release cadence\" — and it is met twice over. (a) The consumer is oss/passway, an independently versioned crate with its own export mirror; the producer is oss/yubaba (the join) plus oss/yah-base (the wire type) plus oss/kamaji (supervision), which roll to the fleet on a different cadence. (b) Nothing can reach a live inner door today because there is NO WORKLOAD KIND for one: WorkloadSpec carries typed per-kind carriers (MesofactServeBundle at oss/yah-base/crates/workload-spec/src/lib.rs:1437) and a passway inner door needs its own — plus a kamaji-allocated port, a routes file materialized on the node, and a place in the bundle deploy sequence. Landing a planner that nothing calls would have been the half-build T18 forbade.")
153//! @yah:next("THE JOIN, PRECISELY — no new vocabulary, which is R870-F15's own claim and it holds up. Inputs: .yah/services/<svc>/service.toml (ServiceComponent { id, kind, mount, ... }, config.rs:3427) and .yah/domains/<zone>.toml (DomainRoute { path, headers, mode }, config.rs:4558, where front_door = passway). Per mount: mount = path_route::mount_from_component(component.mount) — that function already exists and is already the ONE place the \"app\"/None to \"/app\"/\"\" translation happens; headers = the DomainRoute whose route_path_prefix(path) equals normalize_mount(component.mount) (cross_ref_validate already PROVES those two agree, config.rs:1688-1725, so the join cannot silently mismatch); upstreams = the address of the deployed unit serving that mount. Only the last one is placement-time and is why this needs the workload kind above. Group by DEPLOYED UNIT, not by component: every bundle-tier component of a service shares ONE bundle workload (that is config 1, R870-B11), so config-1 mounts collapse to a single root upstream and only independently-deployed units earn their own mount.")
154//! @yah:next("THE TWO ADMISSION RULES, and where each one goes. Both belong to the GENERATOR, never to passway — passway proxies whatever PathRouter it is handed and has no view of how many components a service declares. (1) A service with ONE independently-deployed unit gets NO inner tier at all — enforce by construction: the planner returns Option<InnerDoorPlan> and answers None below two units, so there is no config to write and no process to supervise, and the negative is assertable on the ABSENCE of the plan rather than on a site staying up. (2) A component cannot be both bundle-staged (config 1) and its own workload. R870-B11 landed the config-1-internal half in CloudConfig::cross_ref_validate (config.rs:1621-1657, two bundle components at one mount are refused); put this half in the SAME loop rather than a parallel one. NOTE, checked not assumed: the second half is NOT EXPRESSIBLE TODAY — [providers.bundle] is a per-MIRROR slot, not per-component, so there is no way to say \"give this one component its own workload\" at all. The rule becomes writable in the same commit that introduces that vocabulary, which is this ticket. Do not invent the vocabulary separately.")
155//! @yah:gotcha("DESIGN WRINKLE FOUND WHILE BUILDING R870-T18, and it is an operator call, not a coding one. passway ALWAYS terminates TLS on its listener: TlsMode has exactly two variants, Manual and Acme (oss/passway/crates/passway/src/tls.rs:215), and main() unconditionally calls proxy_service.add_tls_with_settings(&listen, None, tls_settings). So an inner door on loopback still needs a cert on disk, and the outer door still needs PASSWAY_UPSTREAM_TLS=true plus an SNI to reach it. That works — T18's binary-level test does exactly this with an rcgen self-signed leaf — but it means the \"cheap inner tier\" costs a cert, a renewal story, and an upstream TLS handshake per request on loopback. The obvious fix is a plaintext listener mode, and it was deliberately NOT taken in T18: adding a way for a public-facing trust-boundary door to serve cleartext is a security decision with a blast radius past this relay. Decide it before building the supervisor, because it changes what the workload spec has to carry.")
156//! @yah:verify("A two-component service whose components deploy INDEPENDENTLY gets an inner door: one yah cloud apply leaves both https://<host>/ and https://<host>/app/ at 200, and curl -sI on /app/ carries cross-origin-opener-policy: same-origin AND cross-origin-embedder-policy: require-corp from the /app/* route in the domain manifest, while / carries neither.")
157//! @yah:verify("THE NEGATIVE, asserted on absence rather than on uptime: a single-component service (yah-marketing) produces NO inner-door config and NO inner-door process — no routes file materialized on the node, no extra supervised workload in kamaji's table, and a byte-identical workload spec to today. A unit test on the planner returning None is the cheap half; the node-side absence check is the half that matters.")
158//! @yah:gotcha("OPERATOR CALL ASKED AND NOT ANSWERED (R870 relay leader, session:abde2cbb, 2026-09-09). The TLS question in this ticket first gotcha was put to the operator as a three-way choice and the prompt timed out unanswered after 30 minutes, so it remains genuinely open — it was not skipped and not decided by default. The three options as framed, so whoever picks this up does not have to re-derive them: (A) add a plaintext listener mode gated so it is structurally impossible to combine with a public bind — refuse at config load unless the bind is loopback, keep it mutually exclusive with ACME/cert paths; this was the leader recommendation, on the grounds that it makes the inner tier actually cheap as R870-F15 design claimed while keeping the risk a bounded testable invariant rather than an operator remembering not to misconfigure it. (B) keep TLS everywhere and have F23 carry a cert-issuance plus renewal story for every inner door, which is safest by construction and already proven working in R870-T18 binary-level test with an rcgen self-signed leaf, but makes every service with 2+ independently-deployed components pay a cert, a renewal and a loopback handshake per request. (C) park the tier — nothing regresses, because config 1 (bundle staging, R870-B11, in review) already covers the deploy-together case, which is the one noisetable actually needs. THIS IS THE ONLY THING BLOCKING F23 DESIGN; the join itself, both admission rules and the workload-kind vocabulary are all specified in this ticket next entries and need no further decisions.")
159//! @yah:handoff("OPERATOR CALL ANSWERED 2026-09-09: option (A), the loopback-only plaintext listener. It was re-put with ONE fact the earlier framing did not have, and that fact inverts the safety argument the three options were weighed on: option (B) was never \"already proven working\". pingora defaults verify_cert: true (pingora-core-0.8.1/src/upstreams/peer.rs:479, read not assumed) and passway NEVER overrides it — there is no verify_cert anywhere in oss/passway/crates/passway/src. R870-T18's test drove the door from an HTTP client with danger_accept_invalid_certs, not from an outer passway, so the outer-to-inner leg was untested. Since no CA issues for 127.0.0.1, \"keep TLS everywhere\" required a SECOND unbuilt change — a way to disable or pin upstream certificate verification on a public-facing door — traded for encrypting a hop that never leaves the loopback interface. (A) is strictly the smaller security surface, not merely the cheaper one.")
160//! @yah:handoff("PASSWAY: TlsMode::Plaintext, selected by PASSWAY_TLS_MODE=plaintext (a third value on the EXISTING discriminator, not a new bool env var — one variable owns the listener's TLS mode). All guards live in ONE function, tls.rs parse_listener_tls_mode, and each is a boot failure naming what to change: the bind must parse as a LITERAL loopback SocketAddr (0.0.0.0:443 — the default — is refused, and so is a hostname this process cannot prove); PASSWAY_TLS_CERT/KEY must be unset, so a configured public door cannot go cleartext by ADDING a variable rather than removing two; LISTEN_FDS is refused outright because under socket activation PASSWAY_LISTEN is only the key pingora looks the socket up by and proves nothing about the bind. An unrecognized PASSWAY_TLS_MODE is now also a boot failure instead of a silent fall-through to manual. main() reads the mode BEFORE the cert paths (plaintext has none), branches to proxy_service.add_tcp(&listen), and build_tls_settings returns Err rather than panicking on the variant it can no longer be handed.")
161//! @yah:handoff("THE VOCABULARY, and admission rule 2 made UNREPRESENTABLE rather than refused. ServiceComponent gains deploy: DeployTier { Bundle (default), Workload } — oss/yubaba/crates/cloud/src/config.rs. That is the per-component slot [providers.bundle] could not express, and because it is ONE field with two values, \"both bundle-staged and its own workload\" has no spelling at all; there is no rule to enforce. What remained checkable — two components claiming one mount — went into R870-B11's EXISTING cross_ref_validate loop rather than a parallel one. That loop previously filtered on kind and so skipped the workload tier entirely; it now covers both tiers, and only the explanation branches (bundle/bundle = one storage prefix in one bundle; workload/workload = one prefix in the inner-door table; mixed = the mount names two things serving one prefix). skip_serializing_if on the default keeps every existing service.toml byte-identical.")
162//! @yah:handoff("THE JOIN: new module oss/yubaba/crates/cloud/src/inner_door.rs. plan(&ServiceConfig, &domains) -> Result<Option<InnerDoorPlan>>. Rule 1 is by construction — None below two DEPLOYED UNITS, so the negative is assertable on the absence of a plan. Grouping is per unit but mounts are per COMPONENT: N bundle components collapse to one DeployedUnit::Bundle yet keep N mounts, because a bundle sub-mount can carry route headers the root does not and the collapse-to-root shape would silently drop them. Headers come from the DomainRoute whose route_path_prefix equals the component's normalize_mount — cross_ref_validate already PROVES those agree, so the lookup cannot mismatch. Err is reserved for one case: two-plus units with no root mount, which would 503 every unclaimed path. routes_file() refuses an unresolved upstream instead of skipping the mount — a dropped mount does not 503, it falls through to the root and serves the WRONG component with a 200. passway_mount() composes with normalize_mount rather than trimming slashes a second time.")
163//! @yah:handoff("SUPERVISION: WorkloadSpec gains files: Vec<InlineFile { path, content, mode }> (oss/yah-base/crates/workload-spec/src/lib.rs, appended last, serde(default), no skip_serializing_if — postcard is positional, so every pre-existing spec decodes to an empty vec). kamaji's NATIVE backend writes them in spawn_child BEFORE exec and on every respawn (materialize_files, oss/kamaji/crates/kamaji/src/native.rs); containerd/docker/microvm call the new kamaji::reject_unmaterializable_files and REFUSE such a spec by name rather than starting a door against a file that is not there — a silently-skipped route table comes up healthy and routes wrongly, which is worse than not starting. InnerDoorPlan::workload(listen_port, address) renders Workload::Container: argv /usr/local/bin/passway, env PASSWAY_TLS_MODE=plaintext + PASSWAY_LISTEN=127.0.0.1:<port> (the 127.0.0.1 is literal, NOT a parameter, so a wrong port cannot make the door reachable) + PASSWAY_PATH_ROUTES_FILE, and the table itself as the one InlineFile. Not a new Workload variant: TenantPasswayWorkload earns one by carrying config kamaji acts on; an inner door's whole config is an argv, three env vars and a file, so a variant would buy only exhaustive-match churn in peer-owned kamaji-proto (the R572-F1 trade).")
164//! @yah:verify("cargo test -p passway (oss/passway) = 205 lib + 43 + 35 integration, 283 passed / 0 failed, up from the 275 baseline @Ashguard:abde2cbb recorded on R870-T21. 8 new lib tests in tls::tests and 2 new integration tests in tests/path_routes_file.rs. THE END-TO-END ONE IS THE POINT: a_cleartext_inner_door_serves_the_same_mount_table_with_no_certificate forks a REAL passway binary with no PASSWAY_TLS_CERT set at all and asserts the same two-mount split and the same per-mount COOP header over plain http:// — i.e. the tier the operator authorized actually costs a process and nothing else. Its negative, a_cleartext_door_on_a_reachable_bind_refuses_to_start, spawns the binary on 0.0.0.0:0 (the DEFAULT bind, so it is the exact misconfiguration that would make an inner door a public cleartext one) and asserts a non-zero exit whose message names the bind.")
165//! @yah:verify("cargo test -p yah-cloud --lib = 1150 passed / 0 failed (11 new in inner_door::tests, 2 new in config::tests). The cheap half of this ticket's own negative is a_single_unit_service_gets_no_inner_door plus several_bundle_components_are_one_unit_and_still_get_no_door — three components sharing one bundle are still ONE unit and still get no door, which is the case that would be easy to get wrong by counting components. cargo test -p yubaba --lib = 952/0. cargo test -p kamaji --lib --all-features = 208/0 (2 new; the materialization test asserts ORDERING by having the child cat the file into a second path, not merely that the file exists). cargo test -p yah-workload-spec --all-features = 205 + 101, 0 failed. cargo test --workspace --all-features in oss/kamaji = 18+208+5+303, all green in-package.")
166//! @yah:gotcha("ONE PRE-EXISTING FLAKE, DIAGNOSED NOT WAVED THROUGH. kamaji-bin's server::tests::tenant_passway::the_list_reports_the_digest_of_the_spec_it_was_deployed_with FAILS under `cargo test --workspace --all-features` in oss/kamaji, reproducibly, and PASSES 303/303 under `cargo test -p kamaji-bin --lib --all-features` both parallel AND --test-threads=1. So it is cross-PACKAGE contention, not in-package parallelism and not this change: the failing assertion is the second deploy failing to Ack after `free_port()` (server.rs ~:8667) handed back a port another package's test binary had taken between the probe and the bind — a TOCTOU in the helper. Nothing in this ticket adds a port or touches that path; WorkloadSpec::files cannot reach it, since Workload::TenantPassway carries a TenantPasswayWorkload and no WorkloadSpec at all. Worth a real fix (bind-and-hold instead of probe-and-release) but it is not this relay's.")
167//! @yah:gotcha("ROLL ORDER MATTERS AND IS NOT THE USUAL \"JSON IGNORES UNKNOWN KEYS\" ANSWER — flagged by @Ashguard:eclipse (session:e188ccc2, R881-T6) mid-session. All three prod voters now run kamaji+yubaba 0.8.37-h5 (us-south-001 and us-west-001 rolled 2026-09-09; us-east-001 on 0.8.37-h1/h2), all built BEFORE WorkloadSpec::files existed. WorkloadSpec has no deny_unknown_fields, so on the JSON leg an un-rolled node ignores the field exactly as R870-B6's `origin` did. The postcard leg is the one that does NOT forgive: it is positional and non-self-describing, so a new yubaba encoding a spec with a trailing `files` to an old kamaji decoder is a DESYNC, not an ignored key. Before deploying any inner door, confirm which codec that node's kamaji link uses (kamaji-proto/src/codec.rs) and roll kamaji first if it is postcard. Nothing regresses until something actually SETS files — every existing spec encodes an empty vec — but the ordering is a real constraint, not a formality.")
168//! @yah:handoff("WIDER THAN THE TITLE, all mechanical and all compiler-verified. Adding two fields to types this many call sites construct exhaustively meant ~45 initializer repairs across FOUR workspaces: oss/yah-base (workload-spec + local-driver), oss/kamaji (incl. peer-owned kamaji-proto/src/codec.rs and kamaji-containerd-core), oss/yubaba, and the root (crates/yah/hub, app/yah/cli). Each is one line — `files: Vec::new(),` or `deploy: Default::default(),` — with no semantic content; they were driven off E0063 spans, not grep, so none was guessed. NOTE the sweep needs --all-features AND `cargo test --no-run`: `cargo check --all-targets` alone missed sites behind feature gates and in examples/. ALSO REGENERATED (both are pure functions of the tree, so this is not authorship): .yah/schema/{workload,service}.toml.schema.json via `cargo run -p xtask -- emit-schemas` and packages/yah/workload-spec/index.ts via the export-ts bin. Both drift gates still report red because they compare against GIT, and this camp defers commits — they go green with the commit, and the regenerated content is correct.")
169//! @yah:handoff("PHASE 1 DONE — the tier EXISTS and every piece of it is proven in isolation: the cleartext listener (proven through a forked binary), the vocabulary, the join with both admission rules, the wire carrier, and node-side materialization + restart. What is NOT done is the last hop: nothing CALLS plan() yet, so `yah cloud apply` still produces no inner door. That is deliberate rather than abandoned — it is placement work with a live-fleet verify attached, and it is the whole of phase 2.")
170//! @yah:handoff("Tree anchor at handoff: 6f984b53a9dd1a291d29fe7d4cb544b47d4f65e6 — the shared tree as I left it. Diff against it (`git diff 6f984b53a9dd1a291d29fe7d4cb544b47d4f65e6..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
171//! @yah:next("ONE DESIGN QUESTION PHASE 1 LEFT OPEN, stated so it is not rediscovered as a bug. A bundle-tier component at a non-root mount now gets its own entry in the inner-door table pointing at the SAME bundle upstream, purely so the domain manifest's per-path response headers can be applied (see a_bundle_components_sub_mount_keeps_its_headers_and_the_bundle_upstream). That is correct for headers and harmless for routing, but it means the inner door re-states routing the bundle already does internally. If the outer door or the Worker is ALREADY applying those headers for a config-1 service, the inner door would apply them twice — check which tier owns route headers for a passway front door before wiring step 4, because R746 put ROUTE_HEADERS into the Cloudflare Worker and I did not confirm the passway-front-door equivalent.")
172//! @yah:verify("THE LIVE HALF, unrun and needing a fleet: a two-component service whose components deploy independently gets one inner door from one `yah cloud apply`, with https://<host>/ and https://<host>/app/ both 200 and `curl -sI` on /app/ carrying cross-origin-opener-policy: same-origin AND cross-origin-embedder-policy: require-corp while / carries neither. The config-side half of exactly that assertion is already green as inner_door::tests::each_mount_carries_only_its_own_routes_headers, and the transport-side half as the forked-binary cleartext test — what remains unproven is only that apply joins them. THE NEGATIVE'S node-side half is also unrun: for a single-component service (yah-marketing), assert NO routes file is materialized on the node and NO extra workload appears in kamaji's table.")
173//! @yah:next("PHASE 2 IS FIVE STEPS AND EVERY INPUT ALREADY EXISTS. (1) Call inner_door::plan(&svc.service, &cfg.domains) once per service in the apply path; Ok(None) is the common answer and means do nothing at all. (2) Allocate the loopback port. It is deliberately a PARAMETER of InnerDoorPlan::workload rather than config — which port is free is a property of the node — so this is the only genuinely new decision: either take it from kamaji's ledger (oss/kamaji/crates/kamaji/src/ports.rs) or pin one per service. (3) Resolve each DeployedUnit to an address for the `address` closure: DeployedUnit::Bundle is the service's one bundle workload (R870-B11), DeployedUnit::Component(id) is that component's own workload. (4) Deploy the rendered workload in the bundle deploy sequence — BEFORE the outer door is repointed, since the door 503s until its upstream is up. (5) Repoint the outer door's PASSWAY_UPSTREAMS at 127.0.0.1:<port> instead of at the bundle, via IngressPlan::resolve_upstreams (oss/yubaba/crates/cloud/src/reconciler/ingress.rs:397).")
174//! @yah:handoff("DEFECT IN THIS TICKET'S OWN CHANGE, CAUGHT IN REVIEW BY @Ashguard:eclipse (session:e188ccc2) AND FIXED BEFORE IT LEFT THE TREE. WorkloadSpec rides the postcard `Deploy` frame (kamaji-proto/src/messages.rs:373, and V7's own stanza names `Workload::Container(WorkloadSpec)` as what that frame carries), and kamaji-proto/src/version.rs states the rule twice: every field on a postcard message is mandatory and always encoded, and the only compatibility mechanism is a ProtocolVersion bump. V2/V4/V5/V6 were each exactly \"a field appended to a struct\" and each got one. `files` is that shape and I had not bumped. The reasoning that made me miss it is the one V6's stanza already refutes: `#[serde(default)]` makes an OLD spec decode fine, so the JSON leg really is unaffected — but `default` only affects DEserialization, so a new yubaba still ENCODES a length varint an old kamaji reads as the next field and misparses from there. Now V8, CURRENT = V8, with a stanza naming the wrong reasoning rather than only the rule. cargo test -p kamaji-proto --all-features = 33/0; oss/kamaji workspace = 18+208+5+303+2+2+2+1, 0 failed.")
175//! @yah:gotcha("CORRECTION TO THE FLAKE GOTCHA ABOVE — my characterization was too narrow and would mislead the next reader, so read this one instead. I wrote that the tenant_passway digest test is \"green 303/303 in-package, fails only workspace-wide\". @Ashguard:eclipse measured the counter-example on the same tree: `cargo test -p kamaji -p kamaji-bin --lib --all-features`, in-package and parallel, failed a DIFFERENT test in the same module — deploy_arms_the_declared_socket_and_stop_releases_it (server.rs:8590) — and it failed identically before my sweep. On my own later run the workspace-wide invocation came back 303/0. So the truth is: at least two tests in server::tests::tenant_passway are intermittently flaky in BOTH configurations, the cause is `free_port()` probe-and-release losing the port between the probe and the bind (server.rs ~:8667), and it predates R870-F23. Do NOT read an in-package red there as a regression, and do not read a single green run as proof either. The fix is bind-and-hold; it belongs to neither R870 nor R881 and is unfiled — @Ashguard:eclipse tried and board.open refused for want of a parent relay.")
176//! @yah:gotcha("CONSEQUENCE OF THE V8 BUMP FOR ANY FLEET OPERATION, not just for this relay — relayed by @Ashguard:eclipse (session:e188ccc2) who is holding the fleet on R881-T6, and worth acting on before the next roll. The tree is now ProtocolVersion::V8; EVERY node runs a pre-V8 pair (us-east-001 on 0.8.37-h1/h2, us-south-001 and us-west-001 on 0.8.37-h5, the other six on 0.8.28-0.8.34). Nothing is broken, because each node is internally matched and the protocol is a node-local UDS. What changed is that `hotship --binaries yubaba` ALONE — or `kamaji` alone — is now a footgun on every node: it puts a V8 binary against a V7 sibling, and per version.rs:71 that does not fail cleanly, it misreads every field after the desync, \"which is how a wrong image or a wrong volume mount gets deployed instead of an error\". Ship the PAIR. That was harmless before this ticket and is not now.")
177//! @yah:handoff("PHASE 2 LANDED — `yah cloud apply` now produces an inner door. All five steps, with the call sites. (1) PLAN: `service_inner_door` (app/yah/cli/src/cloud.rs:9212) calls `inner_door::plan` once per service; `Ok(None)` is the answer for every service on disk today and returns before anything else runs. (2) PORT: derived, not allocated — `inner_door::listen_port(service)` (oss/yubaba/crates/cloud/src/inner_door.rs:346). (3) RESOLVE: `InnerDoorPlan::resolve_addresses` (inner_door.rs:418) maps each unit to a mesh ident via `unit_ident` (:398) and looks it up with the new `ServiceRecordFanout::address_for_ident` (reconciler/service_discovery.rs:426). (4) DEPLOY: `deploy_inner_door` (cloud.rs:9274), called at the END of the deploy-phase closure in BOTH apply paths — `reconcile_root` (cloud.rs:11877) and `handle_mirror_up` (cloud.rs:7100) — so it is after every unit registered a record and before the front-door phase repoints anything. (5) REPOINT: `IngressPlan::point_at_inner_door` (reconciler/ingress.rs:438), called from `reconcile_ingress_edge` (cloud.rs:7726).")
178//! @yah:handoff("STEP 2 ANSWERED — the port is DERIVED from the service name, not taken from kamaji's ledger, and the three facts that decided it were read rather than assumed. (a) `LedgerPorts` is node-local (a JSON file beside the supervisor's state dir) and yubaba's HTTP surface exposes no allocation verb at all — yubaba/src/lib.rs routes /workloads/*, /services, /node/*, and nothing for ports — so an apply has no way to ask. (b) A stated number is HONOURED, not rejected, on the path this workload takes: R844-F14's pin rule bites inside `LedgerPorts::resolve_set`, and `NativeRuntime::resolve_declared_ports` (oss/kamaji/crates/kamaji/src/native.rs:280) filters `pin.is_none()` BEFORE calling it. That matters because `PASSWAY_LISTEN` must carry the number, and a number the node picks after the spec is rendered cannot be in it. (c) A collision is not representable: the ledger allocates on the workload's MESH ip, an inner door binds loopback, so 100.64.0.3:14210 and 127.0.0.1:14210 are different sockets. The window is 10000-19999, deliberately below Linux's default ephemeral floor (32768) where `pick_free_port`'s bind(:0) draws from. FNV-1a written out inline rather than `DefaultHasher`, whose stability std does not promise — this number goes into a deployed door's env AND the outer door's upstream list, and a toolchain bump silently moving it would repoint one tier and not the other.")
179//! @yah:handoff("THE OPEN HEADER QUESTION IS ANSWERED, AND THE ANSWER IS NO CHANGE — grounded by reading, not assumed. The question was whether the passway FRONT door also applies per-route response headers. It does not: `PassProxy::response_filter` (oss/passway/crates/passway/src/proxy.rs:700) iterates `ctx.route_headers`, and its own doc at :694 states that vector is empty for `RoutingStrategy::ByHost` — which is what every outer door is. So the outer tier owns no headers and there is no double-apply to resolve there. A THIRD tier the question did not name does apply them, and is worth recording: the mesofact bundle ORIGIN, via `MESOFACT_ROUTE_HEADERS` set by `add_declared_route_headers` (app/yah/cli/src/cloud.rs:8612). For a mount served by `DeployedUnit::Bundle` both that origin and the inner door apply the route's headers — but CONVERGENTLY, not duplicatively: both read the same `.yah/domains` route map, `PathRouter` does not strip the mount prefix (oss/passway/crates/passway/src/path_route.rs has no strip/rewrite), so both match the same request path, and both use insert-semantics (`HeaderMap::insert` in mesofact's `RouteHeaderTable::apply`, `insert_header` in passway) — one header, one value. Do NOT collapse it to one owner. The bundle origin's coverage is strictly WIDER: it applies headers for a declared route that has no component mount (a `/docs/*` route served out of the root bundle's dist), which the inner door has no entry for. And the inner door is the ONLY owner for a `DeployTier::Workload` mount, since nothing hands such a component a header table. The two are complementary; removing either loses headers somewhere.")
180//! @yah:handoff("PLUMBING BUILT BECAUSE STEPS 3 AND 5 NEEDED IT, all three of which did not exist. (1) `inner_door::component_workload_ident(service, component_id)` (inner_door.rs:371) — the mesh identity a workload-tier component registers under. It is a NAMING RULE stated here because nothing else states it: a bundle's ident is a mirror fact (`BundleSlot::workload_name`, renameable with `name = \"...\"`), but a workload-tier component has no slot of its own, since `[providers.*]` is per-kind-per-mirror — the exact gap `DeployTier` was added to close. Folded through `reconciler::native_support::sanitize_ident`, which I widened from private to `pub(crate) mod` (reconciler/mod.rs) rather than writing a second normalizer. Getting the ident wrong fails LOUDLY: `routes_file` refuses a mount whose unit resolved to nothing, naming the unit. (2) `ServiceRecordFanout::address_for_ident` — deliberately SINGULAR where `upstreams_for` is plural. An inner door proxies over loopback to a unit on its own node; handed a fleet-wide set it would dial across the mesh, which is not what the cleartext-listener safety argument assumed. Two nodes, two addresses is ambiguity (None), not load balancing. Port selection follows `port_for`'s discipline exactly (`kamaji::DEFAULT_PORT_NAME` first, then the sole anonymous port) so a unit resolves the same way at both tiers or neither. (3) `IngressPlan::point_at_inner_door` OVERRIDES where `resolve_upstreams`/`resolve_ports` fill in — it clears both halves and then goes through those same two methods, so this stays the only place in the crate writing those fields. The ticket's step 5 named `resolve_upstreams`; used alone it is WRONG, because it skips a rule that already has an `upstream_host` and every mirror on disk pins one. A pin names ONE unit, and fronting a two-unit service from one unit serves half the site and 503s the other half, so the pin has to lose here and nowhere else.")
181//! @yah:handoff("TWO PLACEMENT DECISIONS PHASE 2 HAD TO MAKE, both recorded at the site. (a) The inner door lands on the FRONT DOORS, not the workload nodes — the outer door dials 127.0.0.1, so a door anywhere else is a door the outer tier cannot reach. `ingress_topology` (cloud.rs:9230) recomputes `resolve_ingress_placements` + `plan_ingress` in the deploy phase to learn that set; both are pure, so this costs no network and cannot disagree with the front-door phase's own answer. (b) SELF-DISCOVERY IS TURNED OFF for an inner-door service. `PASSWAY_UPSTREAM_SOURCE=yubaba` makes the door poll for the fronted workload's records and use those INSTEAD of its static set — which would route straight past the inner door to whichever unit registered under the mirror's ident, silently undoing step 5. The rendered note says so in its own words rather than reusing R844-F20's \"NOT self-discoverable ... MANUAL step\" wording, because this is not a degradation: the address is derived and byte-identical on every apply. (c) A mirror with two units and NO declared front door SKIPS with a note rather than failing — `reconcile_mirror_ingress` already returns early on `plans.is_empty()`, so there would be no outer door to repoint and nothing that can 503. Every `shape = \"local\"` dev mirror is in that state; bailing there would have broken `yah mirror up`.")
182//! @yah:handoff("DISCOVERED WORK, FIXED IN THIS PASS, NOT FILED AS A FOLLOWUP. `cargo test -p yah-cloud --lib` was 1161/2 on arrival, and the two reds were NOT mine and NOT a flake: `cloud_init::tests::{rendered_runcmd_entries_are_all_strings, coordinator_prestage_only_for_standalone}`. Cause: oss/yubaba/crates/cloud/templates/mirror.yml:107-108, the two R858-F17 turso-backup-helper runcmd entries, were written as BARE YAML scalars containing a `: ` — which makes the whole entry parse as a Mapping, so cloud-init skips it and the helpers never land on a provisioned node. The file is committed and clean (last touched by a8f0d501, i.e. it regressed AFTER phase 1's 1150/0 measurement), no live peer owns it, and the fix is two lines: double-quote the entries and escape the inner quotes. Both tests are green and the comment at the site names the gate. This is a real provisioning defect, not just a red test — a node provisioned since a8f0d501 has no turso-backup-hydrate / turso-backup-tail, and R858-F17's own design makes durability-declaring workloads refuse to deploy without them. Worth a look at whether any node was provisioned in that window.")
183//! @yah:verify("PHASE 2 MEASURED, every number run by me and read. `cargo test -p yah-cloud --lib` = 1163 passed / 0 failed (baseline 1150; +13 — 6 in inner_door::tests, 4 in service_discovery::tests, 3 in ingress::tests). `cargo test -p yah --lib` = 1549 / 0 (+3 new in a new `inner_door_apply_tests` module). `cargo test -p xtask --test main mirror_ingress` = 13 / 0 (baseline 11; +2). `cargo test -p yubaba --lib` = 952 / 0, exactly the baseline. `cargo test -p passway` in oss/passway = 205 + 43 + 37 = 285 / 0 against the 283 baseline, and `cargo test -p kamaji --lib --all-features` = 217 / 0 against 208 — BOTH deltas are peers', not mine: I touched neither crate. Sweeps: `cargo test --workspace --all-features --no-run` clean, and the same in oss/yubaba clean (only the two pre-existing unused-import warnings in a peer's in-flight mesofact_static.rs). NO SCHEMA REGEN NEEDED — this pass added functions, constants and one module-visibility widening, and no serde-visible field on any generator input, so .yah/schema/*.json and packages/yah/workload-spec/index.ts are untouched by construction.")
184//! @yah:verify("THE NEGATIVE IS ASSERTED IN THREE PLACES, at three different altitudes, because it is the claim the live fleet rests on. (1) `xtask/tests/mirror_ingress.rs::no_service_on_disk_gets_an_inner_door` walks the REAL `.yah/services/` tree and asserts every service plans `None`. That is the strongest form available without a fleet: the only way to be wrong about it is for a service to acquire `deploy = \"workload\"`, at which point the test names the service. Sibling `every_services_derived_inner_door_port_is_distinct` pins the port derivation against the real service list. (2) `cloud::inner_door_apply_tests::a_single_unit_service_leaves_the_outer_door_exactly_as_it_was` builds a two-component fixture that is BYTE-FOR-BYTE the positive test's, with one word changed (`workload` -> `bundle`), and asserts the rendered `PASSWAY_UPSTREAMS` is still the mirror's pinned `noisetable.com=100.64.0.3:8080`. So the difference between the two outcomes is provably that one field. (3) `inner_door::tests::{a_single_unit_service_gets_no_inner_door, several_bundle_components_are_one_unit_and_still_get_no_door}` from phase 1, still green. THE POSITIVE: `a_two_unit_service_repoints_the_outer_door_at_its_inner_door` (outer door renders `noisetable.com=127.0.0.1:<derived>`, and the port is asserted equal to what the door itself binds — two call sites in two phases that must not be able to disagree) and `the_rendered_table_splits_the_mounts_and_carries_only_their_own_headers` (both units addressed, COOP+COEP on /app and ABSENT on the root).")
185//! @yah:gotcha("TRANSIENT BUILD FAILURE SEEN AND DISPROVEN, recorded so the next reader does not re-chase it. The first `cargo test --workspace --all-features --no-run` came back with `can't find crate for 'runner'` / `'agent_tools'` / `'camp_service'` and a linker failing on a dozen absent `.rlib`s (libgif, libzune_jpeg, libimagesize...) in crates this ticket never touched — the exact shape CLAUDE.md's orphan-gc warning describes. Followed that procedure rather than cleaning: `cargo orphan-gc log -n 300` names NONE of the missing artifacts (every entry in the window reads `deleted 0 artifacts`), so orphan-gc is NOT confirmed here. The likelier cause is plain target-dir contention: a `yah-release-check` QED pipeline was holding the same `/Users/leif/ss/yah/target` for 29 minutes alongside this build. Re-ran with nothing else on the key: CLEAN, zero errors. Not reproducible, orphan-gc log does not name it, and the artifacts were never deleted per its own record.")
186//! @yah:next("WHAT REMAINS IS THE LIVE HALF ONLY, and it is an operator call the R870 leader is holding — phase 2 deliberately landed code + tests and touched no node. The two assertions: (a) POSITIVE — a two-component service whose components deploy independently gets one inner door from one `yah cloud apply`, with https://<host>/ and https://<host>/app/ both 200 and `curl -sI` on /app/ carrying cross-origin-opener-policy: same-origin AND cross-origin-embedder-policy: require-corp while / carries neither. (b) NEGATIVE, node-side — for a single-component service (yah-marketing), NO routes file materialized under /var/lib/passway/routes and NO extra workload in kamaji's table. Note that (b) is now also asserted statically against the real tree by `xtask/tests/mirror_ingress.rs::no_service_on_disk_gets_an_inner_door`, so the node-side check is confirmation rather than discovery. BEFORE RUNNING (a): there is no service with `deploy = \"workload\"` on disk, so one has to be declared first — and the R870-B6/V8 roll-order gotcha on this ticket applies the moment anything actually SETS `WorkloadSpec::files`. Confirm the target node's kamaji link codec (kamaji-proto/src/codec.rs) and roll the kamaji+yubaba PAIR first if it is postcard.")
187//! @yah:next("ONE THING PHASE 2 DID NOT BUILD, named so it is not mistaken for done: there is still no FLEET deploy path for a `DeployTier::Workload` component. `reconcile_component` (app/yah/cli/src/cloud.rs:8264) dispatches on `component.kind`, and the only non-bundle arms are `container` — which `ContainerReconciler::up` guards on `MirrorShape::Local`, and `LocalProcessReconciler`, which is the camp/dev tier and registers as `local-process-<service>-<env>-<component>`. So on a real mirror such a component is deployed by hand today (`yah cloud workload deploy`). That is exactly why `inner_door::component_workload_ident` had to STATE the ident rather than look it up. The failure mode is loud rather than silent — a component registered under any other ident leaves its unit unresolved and `routes_file` refuses the whole table, naming the unit — but whoever wires that deploy path must make it register under `component_workload_ident(service, id)`, or change both sides together. Related and already filed: R523-F1 (a component kind that deploys a stateful binary to a fleet node) is the same missing arm seen from the other direction.")
188//! @yah:handoff("PHASE 2 COMPLETE — `yah cloud apply` produces an inner door. Everything above this entry is the detail: the five call sites, the derived-port argument, the header-ownership answer (no change — the outer passway door owns no route headers, proven at proxy.rs:694/700, and the bundle origin's overlap is convergent and strictly wider), the three pieces of plumbing built because steps 3 and 5 needed them, the two placement decisions, and the one mirror.yml provisioning defect fixed on the way through. Nothing was deployed and no node was touched, per the dispatch. Green: yah-cloud 1163/0, yah 1549/0, yubaba 952/0, xtask mirror_ingress 13/0, passway 285/0, kamaji 217/0, both --all-features --no-run sweeps clean. Git policy is `defer`, so nothing is committed — the diff is 6 files: oss/yubaba/crates/cloud/src/{inner_door.rs, reconciler/mod.rs, reconciler/ingress.rs, reconciler/service_discovery.rs}, oss/yubaba/crates/cloud/templates/mirror.yml, app/yah/cli/src/cloud.rs, plus xtask/tests/mirror_ingress.rs.")
189//! @yah:handoff("Tree anchor at handoff: 88533e01f7f578b1520b633d05846973fa47f608 — the shared tree as I left it. Diff against it (`git diff 88533e01f7f578b1520b633d05846973fa47f608..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
190//! @yah:handoff("PHASE 2 ACCEPTED BY THE RELAY LEADER (@Ashguard:hydra, session:39386823). `yah cloud apply` now produces an inner door: all five steps wired, plus three pieces of plumbing that did not exist (`component_workload_ident`, `ServiceRecordFanout::address_for_ident`, `IngressPlan::point_at_inner_door`), the derived-port decision argued from three read facts, and the open header question answered NO CHANGE with the proof at proxy.rs:694/700. Implemented by @Ashguard:blade (session:54a6be05). The detail is in the handoff entries above this one; this entry records only that it was accepted and on what evidence.")
191//! @yah:verify("WHAT IS DELIBERATELY NOT VERIFIED, and it is the operator's call rather than an oversight: the LIVE half. No node was touched, nothing was deployed, nothing committed (git policy is `defer`). Running it needs a service with `deploy = \"workload\"` declared — none exists on disk — and the moment anything actually SETS `WorkloadSpec::files`, this ticket's own V8 roll-order gotcha binds: confirm the target node's kamaji link codec (kamaji-proto/src/codec.rs) and roll the kamaji+yubaba PAIR, never one alone.")
192//! @yah:verify("INDEPENDENTLY RE-RUN BY A SECOND COURIER (@Ashguard:dove, session:60d4f41f) who did not implement it, because a courier's self-report is the inner gate and not the outer one. All six commands reproduced the claimed counts EXACTLY: yah-cloud 1163/0 (4 ignored), yubaba 952/0, passway 285/0 (205+43+37), kamaji --lib --all-features 217/0, yah --lib 1549/0 (1 ignored), workspace --all-features --no-run clean. Every content check held: `point_at_inner_door` at ingress.rs:438; `inner_door::plan` reached from the apply path via `service_inner_door` (cloud.rs:9219) through `deploy_inner_door` (cloud.rs:9274, invoked at 7100 and 11877) and the ingress repoint at 7726; the port confirmed a deterministic per-service pin (FNV-1a into 10000-19999, inner_door.rs:346) and NOT kamaji's ledger, with the native.rs:282 `pin.is_none()` justification verified at the site. The negative is asserted three times, not once. CAVEAT ON THE MEASUREMENT ITSELF: the camp skew detector flagged 4 of 6 runs SUSPECT — peers edited kamaji/src/microvm.rs, kamaji-bin/src/main.rs and cloud/reconciler/mesofact_bundle.rs mid-run — so these are shared-tree numbers, not a frozen-tree measurement.")
193//! @yah:verify("ONE CLAIM CORRECTED AND ONE DEFECT FOUND BY THAT RE-RUN, both recorded rather than smoothed over. (1) CORRECTION: `cargo test -p yah-cloud --lib` does NOT run from the repo root — yah-cloud is not a root workspace member and needs dev-dependencies; it only works from `oss/yubaba`. Anyone reproducing the 1163/0 above must cd there first. (2) DEFECT, pre-existing and NOT caused by this ticket: `embedded_template_matches_workspace_canonical` is green VACUOUSLY — it resolves the workspace root via CARGO_MANIFEST_DIR.ancestors() to oss/yubaba, whose .yah/ holds only a .gitignore, so it takes the bootstrap branch and asserts nothing, while the repo-root twin at .yah/infra/cloud-init/mirror.yml is 128 diff-lines stale and missing the whole R858-F17 turso-backup block. FILED AS R870-B25, not left here. Note `rendered_runcmd_entries_are_all_strings` is a DIFFERENT test, is genuinely green, and is the gate that really catches the colon-space footgun this ticket fixed.")
194//! @yah:gotcha("SUPERSEDED BY R870-F27, and the @yah:next that said otherwise has been removed from this block: containerd DOES materialize WorkloadSpec::files now, so an inner door is no longer pinned to native-capable nodes. The backend-check step this ticket told its reader to perform before deploying is gone. Still refusing: Docker and MicroVm. The single fact both halves read is kamaji::Backend::materializes_files (oss/kamaji/crates/kamaji/src/lib.rs), not a list in prose.")
195//!
196//! @yah:ticket(R885-T14, "No cap:bundle-serving mesh capability exists — bundle/almanac/passway workloads are placed with nothing modelling where they can run")
197//! @yah:status(review)
198//! @yah:at(2026-09-12T07:16:50Z)
199//! @yah:assignee(agent:bundle-anthropic-ashguard)
200//! @yah:parent(R885)
201//! @yah:severity(P3)
202//! @yah:next("FOUND WHILE DISPROVING R885-T13, and it is the real gap that ticket's false premise was standing next to. `cap:native-exec` exists and is honoured (config.rs:2353-2357 via wants_native_exec) for `yah.exec = native` Container specs. But the workloads actually running on the fleet — MesofactServeBundle / Almanac / TenantPassway, served by kamaji's BundleBackend and JitRuntime — have NO corresponding mesh capability at all. All four live workloads on us-east-001 are of those kinds. So placement for the entire class of workload this fleet actually runs is unmodelled: nothing declares which nodes can serve bundles, and nothing checks. It works today because there is effectively one node doing it, which is exactly the condition under which an unmodelled constraint stays invisible. THE SHAPE OF THE FIX IS ALREADY IN THE TREE: R860-T5 closed this same gap for native-exec. Follow it — a `cap:bundle-serving` capability declared in .yah/infra/machines/*.toml, a `wants_bundle_serving` predicate beside `wants_native_exec`, and the placement check wired the same way. Establish the right granularity first: whether bundle / almanac / tenant-passway want one shared capability or separate ones is a real design question, and the answer depends on whether a node can serve one kind and not another. Tier: Cleric. ITS NATURAL HOME IS R860's AXIS, NOT R885's — R885 is about workloads running unbounded once placed, this is about where they are placed at all. Filed under R885 because that is where it was found and where the evidence is; move it to R860 if that relay is still live. Not urgent: nothing is broken today, and the cost is that the first multi-node bundle placement decision will be made by something that has no model of the constraint.")
203//! @yah:handoff("GRANULARITY SETTLED BY OPENING KAMAJI, NOT BY GUESSING — and the answer is smaller than the ticket assumed. kamaji gates each backend on its own cargo feature + startup flag: BundleBackend on `bundle-serving` + `--bundle-cache-dir` + `--bundle-origin` (`attach_bundle_backend`, oss/kamaji/crates/kamaji-bin/src/main.rs:924), JitRuntime on `tenant-passway` + `--tenant-passway-dir` (:706). Independent flags means separate capabilities, not one shared tag. But only ONE of the three named in the ticket needs modelling today: (1) BUNDLE gets `cap:bundle-serving`. (2) TENANT-PASSWAY gets nothing — there is no placement decision to gate: `yubaba::tenant_passway::reconcile_once` runs inside a node's own yubaba and drives that node's own kamaji over the local UDS, so which node arms a domain is decided by `YUBABA_TENANT_PASSWAY_STATE_DIR`, not by a selector. A tag would have no reader, which is the same wrong-fact R860-T5 refused to state for `cap:microvm`. (3) ALMANAC gets nothing ever — kamaji refuses `Workload::Almanac` outright (\"almanac and static-asset live in yubaba's reconcilers\", kamaji-bin/src/server.rs:1846) and yubaba reconciles it, so it is never node-placed.")
204//! @yah:verify("MEASURED, every number an exit-visible run. `cargo test -p yah-cloud --lib` (oss/yubaba, CARGO_TARGET_DIR=/tmp/r885t14-target) = 1205 passed / 0 failed / 4 ignored. The baseline is 1200 and it is recoverable from this session's own output rather than asserted: the first post-edit run read 1195 passed / 5 FAILED, those five being pre-existing fixtures the new axis correctly broke (synthetic machines with no `mesh_tags`), so 1200 before, +5 new tests, 1205 after. `cargo test -p xtask --test main` = 69 passed / 0 failed (the real-tree suite, incl. all 11 mirror_ingress, all 4 apex_failover, all 4 fleet_build_placement). `cargo test -p xtask --doc` green. `cargo check -p yubaba --all-targets` (oss/yubaba) clean. `cargo check -p yah --all-targets` (root) clean — flagged SUSPECT by the camp build rail (peers edited app/yah/cli/src/camp.rs, mesh.rs, oss/yah-base/crates/keys/src/spec.rs, oss/yubaba/crates/yubaba/src/lib.rs mid-run); none of those is a file this ticket touched and the CLI's only contact with the change is `resolve_bundle_machines`, whose signature is unchanged.")
205//! @yah:handoff("WHAT LANDED. (1) `pub const BUNDLE_SERVING_MESH_TAG: &str = \"cap:bundle-serving\"` in oss/yubaba/crates/cloud/src/config.rs beside NATIVE_EXEC_MESH_TAG, carrying the node-side gate, why a positive capability and not a taint, and why there is no `cap:tenant-passway` beside it. (2) oss/yubaba/crates/cloud/src/reconciler/mesofact_bundle.rs: `with_bundle_capability(&RequiredSpec) -> RequiredSpec` (idempotent) and `ensure_bundle_capable(&MachineConfig, ..)`, wired into BOTH arms of `resolve_bundle_machines` — the constraint arm gets the tag appended to the derived mesh_tags, the literal `machines = [...]` pin is refused with an error naming the tag AND the .yah/infra/machines/<name>.toml to edit. (3) oss/yubaba/crates/cloud/src/reconciler/ingress.rs: `required_for_role(role, required)` applied in both `resolve_ingress_placements` and `resolve_ingress_candidates`, so the deployer and the discovery fanout stay set-for-set across the new axis — the property resolve_bundle_machines' own doc promises and which a bundle-only check would have broken. (4) .yah/infra/machines/us-east-001.toml declares the tag. (5) W338 §Placement consequences gains item 5.")
206//! @yah:handoff("THE ARCHITECTURAL POINT, because it is why this was not one line beside the native tag. `cap:native-exec` is read in `admission_spec`, which covers `Workload::Container` and NOTHING ELSE. A bundle never reaches that function — it is placed by `resolve_bundle_machines` off the mirror's `providers.bundle` declaration, a completely separate resolver that an operator writes by hand. So the gap was not \"one more tag in the same `if`\"; it was the same class of gap one resolver over. The rule the two instances share, now written into W338: a capability tag belongs wherever a SELECTOR chooses a node, not wherever a spec is admitted. Anything that picks a machine has to know what that machine's kamaji was started with, because every kamaji backend is an opt-in flag.")
207//! @yah:handoff("THREE EXTRA FIXES, all outside the ticket title, all loud. (a) xtask/src/install.rs:717 — `clear_stale_provenance`'s doc comment had an INDENTED log excerpt, which rustdoc compiles as Rust, so `cargo test -p xtask` failed its doctest on main for everyone. Fenced as ```text. Pre-existing, unrelated to this ticket, one line. (b) .yah/schema/mirror.toml.schema.json regenerated (`cargo run -p xtask -- emit-schemas`) — the schema-drift gate was RED on arrival and the drift is @Ashguard:coffee's R584-F2 `local-mailcrab` doc-comment change on `MirrorConfig::drivers`, not mine (verified by reading the one-line diff before regenerating). Regenerated per the generated-artifacts-are-not-ownable rule; they were told. Zero of my own types are schemars-derived, so this change contributes nothing to any schema. (c) Corrected three stale claims at their sites rather than leaving them: config.rs's NATIVE_EXEC_MESH_TAG doc said `Workload::MesofactServeBundle` and `Workload::Almanac` are BundleBackend-served (MesofactServeBundle is a FIELD on Workload::MesofactStatic, not a variant; Almanac is never node-placed at all), and us-east-001.toml's R885-T13 block repeated the same three wrong names for the four processes it observes — all four are bundle workloads.")
208//! @yah:verify("NEW TESTS. Five in mesofact_bundle's `mod tests`, each keeping a capable AND an incapable node in one CloudConfig so a pass provably comes from the capability rather than an empty pool: a_node_without_the_bundle_backend_cannot_be_pinned_to_serve_a_bundle (refusal names the tag and the file; the SAME fleet + same declaration shape at a capable node still resolves); a_constraint_places_past_a_matching_node_that_cannot_serve_bundles (the incapable node is declared FIRST, so a resolver ignoring the axis returns it and the test fails; then dropping the capable node makes the same declaration refuse); the_ingress_planner_applies_the_same_capability_as_the_deployer (set-for-set, plus the candidate widener must not widen past the capability); a_mirror_that_declares_the_capability_itself_is_unchanged (idempotence); a_non_bundle_slot_requires_no_capability (regression guard on the role mapping — a static slot still places on a node with no bundle backend). The shared `machine()` fixture now carries the tag by default, with `machine_without_bundle_backend()` beside it, because every placement test there presupposes an eligible pool.")
209//! @yah:verify("REAL-TREE HALF, which is where the live-safety answer is. New xtask/tests/mirror_ingress.rs::exactly_one_machine_in_the_real_fleet_can_serve_a_bundle asserts the capable set over .yah/infra/machines/ is exactly [\"us-east-001\"] and that a `replicas = 2` bundle constraint therefore refuses NAMING the tag. It is a deliberate tripwire: the day a second node declares the capability, the \"one node serves every bundle\" reasoning scattered through yah-marketing's and noisetable's mirrors stops holding and this is what says so. The pre-existing a_constraint_with_replicas_two_… had to change — its scale-2 assertions need two capable nodes and the real fleet has one — so it now grants the capability to us-south-001/us-west-001 IN MEMORY, asserts first that neither declares it on disk (so the grant cannot go silently vacuous), and says at the site that this is a hypothetical fleet, not a claim about those boxes. Every one of its original assertions (repel-by-default moving the second slot, the toleration lever recovering the pre-B7 answer, absent-replicas being exactly one, the short-count refusal, the two-backend render through collate) is unchanged and green.")
210//! @yah:gotcha("THIS FAILS CLOSED AND THE BLAST RADIUS WAS CHECKED, NOT ASSUMED. A bundle can now only be placed on a node declaring `cap:bundle-serving`, and exactly one does. Every live bundle placement in reach still resolves: yah-marketing's `required = { regions = [\"us-east\"], mesh_tags = [\"tag:cloud-runner\"] }` lands on us-east-001 (green in the real-tree suite), and the noisetable camp's two `providers.bundle.machines = [\"us-east-001\"]` pins are covered because ~/ss/noisetable/.yah/infra/machines/ is an EMPTY DIRECTORY — that camp borrows this repo's inventory read-only through its .yah/infra/sources.toml [[source]] link, so declaring the tag here fixes it there too and no cross-camp edit was needed. NOT EXECUTED, and say so rather than implying it: I did not run `yah cloud validate`/`ingress collate` from ~/ss/noisetable against a rebuilt binary. That conclusion is read off the two mirrors' pins plus the sources.toml link, not observed.")
211//! @yah:assumes("us-east-001's `cap:bundle-serving` rests on a BEHAVIOURAL reading, not on that box's ExecStart line: it is the node every bundle in the fleet is placed on today, R870-T9 records noisetable.com serving live through a bundle from it, and R885-T13 observed four kamaji-forked bundle processes there on 2026-09-11. Nothing in-repo records `--bundle-cache-dir` on its kamaji command line. The tag is therefore right about what the box demonstrably does and unverified about how it was started; if a roll ever drops the flag the tag becomes a lie in the fail-OPEN direction (a deploy that is admitted and then refused — i.e. exactly today's behaviour, not worse). Settle it with `ps -o args= -C kamaji` / the kamaji.service ExecStart, the same evidence us-west-001.toml:61-67 records for the native tag.")
212//! @yah:cleanup("us-south-001 is the obvious second bundle-capable candidate — it already shares the passway demux pair with us-east-001 — and is deliberately left undeclared because nothing in-repo reads its kamaji flags either way. One `ps -o args= -C kamaji` on that box settles it; declaring it wrongly would route a bundle to a node that refuses it, which is the failure this axis exists to remove. Until then the fleet is genuinely single-node for bundle serving and exactly_one_machine_in_the_real_fleet_can_serve_a_bundle says so out loud.")
213//!
214//! @yah:relay(R926, "Register iroh relay + headscale as first-class camp services, with descriptions and HA-aware health checks")
215//! @yah:status(review)
216//! @yah:at(2026-09-20T22:13:30Z)
217//! @yah:assignee(agent:bundle-anthropic-ashguard)
218//! @arch:see(.yah/docs/working/W122-yah-mobile.md)
219//! @yah:handoff("FILED 2026-09-17 from an operator request made during R726-S22's device run (@Ashguard:dove, session:639e7b78). THE ASK, in the operator's words: the iroh relay and headscale should be listed in the yah services with DESCRIPTIONS and HEALTH CHECKS, \"since they can move around in HA mode\". GROUNDED STATE OF THE WORLD AT FILING, read rather than assumed: (1) The service registry is .yah/services/<name>/service.toml. Exactly nine services are registered - scrabcake, yah-analytics, yah-chat, yah-cloud, yah-cloud-admin, yah-cr, yah-dashboard, yah-desktop, yah-marketing. NEITHER an iroh relay NOR headscale is among them, so the operator's observation is correct. (2) The generated schema .yah/schema/service.toml.schema.json has top-level properties [components, db, domain, health_path, name, schema_version], required [domain, name, schema_version]. So a health HOOK already exists in the shape of `health_path` - this is NOT greenfield - but there is NO `description` field at all, which is new schema surface. (3) .yah/infra/machines/*.toml are pure INVENTORY (name, [allocatable], [connect], [registration]); they are not where a service is declared, so do not add it there. (4) The Rust side of health_path lives in oss/yubaba/crates/cloud/src/config.rs and src/lib.rs (also mirrored in oss/mesofact/crates/mesofact/src/lib.rs).")
220//! @yah:next("Wire the iroh relay explicitly rather than implicitly. R726-S22 measured a phone falling back to relay because there was no shared address family with the camp; the relay is the DESIGNED path in that case, not a failure - but today nothing in the service registry says the relay exists, so its health is unobservable from the camp UI.")
221//! @yah:next("Add a `description` field to the service schema (new surface - it does not exist), then REGENERATE the artifacts: `cargo run -p xtask -- emit-schemas` and the workload-spec export. They no longer regenerate on commit and schema-drift-guard in the `check` QED pipeline will fail the build otherwise.")
222//! @yah:next("HA is the actual hard part and deserves a design decision before code: `health_path` is a single path on a single declared domain, which cannot express \"this service currently lives on whichever of N machines won the election\". Decide whether a service gains a set of candidate endpoints with a liveness winner, or whether the registry queries the mesh's own service-records endpoint (the 100.64.0.3:7443/service-records?ready=true surface referenced in us-west-001.toml) as the source of truth. Do NOT bolt a second health mechanism beside health_path - see the repo's below-v1.0.0 rule; change the one that exists.")
223//! @yah:gotcha("HEADSCALE HAS A LOUD, DOCUMENTED FAILURE HISTORY AND THIS TICKET IS PARTLY A RESPONSE TO IT - read R858 before designing the health check. .yah/infra/machines/us-west-001.toml carries R858 (\"Mesh coordination outage: cloud.mesh.yah.dev refuses :443, so no camp machine can reach any 100.64.0.0/10 address\") plus a measured 2026-09-04 gotcha: tailscale reported \"fetch control key ... connect: connection refused\", port 22 answered while 80/443 were REFUSED, and consequently every mesh address stopped answering. THE PART THAT MATTERS FOR A HEALTH CHECK: the same gotcha records that THE PUBLIC SITE STAYED GREEN THROUGHOUT (yah.dev HTTP 200 in 0.81s) because the apex serves from us-east-001's own passway and never traverses the coordination server. So a naive HTTP health check against a public domain would have reported HEALTHY during a total mesh outage. Whatever check this ticket adds for headscale MUST probe the coordination path itself (the control-key fetch, or a mesh-address dial), not a public endpoint that is up for unrelated reasons. The repo's CLAUDE.md also warns that the headscale appliance already accumulated four half-owners of \"does this node have a config.yaml\" and that the seam between two of them took the mesh down twice - so give this ONE owner.")
224//! @yah:handoff("PHASE 1 LANDED — the `description` field exists and the artifacts are regenerated. oss/yubaba/crates/cloud/src/config.rs: ServiceConfig gains `description: Option<String>` (serde default + skip_serializing_if, so the nine pre-existing services still load). 28 struct-literal construction sites updated across config.rs(13), inner_door.rs, pond.rs(5), cloudflare_worker.rs, derive_cache_prune.rs, local_process.rs, mesofact_static.rs, static_asset.rs, static_asset_prune.rs, sync_status.rs, reconciler/mod.rs, tests/whisper_derive_e2e.rs. Two new tests: a_declared_description_survives_the_loader and a_service_without_a_description_still_loads. `cargo run -p xtask -- emit-schemas` re-run; .yah/schema/service.toml.schema.json now carries the field and scripts/check-schema-drift.sh exits 0 (\"ok: .yah/schema is in sync with the Rust types\"). packages/yah/workload-spec/index.ts was NOT regenerated because it did not change — this edit touches no workload-spec source.")
225//! @yah:verify("cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib = 1255 passed / 0 failed / 4 ignored, EXIT=0 (/tmp/r926_full_e2.log). scripts/check-schema-drift.sh = exit 0.")
226//! @yah:gotcha("THE IROH-RELAY HALF OF THIS TICKET RESTS ON A THING THAT DOES NOT EXIST: yah OPERATES NO IROH RELAY. Measured 2026-09-20. (a) mshr's default is n0's PUBLIC relay — oss/mshr/crates/mshr/src/discovery.rs:82 default_relays() returns vec![N0_DNS_PKARR_RELAY_PROD], imported from iroh at discovery.rs:32. (b) The capability to self-host is BUILT AND UNUSED: `mshr::relay::Server` exists with a full ACME/TLS builder (oss/mshr/crates/mshr/src/relay.rs:258-398) and its own round-trip + pebble tests, and grep for `relay::Server|RelayServer|relay_server` across oss/ crates/ app/ returns 13 hits that are ALL the library itself or its own tests — zero production callers, and none at all in crates/yah/, app/yah/ or oss/yubaba/. (c) .yah/infra/workloads/ contains exactly ONE file, yah-cloud-admin.toml. (d) grep for a relay URL across .yah/ returns nothing. So R726-S22's phone \"falling back to relay\" fell back to n0's public infrastructure. Registering that in .yah/services/ — the registry `yah cloud apply` RECONCILES — would declare a domain yah does not own and components it does not deploy. That is a scope question, not a defensible default, which is why it is an operator call below.")
227//! @yah:gotcha("THE HA OPTION THIS TICKET'S OWN next() FAVOURED IS STRUCTURALLY BLOCKED, and the blocker is already documented in-tree. The filing said to consider \"querying the mesh's own service-records endpoint (100.64.0.3:7443/service-records?ready=true) as the source of truth\". SERVICE RECORDS ARE STRICTLY PER-NODE: app/yah/cli/src/mesh.rs:94 records the invariant from service_records.rs's module doc — \"A record's `mesh_ip` must equal the answering node's own mesh address\" (R844-B11) — and `reconcile` rebuilds the set from that node's OWN ContainerRuntime::list_workloads(). THERE IS NO CLUSTER-WIDE SERVICE VIEW. So asking any one node where headscale lives answers only \"on me\" or \"not on me\". Confirmed live in the record: on 2026-09-08 the coordinator MOVED from us-west-001 to us-south-001 (.yah/infra/machines/us-west-001.toml:36), which is exactly the event a health check must survive and exactly the one a per-node query cannot see. Corollary: the R858 gotcha's warning still governs — a check against a public domain reports HEALTHY during a total mesh outage, because yah.dev serves from us-east-001's own passway and never traverses the coordination server.")
228//! @yah:gotcha("TWO SMALL TRAPS FOR THE NEXT EDITOR OF THIS CRATE, both cost me a build. (1) `cargo check -p yah-cloud` FROM THE REPO ROOT reports `error[E0433]: cannot find module or crate serde_yaml` — this is NOT a missing dependency and NOT a vanished artifact. serde_yaml IS declared at oss/yubaba/crates/cloud/Cargo.toml:143 under [dev-dependencies]. oss/yubaba is an independent workspace excluded from the yah root (see CLAUDE.md \"Co-developed OSS repos\"), so from the root the crate is a non-member path dep via [patch.crates-io] and cargo applies no dev-dependencies to it; cargo says so plainly if you use `test` instead of `check` (\"package yah-cloud cannot be tested because it requires dev-dependencies and is not a member of the workspace\"). Correct invocation: `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib`. Also note the package is `yah-cloud`, not `cloud` — `cloud` is the [lib] name only (Cargo.toml:6 vs :31). (2) A LITERAL `@yah:` INSIDE A RUST DOC COMMENT IS STRIPPED OUT OF THE GENERATED JSON SCHEMA, from the marker to the end of that paragraph. My first draft of ServiceConfig::description's doc said \"...whichever W### doc or `@yah:` annotation last touched the thing\", and emit-schemas produced a description truncated mid-sentence at the backtick. Spell it \"board annotation\" in prose on any type that feeds .yah/schema/.")
229//! @yah:handoff("PHASE 2 LANDED — headscale is registered, described, and probed on the coordination path. NEW FILE .yah/services/headscale/service.toml: name=\"headscale\" (matching the service-record ident leader.rs already upserts), domain=\"cloud.mesh.yah.dev\", description set, health_path=\"/key?v=138\". NO components and NO mirrors/ — deliberately: the appliance is placed by the mesh leader (app/yah/cli/src/mesh.rs start_headscale), not by `yah cloud apply`, so this file makes the service observable without claiming its deployment. Verified that shape cannot break the camp's config load by READING the loader: load_services gates mirrors on `if mirrors_dir.exists()` (config.rs:2909) so a missing dir yields an empty map, and both cross_ref_validate bail-loops iterate `&svc.mirrors` and `&svc.service.components` — empty, so zero iterations. The only service-level check is health_path.starts_with('/'), which the query string satisfies. Pinned with a new test, an_observability_only_service_loads_without_components_or_mirrors, so it stays true.")
230//! @yah:handoff("THE HA QUESTION IS ANSWERED FOR HEADSCALE, AND THE ANSWER IS \"health_path IS ALREADY THE RIGHT SHAPE\" — no second mechanism, per the below-v1.0.0 rule. The filing assumed one domain cannot express \"lives on whichever of N machines won the election\". For this service it can, because the thing that moves and the thing the registry names are different things: the appliance really did migrate us-west-001 -> us-south-001 on 2026-09-08 (R858 handoff), but cloud.mesh.yah.dev terminates at a passway front door on ALL THREE doors (R858-T1, mesh.rs:141), so placement is already abstracted behind one stable address and failover is the doors' job. THE PROBE PATH IS THE PART THAT HAD TO BE RIGHT, and all three measurements were taken 2026-09-20: cloud.mesh.yah.dev/key?v=138 -> 200 in 0.23s; cloud.mesh.yah.dev/ -> 404; yah.dev/ -> 200 in 0.38s. So BOTH obvious choices are wrong and each fails in the direction that hurts — the schema default of \"/\" reports headscale BROKEN while it is healthy, and any public yah domain reports it HEALTHY during a total mesh outage (the R858 false-green, still live: measurement three). /key?v=138 is the control-key fetch, the first call every tailscaled makes and the exact request whose failure opened R858, so nothing but the coordination path can serve it.")
231//! @yah:next("The iroh relay is UNBUILT pending an operator call — see the gotcha: yah operates no relay, so there is nothing to register until someone decides between registering a dependency on n0's public relay and standing up mshr::relay::Server as a real yah service.")
232//! @yah:next("SEPARABLE FOLLOWUP, deliberately not done here: the nine pre-existing services (scrabcake, yah-analytics, yah-chat, yah-cloud, yah-cloud-admin, yah-cr, yah-dashboard, yah-desktop, yah-marketing) still carry no `description`. The field is optional so they all load, but a registry where only headscale is described is half a feature. NOT attempted in this pass because writing nine descriptions for services I had not read would be fabrication — each needs its own service.toml and mirrors read first. Worth one ticket that does all nine at once.")
233//! @yah:verify("cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib = 1256 passed / 0 failed / 4 ignored, EXIT=0 (/tmp/r926_full_h5.log), including all three new tests. Live probes 2026-09-20: cloud.mesh.yah.dev/key?v=138 = 200 (0.23s), cloud.mesh.yah.dev/ = 404, yah.dev/ = 200 (0.38s).")
234//! @yah:gotcha("CORRECTION TO THIS TICKET'S OWN RELAY GOTCHA, prompted by the operator (\"I thought we did serve a relay?\") — they were right and my framing was wrong. THERE ARE TWO DIFFERENT RELAYS and the filing conflated them. (1) The iroh/NAT-TRAVERSAL relay (`--xlb-relay` / $YAH_XLB_RELAY): we run none. oss/yubaba/crates/yubaba/src/main.rs:627 says plainly \"unset ships n0's production relays\", and mshr::relay::Server still has zero production callers. That part of the earlier gotcha stands. (2) THE PUSH RELAY: we absolutely do run one, it is ours, and it is a far better fit for this ticket than the iroh relay ever was. crates/yah/push-relay/ is a real crate with its own daemon (src/bin/yah-push-relay.rs), serving ALPN \"yah/push-relay/1\" (protocol.rs:32); yubaba takes --push-relay-node-id whose doc says \"this is normally the relay co-located on this machine\" (main.rs:544-548). It holds the FCM service-account credential and is what makes a phone buzz when a gate is raised (W122 §Push, R726-F7).")
235//! @yah:gotcha("THE PUSH RELAY IS THE CASE health_path GENUINELY CANNOT EXPRESS — and unlike headscale, here the registry really is the wrong shape. It has NO HTTP SURFACE AT ALL: src/bin/yah-push-relay.rs binds an iroh endpoint and serves the ALPN, and grep for health/healthz/axum/http across server.rs + the bin returns nothing but the `.bind()` call at :72. It is reached by hex NodeId over QUIC, so it has no domain and no path — while ServiceConfig REQUIRES `domain` and health_path is defined as \"relative to domain\". Registering it today would mean inventing a `.invalid` domain the way yah-cloud already had to (see .yah/services/yah-cloud/service.toml's note on the R546-S2 domain-requirement tension), and then having no way to probe it. THIS is the real \"a second health mechanism vs change the one that exists\" decision the filing was reaching for — it just belongs to the push relay, not to headscale. The honest shape is probably a dial-the-ALPN liveness check keyed on NodeId, which means `domain` stops being the only way to address a service. That is design work plus an operator call, not a config edit.")
236//! @yah:handoff("SCOPE DELIVERED, AND ONE HALF DELIBERATELY SPLIT OUT. Done: the `description` schema field (new surface, 28 call sites, 3 new tests, artifacts regenerated, drift guard green) and headscale registered at .yah/services/headscale/service.toml with a description and a health check that probes the coordination path itself. NOT done, and split to R926-F1: registering a relay. The filing asked for \"the iroh relay\", which yah does not run — but the operator corrected me that we DO serve a relay, and they are right: it is the PUSH relay (crates/yah/push-relay/), which is ours and is the thing that should be registered. It cannot be registered today because it has no HTTP surface and ServiceConfig requires a `domain` — the genuine schema-shape problem this ticket was reaching for, plus a live operator call on one-shared-relay vs one-per-camp. See the gotchas for both.")
237//! @yah:verify("cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib = 1256 passed / 0 failed / 4 ignored, EXIT=0 (/tmp/r926_full_h5.log). scripts/check-schema-drift.sh = exit 0, \"ok: .yah/schema is in sync with the Rust types\". Live probes 2026-09-20 proving the health check discriminates: cloud.mesh.yah.dev/key?v=138 = 200 (0.23s), cloud.mesh.yah.dev/ = 404 (so the schema default of \"/\" would have reported a false RED), yah.dev/ = 200 (the R858 false-GREEN, still live).")
238//!
239//! @yah:ticket(R931-B2, "yah cloud validate is a false green for the component mount/route cross-check")
240//! @yah:status(review)
241//! @yah:at(2026-09-22T01:05:59Z)
242//! @yah:assignee(agent:bundle-anthropic-miravel)
243//! @yah:parent(R931)
244//! @yah:handoff("Tier: Cleric — a validator gap; the check exists, the surface does not run it. MUTATION-PROVED 2026-09-21, not inferred. ServiceComponent::mount's own doc says a component mounted at /app \"must be routed at /app or /app/*, because a disagreement means requests land on a prefix nothing published to — a 404 whose cause is two files apart\", and names CloudConfig::cross_ref_validate as enforcing it \"at config load\". THE TEST: with a component declaring mount = \"/issues\" and the domain manifest routing \"/issues\" + \"/issues/*\", the mount was mutated to \"/issuez\"; the mutation was confirmed PRESENT by grep (count 1) before the run; `yah cloud validate` still printed \"ok — no alias or port collisions, no inert taints, no retired arch tags, ...\" and exited 0. Reverted by hand, grep count 0 afterwards. So validate does cover alias/port/taint/arch/ingress collisions — it caught a genuine two-upstream ingress collision on the same hostname in the same session, loudly and correctly — but it does NOT run the mount/route cross-check. WHY IT MATTERS MORE THAN IT LOOKS: the failure mode the doc describes is a 404 on a prefix nothing published to, discovered in production, with the two disagreeing strings in different files. `validate` is the obvious place to catch it and currently reports clean. NOT ESTABLISHED: whether cross_ref_validate is simply not called by the validate subcommand, or is called and the mount arm is skipped — I did not read the validate entry point, only the field docs and the loop near config.rs:1773.")
245//! @yah:handoff("FIXED at oss/yubaba/crates/cloud/src/config.rs:1863 (CloudConfig::cross_ref_validate). Removed the `if !matches!(route.mode, RouteMode::Static { .. }) { continue; }` gate that silently skipped the mount/route cross-check for every mode except Static. The loop was already narrowed to component-referencing modes one screen up (`let Some(component_ref) = route.mode.component() else { continue }` — RouteMode::component() returns Some only for Static and Backend, None for StaticBucket and Redirect), so deleting the Static-only re-gate makes the check fire for Backend too while StaticBucket/Redirect remain correctly exempt (neither references a component at all).")
246//! @yah:verify("Baseline (pre-fix, gate reverted by hand via Edit, not git): `cargo test -p yah-cloud --lib` from oss/yubaba = 1266 passed, 0 failed, 4 ignored, exit 0. After fix + 2 new tests: 1268 passed, 0 failed, 4 ignored, exit 0 (the +2 are the new tests, confirmed by name in output: config::tests::a_backend_mode_mount_that_disagrees_with_its_route_path_is_rejected and config::tests::a_backend_mode_route_matching_its_mount_loads, both ok). `cargo build -p yah-cloud` from oss/yubaba: exit 0, no new warnings (one pre-existing unrelated warning in yah-object-store::parse_list_v2). Not committed — .yah/git-policy is `defer`, so the diff is left uncommitted at HEAD f7daeb07 for the camp's git sweep.")
247//! @yah:gotcha("Investigated whether widening to Backend could false-positive against a legitimate use: R898-F3's documented pattern of a Backend route proxying `/api/issues` to an origin's `/issues` via `origin_path` rewrite, where public path and origin path deliberately diverge. Traced route_table.rs::resolve_mode/backend_origin: Backend's resolved origin comes from `component_workload_ident` (direct mesh-address lookup), and any path rewrite is via `origin_path`/route.path — `mount` is never consulted there. So the widened check compares `component.mount` against `route.path`'s prefix (same as Static), which is orthogonal to `origin_path` and does not conflict with the rewrite pattern. Confirmed empirically: grepped the whole repo (`.yah/domains/*.toml`) and the whole yah-cloud test suite for `mode = \"backend\"` — zero hits anywhere, so no existing fixture or real config exercises this path today, and none broke.")
248//! @yah:assumes("Did not touch app/yah/cli/src/cloud.rs (owned by a sibling courier this relay) — handle_validate() already calls CloudConfig::load() correctly per the original diagnosis, so no change was needed there.")
249//! @yah:verify("INDEPENDENTLY RE-VERIFIED by a separate courier against the current tree (not the implementing courier's self-report). cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib = 1268 passed / 0 failed / 4 ignored, matching the claim exactly. Both new tests confirmed present and passing by name: config::tests::a_backend_mode_mount_that_disagrees_with_its_route_path_is_rejected and config::tests::a_backend_mode_route_matching_its_mount_loads. The Static-only re-gate is confirmed GONE from cross_ref_validate at config.rs:1863, with the loop now narrowing via RouteMode::component() upstream instead. cargo build -p yah exit 0.")
250
251use anyhow::{bail, Context, Result};
252// R870-F26: an `[[ingress]]` edge embeds the renderer's own auth type rather
253// than a config-side copy of its five fields — see `IngressEdge::auth`.
254use local_driver::passway_ingress::{AuthSpelling, PasswayAuth};
255use serde::{Deserialize, Serialize};
256use std::collections::BTreeMap;
257use std::path::Path;
258use thiserror::Error;
259use workload_spec::secrets::SecretAccess;
260use workload_spec::sovereign::Membership;
261pub use workload_spec::sovereign::SovereignRole;
262use workload_spec::{validate, LifecycleArchetype, Locality, WorkloadSpec};
263
264/// Static node capacity declaration on `machine.toml` (R572-F3).
265///
266/// `memory_mb` and `cpu_millis` express the node's *total* hardware budget.
267/// F5's bin-packer subtracts the sum of committed workload requests from
268/// this floor to determine available headroom; an absent `allocatable`
269/// block means no capacity constraint is enforced (any workload fits).
270#[derive(Debug, Clone, Serialize, Deserialize)]
271#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
272pub struct NodeAllocatable {
273 /// Total physical RAM in mebibytes (e.g. 512 for a 512 MB node).
274 pub memory_mb: u32,
275 /// Total CPU in k8s millicores (1000 = 1 core, 250 = 0.25 CPU).
276 pub cpu_millis: u32,
277}
278
279/// `[registration]` — facts **observed** about a running box, written by the
280/// fleet rather than declared by an operator (R707-T1).
281///
282/// The rest of `machine.toml` is *declaration*: intent, operator-authored,
283/// reviewed and diffed like any other source. This block is the other half —
284/// what the box turned out to be once it booted and joined. Keeping the two
285/// apart is what lets the published fleet index (R707-F3) say which half it is
286/// carrying; publishing them under one schema would bake the confusion into a
287/// permanent record.
288///
289/// The split is a **provenance** boundary, not a trust or reach one:
290/// - *Declaration* answers "what did we ask for" — `name`, `region`, `arch`,
291/// `mesh_tags`, `[allocatable]`, and the declared reach in [`ConnectSpec`].
292/// - *Registration* answers "what did we observe" — the hostkey TOFU'd at
293/// attach, the mesh address headscale assigned at join.
294///
295/// It stays in the git-tracked TOML on purpose. Registration is not local
296/// scratch state: every consumer needs the mesh address to dial a node, so it
297/// has to travel with the declaration. (`.yah/infra/state/machines/<name>.json`
298/// — [`crate::state::MachineState`] — remains the *gitignored* sidecar for
299/// provider-side derivatives that nobody but this camp needs.)
300#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq, Eq)]
301#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
302pub struct MachineRegistration {
303 /// Yubaba's ed25519 `/identity` fingerprint, TOFU-recorded by
304 /// `yah cloud machine attach` on first contact (`SHA256:…`). An observed
305 /// property of a running process — not the operator's intent — which is
306 /// why it moved out of the top level here.
307 #[serde(default, skip_serializing_if = "Option::is_none")]
308 pub hostkey_fingerprint: Option<String>,
309 /// Mesh (headscale/tailnet) IPv4 assigned at join, e.g. `"100.64.0.1"`.
310 /// Bare address, not a URL: the *port* is declared reach and lives on
311 /// [`ConnectSpec::yubaba_port`]. [`MachineConfig::yubaba_url`] composes the
312 /// two. Absent until the node has joined the mesh.
313 #[serde(default, skip_serializing_if = "Option::is_none")]
314 pub mesh_ipv4: Option<String>,
315 /// RFC3339 timestamp of the mesh join that produced `mesh_ipv4`. Free-form
316 /// audit; nothing keys off it.
317 #[serde(default, skip_serializing_if = "Option::is_none")]
318 pub joined_at: Option<String>,
319}
320
321impl MachineRegistration {
322 /// True when nothing has been observed yet — used to omit the whole
323 /// `[registration]` table from a serialized machine TOML.
324 pub fn is_empty(&self) -> bool {
325 self.hostkey_fingerprint.is_none() && self.mesh_ipv4.is_none() && self.joined_at.is_none()
326 }
327}
328
329/// Per-machine TOML from `.yah/infra/machines/<name>.toml`.
330///
331/// Two halves, split by provenance (R707-T1): everything here is *declaration*
332/// — operator intent under review and blame — except [`registration`], which
333/// carries what the fleet observed. See [`MachineRegistration`] for why the
334/// boundary is drawn there and what depends on it.
335///
336/// @yah:ticket(R860-T5, "Model per-node native-exec capability as an admission axis (W338 §Placement consequences 3 / R858-T4 gap)")
337/// @yah:status(review)
338/// @yah:phase(P1)
339/// @yah:at(2026-09-05T18:29:19Z)
340/// @yah:assignee(agent:bundle-anthropic-ashguard)
341/// @yah:parent(R860)
342/// @yah:next("Cheapest defensible shape: express it on MachineConfig, which already has the two vocabularies — `mesh_tags: Vec<String>` (config.rs:246, superset match, already carries `arch:`/`os:`/`tag:build-worker`) and `taints: Vec<String>` (config.rs:337). A `native-exec` mesh tag required by any group member whose kind is native is a one-line admission axis in `admission_spec()`. Whichever is chosen, it must be declared in .yah/infra/machines/*.toml for the nodes that actually run kamaji with --native-exec-dir, and `check_inert_taints` (config.rs:703) lints unread taint keys dead — so a taint nobody reads will be flagged.")
343/// @yah:verify("cargo test -p cloud --lib config")
344/// @arch:see(.yah/docs/working/W338-workload-dependencies-and-appliance-composition.md)
345/// @yah:depends_on(R860-T4)
346/// @yah:gotcha("Verified 2026-09-04: native-exec capability is modelled NOWHERE in placement — `rg \"native\" oss/yubaba/crates/cloud/src/config.rs` returns zero hits, and the raft state machine models no member attributes, labels or taints at all (`rg \"taint|capabilit|labels|mesh_tag\"` over raft/{mod,store,network}.rs yields one unrelated comment at raft/store.rs:591). Native-exec is a node-local kamaji startup decision today: `--native-exec-dir` (oss/kamaji/crates/kamaji-bin/src/main.rs:152-156, :51-55) plus the `native-exec` cargo feature (kamaji-bin/src/server.rs:329-330). A node without it refuses the deploy at dispatch time and nothing upstream can see that in advance — which is exactly the deploy-time surprise W338 wants turned into a placement precondition.")
347/// @yah:handoff("NATIVE-EXEC IS NOW A PLACEMENT PRECONDITION, NOT A DISPATCH-TIME SURPRISE. New `pub const NATIVE_EXEC_MESH_TAG: &str = \"cap:native-exec\"` in oss/yubaba/crates/cloud/src/config.rs (declared just above `node_selector_mesh_tags`), and one axis in `admission_spec()` immediately after the R860-T4 group loop: if ANY member of `placement_group(ws, declared)` returns true from `WorkloadSpec::wants_native_exec()`, the tag is appended to the derived `RequiredSpec.mesh_tags` (deduped). No new field on `RequiredSpec`, no signature change anywhere, no wire or serde change — the mesh_tags axis is already an AND-ed superset check against `machine.mesh_tags` in `matches` and is already rendered by `describe`, so a refusal now reads `required.mesh_tags=[...,cap:native-exec]`.")
348/// @yah:handoff("ITEM 1 — HOW A NATIVE WORKLOAD IS DETECTED, settled by opening the type rather than guessing. There is no `kind` on `WorkloadSpec`: on the wire a native workload is still `Workload::Container(WorkloadSpec)`, and the ONLY difference is the annotation `yah.exec = native`, read through `WorkloadSpec::wants_native_exec()` (oss/yah-base/crates/workload-spec/src/lib.rs:2939; consts `NATIVE_EXEC_ANNOTATION` / `NATIVE_EXEC_VALUE` at :3402/:3407). That accessor is what the admission axis calls — matching kamaji, whose `deploy_container` checks the same marker first and routes to `deploy_native_exec` (oss/kamaji/crates/kamaji-bin/src/server.rs). The `yah.exec` key is a substrate selector with a second value, `microvm` (`wants_microvm`, same key, R605-F8), so per-node microVM capability is the obvious sibling axis and is NOT modelled here — see next-steps.")
349/// @yah:handoff("ITEM 2 — DECLARATIONS LANDED ON TWO NODES, FROM READINGS RECORDED IN-REPO, NOT INFERRED. `cap:native-exec` added to `mesh_tags` in .yah/infra/machines/us-west-001.toml and .yah/infra/machines/us-west-003.toml, each with a comment naming its evidence and its re-check condition. us-west-001: the R858 gotcha in its own header records a `ps` reading taken on the box 2026-09-05 — pid 515908 is `/usr/local/bin/kamaji --native-exec-dir /var/lib/yah/kamaji/native`, supervising headscale as a native child. us-west-003: its header's 'THE DEPLOYED KAMAJI PREDATES THE microVM BACKEND' note quotes the box's actual ExecStart, read over ssh 2026-09-01, carrying `--native-exec-dir /var/lib/yah/kamaji/native` (corroborated by .yah/docs/architecture/A043-yah-on-machine-daemons.md's @yah:verify for the same probe). Both comments say plainly that the capability lives in the systemd unit's ExecStart, not in the TOML, so it must be re-checked after any roll.")
350/// @yah:handoff("ITEM 2, THE NEGATIVES — TWO NODES ARE KNOWN NOT TO HAVE IT AND WERE DELIBERATELY LEFT UNSET. us-south-001: kamaji refused headscale there 2026-09-03 with 'native backend not configured — start kamaji with --native-exec-dir' (the R858 chain, quoted in .yah/infra/machines/us-west-001.toml and W267). I did NOT edit us-south-001.toml — it was already dirty in the working tree at the anchor SHA and @Ashguard:eclipse is live on R858, so I left it alone rather than race it; the mechanism fails closed there, which is the correct state. us-west-015 (the sole darwin builder): W254-darwin-build-nodes.md's own next-step records that its kamaji is built/started `--docker` only. I added a comment to us-west-015.toml explaining that the tag is deliberately absent, that this is the node where the axis changes an error message (a darwin build row is native by construction, so it is now refused at ELECTION naming cap:native-exec instead of reaching the box and being refused by kamaji), and the exact enable sequence: rebuild with `--features native-exec`, restart with `--native-exec-dir <dir>`, THEN add the tag. us-west-002/011/013/014 are unestablished from the repo and left unset. THE OPERATOR-FACING ANSWER: the file is `.yah/infra/machines/<node>.toml` and the key is `mesh_tags`; add the literal string `cap:native-exec` to that array, and only after the roll.")
351/// @yah:handoff("DECISIONS THE BRIEF LEFT OPEN, all recorded in doc comments at the site. (1) MESH TAG, NOT TAINT — as recommended, and the doc says why in the terms the brief asked for: mesh tags are positive capability with superset matching ('this node CAN'), which is the claim being made; a taint is repulsion and would have to be inverted to `no-native-exec` on every node LACKING the backend (declaration burden on the majority, and silently wrong for a node nobody has edited) AND taught to `taint_effect`, or `check_inert_taints` would correctly lint the key dead. (2) THE `cap:` NAMESPACE IS NEW. Live prefixes are `tag:` (operator-assigned role), `arch:`/`os:` (silicon and userland facts, emitted as requirements by qed::platform::build_worker_mesh_tags), and `tier:` which R763 RETIRED for architecture and reserved for the environment axis — so reusing any of them would have stated the wrong kind of fact. A capability the daemon was configured with is none of those. Nothing validates tag prefixes (only `check_retired_arch_tags` looks at one), so this costs no wiring. (3) COMPUTED OVER THE GROUP, not the requirer — that is literally W338's sentence ('supply = self specs must be placeable where their requirer lands'), and the second test proves it: an ordinary container requirer with a `local` edge to a native provider is pulled onto a capable node. (4) FAILS CLOSED, accepted deliberately: an undeclared node is simply not a candidate, so an undeclared fleet reports 'no node admits' at election rather than dispatching to a node that refuses. Nothing in `.yah/infra/workloads/` is native-marked today (only yah-cloud-admin.toml exists there), so the only live consumer is the qed darwin build row, where failing closed is strictly the better error.")
352/// @yah:handoff("BLAST RADIUS, MEASURED. `admission_spec` is private and its callers are unchanged: `admit_workload` / `admit_workload_candidates` / `admit_workload_in_group` (config.rs), reached from app/yah/cli/src/cloud.rs (deploy, rolling, topology analyzer), app/yah/cli/src/yubaba_client.rs `elect_node`, and cloud/src/migrate.rs. The headscale appliance path inside yubaba (headscale_appliance.rs) does NOT go through admission — it is node-internal — so nothing eclipse holds on R858 is touched by this. Files edited, in full: oss/yubaba/crates/cloud/src/config.rs; .yah/infra/machines/{us-west-001,us-west-003,us-west-015}.toml. Nothing in oss/yubaba/crates/yubaba/ was opened, and oss/kamaji/crates/kamaji-bin/src/server.rs was READ ONLY (to confirm the marker check), per @Ashguard:hydra's contention triage.")
353/// @yah:handoff("ONE SCOPE ADDITION, stated loudly rather than slipped in: `MachineConfig::mesh_tags` (config.rs:256) had NO doc comment at all — the operator-facing declaration key for four tag namespaces was undocumented. I gave it one enumerating `tag:` / `arch:`+`os:` / the new `cap:` / retired `tier:`, and noting that nothing validates the prefix (which is why the two lints exist). CONSEQUENCE TO KNOW: that field's doc is the source of the `mesh_tags` description in the GENERATED .yah/schema/machine.toml.schema.json, so it is schema-drift-affecting — see the gotcha.")
354/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I found it and left it. Quote this SHA rather than 'HEAD' in any revert/restore instruction; to undo a hunk, read it with `git show 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2:<path>` and put it back with Edit, never `git checkout`/`restore` (they restore whole files and would delete peers' uncommitted work).")
355/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
356/// @yah:next("MICROVM IS THE IDENTICAL UNMODELLED GAP, one line away. `yah.exec` is a substrate selector with a second value: `WorkloadSpec::wants_microvm()` (workload-spec/src/lib.rs, R605-F8), and kamaji constructs MicroVmRuntime only when started with `--microvm-dir` — A043's probe records that us-west-003's deployed kamaji has `--native-exec-dir` but NOT `--microvm-dir`, so a microvm-marked deploy is refused there by exactly the same dispatch-time surprise this ticket removed for native. The shape is `cap:microvm` alongside NATIVE_EXEC_MESH_TAG in the same `if` in `admission_spec`. Not done here because no node in the fleet can host one yet (R605-F14 must land a guest kernel + rootfs first), so declaring the tag anywhere today would be the wrong fact.")
357/// @yah:next("us-south-001 needs `cap:native-exec` DECIDED, not defaulted, and it is the R858 node. It is the one machine the repo positively records as LACKING the backend (kamaji refused headscale there 2026-09-03), so leaving the tag off is correct TODAY — but if R858's fix is 'give us-south-001 a native-capable kamaji' rather than 'stop moving headscale', then the roll and the tag must land together, in that order. I left .yah/infra/machines/us-south-001.toml untouched because it was already dirty at anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 and @Ashguard:eclipse is live on R858.")
358/// @yah:next("R860-T6 (`supply = \"self\"` provisioning) inherits this for free — `admission_spec` already requires the capability of the whole group, so a self-provisioned native member cannot be elected onto a node that cannot run it. What T6 must still not do is re-elect per member: reuse the node URL `elect_node` returned for the requirer, per R860-T4's handoff.")
359/// @yah:verify("BASELINE RECORDED BEFORE EDITING, at tree anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2: `cargo test -p yah-cloud --lib` from oss/yubaba = 1090 passed / 0 failed / 4 ignored, exit 0 — exactly the count the brief predicted. AFTER: 1093 passed / 0 failed / 4 ignored, exit 0 (+3, exactly the three tests added). `cargo check -p yah-cloud --all-targets` exit 0 and `cargo check -p yubaba --all-targets` exit 0 (yubaba consumes cloud, so it is where any signature change would surface — there is none). Every exit code echoed explicitly via an `EXIT=$?` / `${PIPESTATUS[0]}` marker and read back, never inferred from an empty grep. The four `yah-cloud` warnings are all pre-existing and in other files (object-store r2.rs, reconciler/mesofact_static.rs unused imports, app_manifest.rs, reconciler/mod.rs non_snake_case); config.rs contributes none.")
360/// @yah:verify("NEW TESTS (config.rs `mod tests`, R860-T5 section at the end, after the R860-T4 block). (1) a_node_without_the_native_exec_capability_cannot_host_a_native_workload — a `yah.exec = native` spec is refused by a bare node with an error naming `cap:native-exec`, and admitted by a node declaring it, with both nodes in the same fleet so the choice is provably the tag. (2) a_local_edge_to_a_native_provider_makes_the_requirer_need_the_capability — an ordinary container requirer (asserted `!wants_native_exec()`) with a `local` edge to a native provider lands on the capable node, while the SAME spec without the edge still lands on the plain one, so the constraint provably comes from the group. (3) a_group_with_no_native_member_does_not_require_the_capability — the regression guard: the axis is absent from `admission_spec`'s mesh_tags and a group with a local edge between two ordinary specs still admits on a node declaring nothing. Helper `native_spec()` asserts the marker reads back through `wants_native_exec()` before the test uses it, so a typo cannot make the test pass vacuously.")
361/// @yah:verify("Machine-config lints were considered and are unaffected by construction: `check_inert_taints` reads `taints` (I touched none), and `check_retired_arch_tags` flags only the `tier:` prefix. `cap:` is a new namespace and nothing validates prefixes, so no lint fires and no lint needs teaching.")
362/// @yah:gotcha("SCHEMA DRIFT IS EXPECTED FROM THIS TICKET AND WAS ALREADY RED BEFORE IT. `.yah/schema/machine.toml.schema.json` is generated from `cloud::config` by `cargo run -p xtask -- emit-schemas`, and MachineConfig's DOC COMMENT is what the generator emits as its `description` — which means (a) my new `mesh_tags` doc changes it, and (b) so does this very handoff, because R860-T5's @yah: annotation block lives inside MachineConfig's doc at config.rs:201. That file was ALSO already dirty in the working tree at anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2, before I touched anything — `scripts/check-schema-drift.sh` compares the regenerated tree against git, so it is red for any uncommitted schema edit regardless of author. Regenerate with `cargo run -p xtask -- emit-schemas` (or `scripts/check-schema-drift.sh --update`) when the root target dir is not contended; the pre-commit hook no longer does it (disabled 2026-08-15, see CLAUDE.md).")
363/// @yah:verify("SCHEMA REGENERATED IN THIS SESSION, so the drift gate is not left for the next reader: `cargo run --quiet -p xtask -- emit-schemas` exit 0, run from the repo root after the handoff was written (so it captures the annotation text too). Two files moved. `.yah/schema/machine.toml.schema.json`: MachineConfig's `description` grows by this ticket's annotation block, plus a genuinely new `mesh_tags.description` from the doc comment I added. `.yah/schema/workload.toml.schema.json`: +104 lines that are NOT mine — the `Locality` / `Requirement` / `Supply` / `WorkloadSpec.requires` types R860-T1 landed had never been emitted, so the sibling ticket's schema drift was still outstanding and my regen swept it in. Derived artifacts are not ownable (shared-tree doctrine), so this is deliberate rather than accidental; @Ashguard, whoever picks up R860-T1's review should know the schema now describes `requires`.")
364/// @yah:verify("FINAL RE-RUN AFTER THE HANDOFF ANNOTATION WAS WRITTEN INTO config.rs (the board write edits MachineConfig's doc block, so the file changed under the earlier green): `cargo test -p yah-cloud --lib` = 1093 passed / 0 failed / 4 ignored, exit 0. Unchanged. Note for anyone reading the camp build rail's skew warnings on this session: the one `SUSPECT RESULT` it emitted names `oss/yubaba/crates/cloud/src/config.rs` as modified mid-run, and that modification was MY OWN board_handoff annotation write, not a peer — the two authoritative runs (full lib test, and both cargo checks) each came back `Input closure unchanged across the whole run: no skew`.")
365/// @yah:verify("All builds were run with `CARGO_TARGET_DIR=/tmp/r860t5-target` rather than the shared oss/yubaba/target, following R860-T4's recorded gotcha — a peer (session:83093d9d) held the shared target lock for the entire session (20+ minutes of `cargo check -p yubaba --lib`). Costs one cold dep build, then every subsequent run is seconds. Worth reaching for immediately when the queue message says you are behind someone.")
366/// @yah:handoff("LEADER RE-VERIFIED (session:69b18855, independent of the courier's self-report). `cargo test -p yah-cloud --lib` from oss/yubaba: 1093 passed / 0 failed / 4 ignored, exit 0, against the 1090/0/4 baseline this relay's own T4 established — +3 = exactly its new tests. Axis confirmed by content: `NATIVE_EXEC_MESH_TAG = \"cap:native-exec\"` at config.rs:2304, appended to the derived `RequiredSpec.mesh_tags` at :2112-2114 when any `placement_group` member returns true from `WorkloadSpec::wants_native_exec()`. No new `RequiredSpec` field, no signature change, no wire change — it rides the existing AND-ed superset check, so a refusal now reads `required.mesh_tags=[...,cap:native-exec]`.")
367/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
368/// @yah:verify("MACHINE DECLARATIONS AUDITED FOR PROVENANCE, because a wrong capability declaration is worse than an absent one. Both are traceable to measurements ALREADY RECORDED IN-REPO, not inferred: us-west-001 from the `ps` reading at us-west-001.toml:21 (pid 517125, ppid 515908 = `/usr/local/bin/kamaji --native-exec-dir /var/lib/yah/kamaji/native`, cgroup `0::/yubaba.slice/kamaji.service/native`, 2026-09-05); us-west-003 from the actual ExecStart read over ssh 2026-09-01 at us-west-003.toml:141. us-west-015 was deliberately left WITHOUT the tag and carries enable instructions at :207-217 — unknown fails closed, which is the correct direction. No node was guessed at and nothing was probed live.")
369/// @yah:handoff("THIS TICKET MODELS THE EXACT DRIFT THAT CAUSED THE 25-HOUR MESH OUTAGE, which is worth stating because it turns an abstract W338 bullet into a measured one. us-west-001.toml:8 records the root-cause chain: on 2026-09-03T06:03:03Z leadership moved to us-south-001, which tried to deploy headscale and kamaji refused — \\\"workload requests native host execution (yah.exec=native) but no native backend is available (native backend not configured — start kamaji with --native-exec-dir)\\\" — then the systemd fallback failed too, both at WARN, and the mesh had no coordination server for 25 hours. us-west-001.toml:10 names it explicitly as \\\"a silent per-node capability drift that placement does not model\\\". After this ticket, placement models it: a group needing native exec can no longer be admitted onto a node that has not declared `cap:native-exec`.")
370/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
371/// @yah:verify("RE-VERIFIED AT HEAD 00ee20d1 (session:aa5e882d, 2026-09-05). `NATIVE_EXEC_MESH_TAG` present in oss/yubaba/crates/cloud/src/config.rs (declared above `node_selector_mesh_tags`, appended to the derived `RequiredSpec.mesh_tags` when any `placement_group` member wants native exec). `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` = 1141 passed / 0 failed / 4 ignored, exit 0.")
372/// @yah:cleanup("cap:microvm remains the identical unmodelled axis, one line from done in the same `if` in `admission_spec`. Deliberately NOT taken: no node in the fleet can host a microvm until R605-F14 lands a guest kernel + rootfs, so declaring the tag today would assert a false fact. Do it when R605-F14 lands, not before.")
373#[derive(Debug, Clone, Serialize, Deserialize)]
374#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
375pub struct MachineConfig {
376 pub name: String,
377 pub provider: String,
378 /// Who the hardware actually comes from (`"ovh"`, `"vultr"`, `"on-prem"`).
379 ///
380 /// Deliberately *not* [`provider`](Self::provider), which selects the
381 /// auto-provision driver: a box we rented by hand and brought up over SSH
382 /// is `provider = "static"` for its whole life, and writing the vendor
383 /// there instead would flip it driver-backed and make
384 /// [`validate`](Self::validate) demand `location` + `server_type` it has no
385 /// answer for. The two axes genuinely differ — vendor is who bills you,
386 /// `provider` is who yah can call an API against.
387 ///
388 /// Worth recording because vendor-scoped policy is invisible in every other
389 /// field and decides real work: outbound port 25, rDNS/PTR control, IP
390 /// reputation, egress billing. It survived only in TOML prose until now,
391 /// which made it ungreppable at exactly the moment you need it.
392 #[serde(default, skip_serializing_if = "Option::is_none")]
393 pub vendor: Option<String>,
394 /// Human label for the box (`"gamer"`, `"the GEEKOM"`). Free-form and never
395 /// matched on — [`name`](Self::name) stays the identity everywhere. This is
396 /// only so operators and agents can say which box they mean out loud.
397 #[serde(default, skip_serializing_if = "Option::is_none")]
398 pub nickname: Option<String>,
399 /// Provider DC code (e.g. Hetzner `"hil"`). **Provisioning-only**: required
400 /// iff the provider has an auto-provision driver ([`provider_has_machine_driver`]);
401 /// a BYO `static` node we brought up over SSH has no such code. Optional at
402 /// load time so static machine.tomls omit it; [`MachineConfig::validate`]
403 /// enforces presence at the right moment for driver-backed providers.
404 #[serde(default, skip_serializing_if = "Option::is_none")]
405 pub location: Option<String>,
406 /// Provider SKU/size (e.g. Hetzner `"ccx13"`). Provisioning-only, same
407 /// optionality contract as [`location`](Self::location).
408 #[serde(default, skip_serializing_if = "Option::is_none")]
409 pub server_type: Option<String>,
410 /// **Deprecated (R330-F16).** A machine should describe *itself* (region,
411 /// zone, provider, mesh_tags); *which* mirrors run on it is derived by the
412 /// reconciler from each mirror's `required` placement spec, not declared
413 /// here. Now optional + omitted-when-empty so new machine.tomls leave it
414 /// out. The legacy `resolve_mirror_machine` topology fallback still reads
415 /// it until yubaba's reverse-index supersedes the topology.toml path; once
416 /// that lands, this field and its readers are removed wholesale.
417 #[serde(default, skip_serializing_if = "Vec::is_empty")]
418 pub hosts_mirrors: Vec<String>,
419 /// Positive placement facts about this node, matched as a **superset**:
420 /// a workload is admitted only where every tag it requires is present, so
421 /// adding a tag can only ever make a machine match more, never fewer.
422 ///
423 /// Four namespaces are live, and they are not interchangeable:
424 /// - `tag:<role>` — a role the operator assigns (`tag:build-worker`,
425 /// `tag:qed`, `tag:cloud-runner`, `tag:mac-builder`);
426 /// - `arch:<x86|arm>` / `os:<linux|darwin>` — facts about the silicon and
427 /// userland, emitted as *requirements* by
428 /// [`qed::platform::build_worker_mesh_tags`];
429 /// - `cap:<capability>` — something the node's daemons were configured to
430 /// be able to do. Today just [`NATIVE_EXEC_MESH_TAG`] (R860-T5);
431 /// - `tier:` is **retired** for architecture (R763) and reserved for the
432 /// environment axis — [`crate::validate::check_retired_arch_tags`]
433 /// flags a machine still carrying `tier:<arch>`.
434 ///
435 /// Nothing validates the prefix, which is why the lint above exists: a tag
436 /// nobody requires is silently inert, and a *stale* one silently stops
437 /// matching and reports "no node" rather than "wrong tag".
438 pub mesh_tags: Vec<String>,
439 /// Canonical geo region label (latency axis), e.g. `"us-west"`. F16's three
440 /// topology axes are orthogonal: `region` = geo (latency), `zone` = failure
441 /// domain within a region (HA), `provider` = network/cost. `region` is
442 /// distinct from `location` (the provider's DC code, e.g. Hetzner `"hil"`):
443 /// `location` is provider-scoped, `region` is our provider-neutral label.
444 /// Optional for backward-compat; a machine without it never satisfies a
445 /// `required.regions` constraint.
446 #[serde(default, skip_serializing_if = "Option::is_none")]
447 pub region: Option<String>,
448 /// Failure-domain label within a region (HA axis), e.g. `"hil"`. For
449 /// single-DC Hetzner this typically mirrors `location`. F16 placement
450 /// matches `required.zones` against this. Optional for backward-compat.
451 #[serde(default, skip_serializing_if = "Option::is_none")]
452 pub zone: Option<String>,
453 /// Declared CPU architecture (`"x86_64"` / `"aarch64"`). A machine has
454 /// exactly one — it's a first-class property of the box, not a reach
455 /// detail and not a mesh tag. Drives the yubaba release triple. Optional
456 /// only because there's no provider API to probe it (static nodes declare
457 /// it; a driver-backed provider may leave it unset until known).
458 #[serde(default, skip_serializing_if = "Option::is_none")]
459 pub arch: Option<String>,
460 pub bucket: Option<BucketSpec>,
461 /// **Legacy location, superseded by `[registration].hostkey_fingerprint`**
462 /// (R707-T1). Still deserialized so machine TOMLs written before the split
463 /// keep parsing; never *read* directly — go through
464 /// [`MachineConfig::hostkey_fingerprint`], which prefers the registration
465 /// block. [`MachineConfig::normalize`] folds this into `registration`, and
466 /// [`MachineConfig::save`] normalizes before writing, so a load→save cycle
467 /// migrates the file rather than dropping the value.
468 #[serde(
469 rename = "hostkey_fingerprint",
470 default,
471 skip_serializing_if = "Option::is_none"
472 )]
473 pub legacy_hostkey_fingerprint: Option<String>,
474 /// Provider-side SSH-key IDs (Hetzner: from `GET /v1/ssh_keys`)
475 /// authorized for `root` at create time. Defaults to empty for
476 /// backwards-compat with existing machine declarations; an empty
477 /// list yields a Hetzner-emailed random root password (which the
478 /// driver currently discards). Populate this when you want pre-mesh
479 /// SSH access for bootstrap deploys or recovery.
480 #[serde(default, skip_serializing_if = "Vec::is_empty")]
481 pub ssh_keys: Vec<u64>,
482 /// Cloudflare Tunnel ID this machine joins (e.g. `abc123.cfargotunnel.com`).
483 /// `None` → no tunnel (mesh-only node, no public ingress).
484 /// When set, `yah cloud machine provision` reads `cloudflare-tunnel-token`
485 /// from the keys vault and injects the cloudflared install block into
486 /// cloud-init so the new machine connects to CF edge on first boot.
487 #[serde(default, skip_serializing_if = "Option::is_none")]
488 pub cloudflared: Option<String>,
489 /// Provider-issued floating/reserved IP that follows **public-ingress
490 /// ownership** onto this box — R859-F2 (W267 §Tier 1).
491 ///
492 /// The value is the provider's own identifier, opaque here and interpreted
493 /// only by the matching adapter: a Hetzner numeric floating-IP id as a
494 /// string, an OVH Additional-IP address (`"51.81.85.200"`), a Vultr
495 /// reserved-IP UUID. Same "the adapter is the boundary" convention
496 /// [`crate::envoy::floating_ip::FloatingIpAssignInput::ip_id`] documents.
497 ///
498 /// # Why it lives on the machine
499 ///
500 /// [`crate::envoy::floating_ip`] shipped the `floating_ip.*` verbs and
501 /// three provider adapters with no config anywhere saying *which* floating
502 /// IP is "the" ingress IP — the gap R594-F5 recorded and deliberately left.
503 /// This is that field, and it sits beside [`cloudflared`](Self::cloudflared)
504 /// on purpose: that is already the per-node "how the world reaches this
505 /// box" handle, and a floating IP is the sovereign-tier answer to the same
506 /// question. `[[ingress]]`'s
507 /// [`tunnel_id`](crate::config::IngressEdge::tunnel_id) is the *service*
508 /// side of ingress identity — which cohort a given service fronts through —
509 /// and a floating IP is not per-service: one IP moves between boxes, so it
510 /// cannot be partitioned by slot or hostname.
511 ///
512 /// # Absent means "no floating-IP path", never an error
513 ///
514 /// Most machines have none, and that is the normal case: mesh-only nodes,
515 /// boxes behind a Cloudflare tunnel, and every provider without a
516 /// floating-IP adapter. The effector skips such a machine cleanly rather
517 /// than refusing — see
518 /// [`plan_ingress_owner_effect`](crate::provider::floating_ip::plan_ingress_owner_effect).
519 ///
520 /// # The cohort has to agree
521 ///
522 /// Every machine that can hold the same ingress IP must declare the *same*
523 /// id: the IP is one resource that moves, so two ids inside one
524 /// [`sovereign_group`](Self::sovereign_group) means an ownership flip
525 /// silently reassigns a *different* IP than the one currently serving
526 /// traffic. `yah cloud validate` refuses that
527 /// ([`crate::validate::check_ingress_floating_ip`]) rather than leaving it
528 /// to be discovered during a failover.
529 #[serde(default, skip_serializing_if = "Option::is_none")]
530 pub ingress_floating_ip: Option<String>,
531 /// When `true`, this machine hosts operator-bridge workloads (Tailscale
532 /// operator access to mesh-internal services). `yah cloud machine provision`
533 /// will install tailscaled and run `tailscale up` during cloud-init via the
534 /// `{{OPERATOR_BRIDGE_BLOCK}}` placeholder. Defaults to `false` for
535 /// backward-compat with existing machine declarations.
536 #[serde(default)]
537 pub hosts_operator_bridge: bool,
538 /// BYO `static`-node reach descriptor. Static nodes have no provider API to
539 /// probe, so how the camp reaches them (SSH user@host + the yubaba URL,
540 /// which is loopback until the WireGuard mesh lands) is *declared* here.
541 /// `None` for driver-backed providers (Hetzner/Vultr), whose address is
542 /// resolved from the provider API / mesh at provision time.
543 #[serde(default, skip_serializing_if = "Option::is_none")]
544 pub connect: Option<ConnectSpec>,
545 /// Static node capacity (R572-F3). Declares the node's total hardware
546 /// budget; F5's scheduler subtracts committed workload requests from this
547 /// to check whether a new workload fits. Absent means unconstrained.
548 #[serde(default, skip_serializing_if = "Option::is_none")]
549 pub allocatable: Option<NodeAllocatable>,
550 /// Placement taint keys (R572-F3). A repelling key blocks placement by
551 /// default, and a placement opts back in by naming that exact key in
552 /// [`RequiredSpec::tolerates`] (R876-B7).
553 ///
554 /// The `unless` is real now. It was not between R742-T4 and R876-B7: the
555 /// spec side declared which archetypes it *was* rather than which taints it
556 /// tolerated, that field was `#[serde(skip)]`, and so every placement
557 /// declared as `required = {...}` in a mirror TOML read this list as empty
558 /// and could not be drained at all. See [`RequiredSpec::tolerates`].
559 ///
560 /// A key in this list influences placement in exactly one of two ways, and
561 /// [`taint_effect`] is the authority on which:
562 ///
563 /// - **repulsion** — `"no-server"` / `"no-appliance"` / `"no-job"` reject
564 /// any placement that does not tolerate them. The archetype in the key is
565 /// now vocabulary rather than a filter: `matches` does not compare it
566 /// against the workload's class, it checks the toleration list, and
567 /// [`admission_spec`] is what turns a workload's class into the
568 /// tolerations that reproduce the old archetype-scoped behaviour;
569 /// - **affinity** — a key in [`AFFINITY_TAINT_KEYS`] (today just
570 /// `"public-ip"`) that a workload names in
571 /// `yah.placement.requires-taint`, which then *requires* this node.
572 ///
573 /// Anything else is **inert**: it parses, it round-trips, and no scheduler
574 /// decision can ever read it. `yah cloud validate` rejects such keys
575 /// (`validate::check_inert_taints`) rather than letting them sit looking
576 /// load-bearing — which is how `no-voter` spent months asserting a
577 /// falsehood on three nodes. Facts about a node that are not placement
578 /// inputs belong in [`mesh_tags`](Self::mesh_tags) or a comment.
579 #[serde(default, skip_serializing_if = "Vec::is_empty")]
580 pub taints: Vec<String>,
581 /// Which consensus group this node belongs to — W305/R742-F1. `None` means
582 /// standalone: in no group at all, which is us-west-002 and us-west-015.
583 ///
584 /// Membership is not by itself quorum eligibility; that is
585 /// [`sovereign_role`](Self::sovereign_role), added by R605-F12 because
586 /// us-west-003 is in prod's blast radius *and* must never vote in it.
587 ///
588 /// **Not a placement input.** It is deliberately absent from
589 /// [`RequiredSpec::matches`], and adding it there would be a category
590 /// error: a sovereign group is a *blast radius*, not a filter. Nothing
591 /// about "which quorum does this box vote in" should decide where a
592 /// workload runs — that is what made the fleet express three unrelated
593 /// properties through one taint list and get all three wrong (W305).
594 ///
595 /// What it *is* for is refusal. [`judge_join`] answers "may this node join
596 /// that node's cluster", and the answer is no unless both declare the same
597 /// group. Before this field the only guard was a comment in three machine
598 /// TOMLs saying "never run a raft join against this box from a shell
599 /// pointed at prod" — habit, with no mechanism behind it, which is the
600 /// same class of guard W257 §8 admitted to.
601 ///
602 /// # Why `sovereign_group` and not `raft_group`
603 ///
604 /// Raft is today's mechanism (operator, 2026-08-10). A field named for the
605 /// mechanism goes stale the day the mechanism is swapped, and every
606 /// consumer that reads it inherits the lie. `sovereign` names what the
607 /// group *has* — its own authority, its own upgrade cadence, its own
608 /// destruction — which stays true under any consensus protocol.
609 ///
610 /// Note the word already appears in this tree as prose (W267's title, the
611 /// `IngressProvider::Passway` doc comment's "sovereign edge"). That is an
612 /// adjective meaning "self-hosted, not SaaS"; this is the first time it
613 /// carries structure.
614 #[serde(default, skip_serializing_if = "Option::is_none")]
615 pub sovereign_group: Option<String>,
616 /// Whether this node may hold a seat in its group's quorum — R605-F12.
617 /// Meaningless without [`sovereign_group`](Self::sovereign_group): a
618 /// standalone box has no quorum to be eligible for.
619 ///
620 /// **`None` is "not written", not a third role.** Read it through
621 /// [`sovereign_membership`](Self::sovereign_membership), which resolves the
622 /// absence to [`SovereignRole::Voter`] — what declaring a group has always
623 /// meant, so the six nodes stamped before this field keep their seats
624 /// without an edit. The distinction is kept only so
625 /// [`crate::validate::check_unroled_sovereign_members`] can tell an
626 /// operator who *chose* voter from one who never considered the question;
627 /// no join decision reads the `Option` directly.
628 ///
629 /// # Why this is not a taint
630 ///
631 /// It was, once: `no-voter` sat in [`taints`](Self::taints) on three nodes
632 /// for months, read by nothing, and R742-T4 removed it because the taint
633 /// list is a *placement* vocabulary and this is not a placement input (see
634 /// [`taint_effect`]). Nor is it a second group label. It is a modifier on
635 /// the membership this node already declares, which is why it lives beside
636 /// the group and is judged with it in one predicate,
637 /// [`workload_spec::sovereign::join_permitted`].
638 #[serde(default, skip_serializing_if = "Option::is_none")]
639 pub sovereign_role: Option<SovereignRole>,
640 /// `[registration]` — the observed half (R707-T1). Empty until the box has
641 /// been attached / mesh-joined. See [`MachineRegistration`].
642 #[serde(default, skip_serializing_if = "MachineRegistration::is_empty")]
643 pub registration: MachineRegistration,
644}
645
646/// True iff `provider` has an auto-provision driver (create/destroy via API).
647/// Driver-backed providers require `location` + `server_type`; BYO `static`
648/// nodes (brought up over SSH) do not. The cloud-vs-vps distinction the fleet
649/// cares about lives here — at the provider-capability layer — not as a
650/// separate machine type (W242 BYO Phase-0 decision).
651pub fn provider_has_machine_driver(provider: &str) -> bool {
652 matches!(provider, "hetzner" | "vultr" | "digitalocean")
653}
654
655/// Taint keys a workload may name in `yah.placement.requires-taint` to
656/// *require* a node (W305/R742-T4 affinity vocabulary).
657///
658/// This is a closed list on purpose. `WorkloadSpec::requires_taint` returns
659/// free text, but every producer in the tree is code — `passway_ingress.rs`
660/// and `cloudflared_ingress.rs`, both emitting
661/// [`workload_spec::PUBLIC_IP_TAINT`] — and no on-disk `workload.toml` sets the
662/// annotation at all. So the set of keys a node can usefully carry for
663/// affinity is knowable at compile time, which is what lets
664/// [`taint_effect`] call anything outside it inert instead of guessing.
665///
666/// **Adding an affinity key means adding it here**, in the same change that
667/// teaches a workload to require it. That coupling is the point: it makes the
668/// node side and the workload side impossible to land apart.
669pub const AFFINITY_TAINT_KEYS: &[&str] = &[workload_spec::PUBLIC_IP_TAINT];
670
671/// How a key in [`MachineConfig::taints`] can affect placement.
672///
673/// W305 finding 1: before R742-T4 nothing asked this question, so a key that
674/// no scheduler path could read — `"qa"`, `"no-voter"` — parsed, validated,
675/// and quietly did nothing. Both of the findings that cost real fleet state
676/// were invisible for exactly that reason.
677#[derive(Debug, Clone, Copy, PartialEq, Eq)]
678pub enum TaintEffect {
679 /// `"no-<archetype>"`: rejects placement outright unless the constraint
680 /// names this key in [`RequiredSpec::tolerates`]. Read by
681 /// [`RequiredSpec::matches`], which walks `machine.taints` and classifies
682 /// each key through [`taint_effect`] (R876-B7).
683 Repels(LifecycleArchetype),
684 /// A key in [`AFFINITY_TAINT_KEYS`]: a workload naming it in
685 /// `yah.placement.requires-taint` is restricted to nodes carrying it.
686 Attracts,
687 /// Neither. No placement decision can read this key.
688 Inert,
689}
690
691/// Classify one node taint key. See [`TaintEffect`].
692///
693/// The repulsion half is derived from [`LifecycleArchetype::ALL`] rather than
694/// a literal list, so a fourth archetype makes `no-<its key>` live without an
695/// edit here.
696pub fn taint_effect(key: &str) -> TaintEffect {
697 if let Some(arch) = LifecycleArchetype::ALL
698 .into_iter()
699 .find(|a| key == format!("no-{}", a.taint_key()))
700 {
701 return TaintEffect::Repels(arch);
702 }
703 if AFFINITY_TAINT_KEYS.contains(&key) {
704 return TaintEffect::Attracts;
705 }
706 TaintEffect::Inert
707}
708
709/// Every key the scheduler *can* act on, sorted — for error messages that
710/// tell the operator what the legal vocabulary actually is instead of only
711/// what was wrong.
712pub fn live_taint_keys() -> Vec<String> {
713 let mut keys: Vec<String> = LifecycleArchetype::ALL
714 .into_iter()
715 .map(|a| format!("no-{}", a.taint_key()))
716 .chain(AFFINITY_TAINT_KEYS.iter().map(|k| (*k).to_string()))
717 .collect();
718 keys.sort();
719 keys
720}
721
722/// What [`judge_join`] decided about one proposed cluster join.
723///
724/// Shaped like yubaba's `PromotionVerdict` / `GeographyVerdict` and for the
725/// same reason: the rule stays unit-testable without a live cluster, and a
726/// refusal carries its reason from the place that knows it.
727#[derive(Debug, Clone, PartialEq, Eq)]
728pub enum JoinVerdict {
729 /// Both nodes declare the same sovereign group and both are voters. The
730 /// join is within one blast radius and grows a quorum both sides are
731 /// eligible for.
732 Permit,
733 /// The join is refused. Carries an operator-readable reason naming both
734 /// declared values and the file to edit — a refusal that only says
735 /// "invalid" gets worked around rather than fixed.
736 Refuse(String),
737}
738
739/// May `joiner` join the cluster `target` belongs to? — W305/R742-F1.
740///
741/// **A join is permitted iff both nodes declare the same non-`None`
742/// [`sovereign_group`](MachineConfig::sovereign_group) and both are
743/// [`SovereignRole::Voter`].** One rule, no special cases, and it makes the
744/// declaration mandatory before any quorum grows.
745///
746/// The case this exists for is two *different* declared groups: joining a dev
747/// Pi into prod is refused rather than trusted, where today the only guard is
748/// a comment saying not to do it. But an undeclared node is refused too, and
749/// that is the deliberate half — `None` means "in no group", not "unknown", so
750/// growing prod with an unstamped box is exactly as much a cross-group join as
751/// the dev case is. Failing open there would leave the operator believing a
752/// guarantee that was never evaluated, which is the reasoning
753/// `QuorumGeography::judge` already applies to untagged voters.
754///
755/// No legitimate flow pays for that strictness: prod and dev are both stamped,
756/// and us-west-002/015 are deliberately in no group at all. Adding a real
757/// member means declaring it first, which is the point.
758///
759/// # The non-voting refusal (R605-F12)
760///
761/// Same group and still refused, when either side declares
762/// [`SovereignRole::NonVoter`]. This is the case a group label alone could not
763/// express. us-west-003 is a residential-uplink build box the operator counts
764/// as part of prod — same secrets, same upgrade cadence, same destruction — and
765/// which must never hold a prod raft seat, because a home-internet partition
766/// should not be able to stall the quorum. Until R605-F12 the only thing
767/// refusing it was its *absent* stamp, so recording the operator's real intent
768/// (`sovereign_group = "prod"`) would have removed the guard. Now the intent
769/// and the guard are the same two lines.
770///
771/// Note what this is not: the refusal here is about *voting*, and it says
772/// nothing about the mesh. One mesh spans the whole fleet regardless of group
773/// or role (operator, 2026-08-19); a non-voter is reachable, schedulable and
774/// rollable like any other node.
775///
776/// This is the **camp-side** rendering of the rule. The predicate itself lives
777/// in [`workload_spec::sovereign::join_permitted`] because yubaba's
778/// `POST /raft/add-learner` gate asks the same question and cannot see this
779/// crate (there is deliberately no yubaba → cloud edge). Only the prose is
780/// duplicated, and it has to be: a refusal here names
781/// `.yah/infra/machines/<name>.toml`, while the node-side one has no machine
782/// name in hand and must also name `yubaba serve --sovereign-group`.
783///
784/// The node-side gate is *narrower* on purpose, and the difference is worth
785/// knowing when reading either: a daemon started without `--sovereign-group`
786/// has declared nothing rather than declared standalone, so yubaba resolves
787/// that unknown before it judges, and its gate is in force only once the
788/// cluster being joined declares a group. See `yubaba::sovereign_group`.
789pub fn judge_join(joiner: &MachineConfig, target: &MachineConfig) -> JoinVerdict {
790 let stamp_hint = |m: &MachineConfig| {
791 format!(
792 "declare `sovereign_group = \"<group>\"` in .yah/infra/machines/{}.toml",
793 m.name
794 )
795 };
796 let role_hint = |m: &MachineConfig| {
797 format!(
798 "set `sovereign_role = \"voter\"` in .yah/infra/machines/{}.toml",
799 m.name
800 )
801 };
802 if workload_spec::sovereign::join_permitted(
803 joiner.sovereign_membership(),
804 target.sovereign_membership(),
805 ) {
806 return JoinVerdict::Permit;
807 }
808 let (j, t) = (
809 joiner.sovereign_group.as_deref(),
810 target.sovereign_group.as_deref(),
811 );
812 // Everything below is a refusal; the only permitted shape returned above.
813 //
814 // R605-F12: when both sides name the SAME group, the role is the only thing
815 // left that can have refused, and it gets its own message. Falling through
816 // to the arms below would print "cross-group join refused: 'us-west-003' is
817 // in "prod" and 'us-west-001' is in "prod"" — a message that reads as a bug
818 // in the check rather than a decision about the fleet.
819 //
820 // Deliberately not hoisted above the group comparison. A non-voting joiner
821 // whose target is standalone is refused for *both* reasons, and naming the
822 // role there would send the operator to fix a field that would not have
823 // made the join legal anyway.
824 if let (Some(a), Some(b)) = (j, t) {
825 if a == b {
826 for (m, side, other) in [
827 (joiner, "the joiner", &target.name),
828 (target, "the target", &joiner.name),
829 ] {
830 if m.sovereign_membership().role.is_voter() {
831 continue;
832 }
833 return JoinVerdict::Refuse(format!(
834 "join refused: {side} '{}' is a NON-VOTING member of sovereign group {a:?}, \
835 the same group as '{other}'. It is inside that blast radius — same secrets, \
836 same upgrade cadence, same destruction — but declares itself ineligible for \
837 the quorum, so this is refused by declaration rather than by omission. If it \
838 should genuinely vote, {}; if it should not, this refusal is the field doing \
839 its job and the join is the thing to reconsider.",
840 m.name,
841 role_hint(m),
842 ));
843 }
844 }
845 }
846 match (j, t) {
847 (Some(a), Some(b)) => JoinVerdict::Refuse(format!(
848 "cross-group join refused: '{}' is in sovereign group {a:?} and '{}' is in {b:?}. \
849 These are separate blast radii — separate quorums, separate upgrade cadences, \
850 separately destroyable — and merging them is not something a join can undo. If \
851 the move is genuinely intended, restamp '{}' to {b:?} first and treat it as \
852 leaving its old group.",
853 joiner.name,
854 target.name,
855 joiner.name,
856 )),
857 (None, Some(b)) => JoinVerdict::Refuse(format!(
858 "join refused: '{}' declares no sovereign_group, so it is standalone — in no \
859 group — while '{}' is in {b:?}. That is a cross-group join, not an unchecked \
860 one. To make '{}' a member of {b:?}, {}.",
861 joiner.name,
862 target.name,
863 joiner.name,
864 stamp_hint(joiner),
865 )),
866 (Some(a), None) => JoinVerdict::Refuse(format!(
867 "join refused: '{}' is in sovereign group {a:?} but '{}' declares none, so the \
868 target is standalone and has no group to join. Either {}, or found the group on \
869 '{}' rather than growing it.",
870 joiner.name,
871 target.name,
872 stamp_hint(target),
873 joiner.name,
874 )),
875 (None, None) => JoinVerdict::Refuse(format!(
876 "join refused: neither '{}' nor '{}' declares a sovereign_group, so this join \
877 would form a group nobody declared and nothing could later reason about. Name \
878 the group on both boxes first: {}, and the same for '{}'.",
879 joiner.name,
880 target.name,
881 stamp_hint(joiner),
882 target.name,
883 )),
884 }
885}
886
887impl MachineConfig {
888 /// This node's declared place in a sovereign group, as the shared join rule
889 /// wants it — R605-F12.
890 ///
891 /// The one place `sovereign_role`'s `None` is resolved. Absence means
892 /// [`SovereignRole::Voter`], which is what declaring a group meant before
893 /// the role existed; resolving it here rather than at each call site is what
894 /// keeps the camp-side and node-side gates from disagreeing about a node
895 /// that never wrote the field.
896 pub fn sovereign_membership(&self) -> Membership<'_> {
897 Membership {
898 group: self.sovereign_group.as_deref(),
899 role: self.sovereign_role.unwrap_or_default(),
900 }
901 }
902
903 /// Provider DC code, or `""` when omitted (static nodes). Most readers want
904 /// a `&str`; the driver-backed provision/status paths still go through
905 /// [`validate`](Self::validate) which guarantees presence for those.
906 pub fn location(&self) -> &str {
907 self.location.as_deref().unwrap_or("")
908 }
909
910 /// Provider SKU, or `""` when omitted (static nodes).
911 pub fn server_type(&self) -> &str {
912 self.server_type.as_deref().unwrap_or("")
913 }
914
915 /// Enforce the provisioning-only-field contract: a machine whose provider
916 /// has an auto-provision driver MUST declare `location` + `server_type`
917 /// (the driver can't create a server without them). Static nodes may omit
918 /// both. Call this before any provision/diff that assumes a driver.
919 pub fn validate(&self) -> Result<()> {
920 if provider_has_machine_driver(&self.provider) {
921 if self.location.is_none() {
922 anyhow::bail!(
923 "machine '{}' (provider '{}') has an auto-provision driver but no `location`",
924 self.name,
925 self.provider
926 );
927 }
928 if self.server_type.is_none() {
929 anyhow::bail!(
930 "machine '{}' (provider '{}') has an auto-provision driver but no `server_type`",
931 self.name,
932 self.provider
933 );
934 }
935 }
936 Ok(())
937 }
938
939 /// Declared taints that no placement decision can read (W305/R742-T4).
940 ///
941 /// Deliberately **not** folded into [`validate`](Self::validate): that
942 /// guard runs on the provision/diff hot path and answers a different
943 /// question (can the driver create this server). An inert taint is a lint
944 /// — it never breaks an operation in flight, it just means the file is
945 /// asserting something the scheduler will not honour. `yah cloud validate`
946 /// is where the operator asks for that judgement; see
947 /// [`crate::validate::check_inert_taints`].
948 pub fn inert_taints(&self) -> Vec<&str> {
949 self.taints
950 .iter()
951 .filter(|t| taint_effect(t) == TaintEffect::Inert)
952 .map(String::as_str)
953 .collect()
954 }
955
956 /// Yubaba's TOFU'd hostkey fingerprint, from `[registration]` and falling
957 /// back to the pre-R707-T1 top-level field. **The only read path** — a
958 /// caller that reaches for `legacy_hostkey_fingerprint` directly sees
959 /// `None` on every migrated machine.
960 pub fn hostkey_fingerprint(&self) -> Option<&str> {
961 self.registration
962 .hostkey_fingerprint
963 .as_deref()
964 .or(self.legacy_hostkey_fingerprint.as_deref())
965 }
966
967 /// Record (or clear) the observed hostkey fingerprint. Writes
968 /// `[registration]` and drops any pre-R707-T1 top-level value, so the two
969 /// locations can never disagree after a writeback.
970 pub fn set_hostkey_fingerprint(&mut self, fingerprint: Option<String>) {
971 self.registration.hostkey_fingerprint = fingerprint;
972 self.legacy_hostkey_fingerprint = None;
973 }
974
975 /// Mesh (tailnet) IPv4 for this node, or `None` pre-mesh.
976 ///
977 /// Prefers `[registration].mesh_ipv4`; falls back to the host of a legacy
978 /// `[connect].yubaba` URL when that host is in the `100.64.0.0/10` CGNAT
979 /// range the mesh uses. A loopback placeholder (`http://127.0.0.1:7443`,
980 /// meaning "pre-mesh, reachable only through an SSH tunnel") is *not* a
981 /// mesh address and yields `None`.
982 pub fn mesh_ipv4(&self) -> Option<&str> {
983 if let Some(ip) = self.registration.mesh_ipv4.as_deref() {
984 return Some(ip);
985 }
986 let url = self.connect.as_ref()?.yubaba.as_deref()?;
987 mesh_ipv4_from_url(url)
988 }
989
990 /// Base URL for this node's yubaba, or `None` when no reach resolves.
991 ///
992 /// Thin wrapper over [`reach`](Self::reach) for the many call sites that
993 /// only branch on presence. Prefer `reach` anywhere the operator sees the
994 /// outcome — a `None` here throws away a refusal that names exactly which
995 /// address is missing.
996 pub fn yubaba_url(&self) -> Option<String> {
997 self.reach().ok()
998 }
999
1000 /// The **one** address automation dials for this node — mesh-only.
1001 ///
1002 /// `Err` is a *named refusal*, not an absence: a node with no mesh address
1003 /// is unresolvable to every automated path, and R605-T10's whole complaint
1004 /// is that this used to surface as a connect timeout against an address the
1005 /// caller has no route to.
1006 ///
1007 /// Resolution order:
1008 ///
1009 /// 1. A declared `[connect].yubaba` on a **private** host (10/8,
1010 /// 172.16/12, 192.168/16) is **not dialed** — see below.
1011 /// 2. Any other declared `[connect].yubaba` wins verbatim. That includes
1012 /// the pre-mesh loopback placeholder (`http://127.0.0.1:7443`, "I have
1013 /// no mesh address; reach me through the SSH tunnel to `ssh`"), which is
1014 /// a genuine declaration and stays honoured.
1015 /// 3. Otherwise `[registration].mesh_ipv4` composed with
1016 /// `[connect].yubaba_port`.
1017 ///
1018 /// **Why a LAN literal loses (R605-T10, operator 2026-08-19).** The LAN
1019 /// address is an emergency break-glass route, never an official one, and
1020 /// automation must ALWAYS assume the caller is not on that LAN — this camp
1021 /// sits on 192.168.22.0/22 with no route to the fleet's 192.168.10.0/24 at
1022 /// all. Writing one into the field every resolver dials does not sit beside
1023 /// the mesh route, it *overrides* it: R707-T6 made a declared literal beat
1024 /// `mesh_ipv4` outright, so us-west-011 (mesh-joined, healthy) was elected
1025 /// for every aarch64 build and then dialed at an address that answers only
1026 /// from inside bldg-2506.
1027 ///
1028 /// **What R707-T6 wanted is preserved elsewhere.** Its forcing case was
1029 /// identity, not reach: the dev raft group advertises LAN addrs
1030 /// (`192.168.10.11:7443`, verified live off `/raft/status` 2026-08-27), and
1031 /// `rollout::yubaba::membership_to_nodes` has to map those back to declared
1032 /// machines. That match now runs against [`lan_endpoint`](Self::lan_endpoint),
1033 /// which is composed from the break-glass `[connect].address` metadata and
1034 /// is never dialed — so the two concerns the old precedence rule fused are
1035 /// split, and the literal can stop squatting a dialed field.
1036 ///
1037 /// The LAN address itself STAYS in the machine TOML. It is useful metadata
1038 /// and the manual `ssh` path is entitled to it; it is only disconnected
1039 /// from every automated process.
1040 pub fn reach(&self) -> Result<String, String> {
1041 let Some(connect) = self.connect.as_ref() else {
1042 return Err(format!(
1043 "machine {:?} declares no [connect] block, so nothing knows how to reach it \
1044 \u{2192} declare one, or leave it unprovisioned and out of placement",
1045 self.name
1046 ));
1047 };
1048 let mesh = || {
1049 self.registration
1050 .mesh_ipv4
1051 .as_deref()
1052 .map(|ip| format!("http://{ip}:{}", connect.yubaba_port()))
1053 };
1054 if let Some(literal) = &connect.yubaba {
1055 let Some(lan) = private_ipv4_from_url(literal) else {
1056 return Ok(literal.clone());
1057 };
1058 return mesh().ok_or_else(|| {
1059 format!(
1060 "machine {:?} is unresolvable to automation: its only declared yubaba reach \
1061 is the private literal {:?} and it has no [registration].mesh_ipv4\n\
1062 \u{2192} a LAN address is an emergency break-glass route, never an official \
1063 one (R605-T10) — every automated path assumes the caller is NOT on {}/24\n\
1064 \u{2192} mesh-join the box and record `mesh_ipv4` under [registration], then \
1065 delete `[connect].yubaba` so the port composes with it",
1066 self.name,
1067 literal,
1068 lan.rsplit_once('.').map(|(net, _)| net).unwrap_or(lan),
1069 )
1070 });
1071 }
1072 mesh().ok_or_else(|| {
1073 format!(
1074 "machine {:?} has no [registration].mesh_ipv4 and declares no \
1075 [connect].yubaba, so no automated path can reach it\n\
1076 \u{2192} mesh-join the box and record its tailnet address, or taint it out of \
1077 placement — do not point `[connect].yubaba` at a LAN address (R605-T10)",
1078 self.name
1079 )
1080 })
1081 }
1082
1083 /// The LAN `host:port` this node's yubaba answers on, composed from the
1084 /// break-glass `[connect].address` metadata plus the declared port.
1085 ///
1086 /// **Identity only — never dial this.** It exists so a raft membership
1087 /// entry that names a node by its LAN address can be mapped back to the
1088 /// declared machine (`rollout::yubaba::membership_to_nodes`) without that
1089 /// address having to live in a field a resolver reads. `None` when the
1090 /// machine is unprovisioned.
1091 pub fn lan_endpoint(&self) -> Option<String> {
1092 let connect = self.connect.as_ref()?;
1093 Some(format!("{}:{}", connect.address, connect.yubaba_port()))
1094 }
1095
1096 /// Fold the pre-R707-T1 top-level `hostkey_fingerprint` into
1097 /// `[registration]`, and lift a mesh IP out of a legacy `[connect].yubaba`
1098 /// URL. Idempotent; a machine already on the split shape is untouched.
1099 ///
1100 /// [`save`](Self::save) calls this, so writing a machine TOML migrates it
1101 /// rather than round-tripping the old shape back out.
1102 pub fn normalize(&mut self) {
1103 if let Some(fp) = self.legacy_hostkey_fingerprint.take() {
1104 self.registration.hostkey_fingerprint.get_or_insert(fp);
1105 }
1106 if self.registration.mesh_ipv4.is_none() {
1107 if let Some(ip) = self
1108 .connect
1109 .as_ref()
1110 .and_then(|c| c.yubaba.as_deref())
1111 .and_then(mesh_ipv4_from_url)
1112 .map(str::to_string)
1113 {
1114 self.registration.mesh_ipv4 = Some(ip);
1115 // The URL was pure derivation from mesh IP + port; keep only
1116 // the declared half so the two can't drift apart.
1117 if let Some(c) = self.connect.as_mut() {
1118 c.yubaba = None;
1119 }
1120 }
1121 }
1122 }
1123
1124 /// Persist to `<cloud_dir>/machines/<name>.toml`, creating the dir if needed.
1125 ///
1126 /// ⚠ Serializes the struct, so **operator comments in the target file are
1127 /// lost**. Pre-existing behaviour, not introduced here, but it is why
1128 /// registration writeback (`yah cloud machine attach`) goes through
1129 /// [`crate::state::MachineState`] and the comment-preserving path in the
1130 /// CLI rather than calling this on a hand-authored inventory file.
1131 pub fn save(&self, cloud_dir: &Path) -> Result<()> {
1132 let dir = cloud_dir.join("machines");
1133 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
1134 let path = dir.join(format!("{}.toml", self.name));
1135 let mut normalized = self.clone();
1136 normalized.normalize();
1137 let s = toml::to_string_pretty(&normalized)
1138 .with_context(|| format!("serializing machine {}", self.name))?;
1139 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
1140 }
1141}
1142
1143/// Host of an `http://host:port` URL iff it is a mesh (headscale) IPv4 in the
1144/// `100.64.0.0/10` CGNAT range. String-level rather than URL-parsed: the
1145/// inventory format is stable and this crate carries no URL dependency (same
1146/// reasoning as `fleet_metrics::extract_host` and
1147/// `hub::coordinator::is_loopback_url`).
1148fn mesh_ipv4_from_url(url: &str) -> Option<&str> {
1149 let host = ipv4_host_of(url)?;
1150 let ip: std::net::Ipv4Addr = host.parse().ok()?;
1151 let [a, b, ..] = ip.octets();
1152 // 100.64.0.0/10 ⇒ first octet 100, second octet 64..=127.
1153 (a == 100 && (64..=127).contains(&b)).then_some(host)
1154}
1155
1156/// Host of an `http://host:port` URL iff it is an **RFC1918 private** IPv4 —
1157/// `10/8`, `172.16/12`, `192.168/16`. `None` for anything else, loopback and
1158/// the `100.64/10` mesh range included: neither is a LAN literal.
1159///
1160/// The judgement R605-T10 turns on. A private literal is only ever reachable
1161/// from inside one building, so it is metadata about where the box physically
1162/// sits and never an address automation may dial — see
1163/// [`MachineConfig::reach`] and [`crate::validate::check_lan_dial_targets`].
1164pub fn private_ipv4_from_url(url: &str) -> Option<&str> {
1165 let host = ipv4_host_of(url)?;
1166 is_private_ipv4(host).then_some(host)
1167}
1168
1169/// Whether a bare host string is an RFC1918 private IPv4 literal.
1170pub fn is_private_ipv4(host: &str) -> bool {
1171 let Ok(ip) = host.parse::<std::net::Ipv4Addr>() else {
1172 return false;
1173 };
1174 ip.is_private()
1175}
1176
1177/// Bare host of a `[scheme://]host[:port][/path]` string.
1178fn ipv4_host_of(url: &str) -> Option<&str> {
1179 let after_scheme = url.split("://").nth(1).unwrap_or(url);
1180 after_scheme.split(['/', ':']).next()
1181}
1182
1183/// Declared **reach** for a BYO `static` node (no provider API). Lives under
1184/// `[connect]` in the machine TOML.
1185///
1186/// Reach only — how the camp gets to the box. *Permission* is a separate axis
1187/// that belongs to cheers' scopes (W295 §"Deliberately deferred"); the two
1188/// collapse in practice today (mesh membership grants everything) and the data
1189/// model must not fuse them, so do not add an authorization field here.
1190///
1191/// `address`, `ssh` and `identity_file` stay whole, literal, operator-authored
1192/// strings even though their values often *look* derived. They are not:
1193/// us-west-001 dials SSH over its public IP while us-west-002 was deliberately
1194/// repointed at its tailnet IP (R608-F10) precisely because the LAN address is
1195/// unreachable off-LAN. Decomposing them into user + host and recomposing
1196/// would silently undo per-machine decisions like that one. `yubaba` is the
1197/// field that *was* derived — mesh IP plus a fixed port, rewritten by
1198/// mesh-join — so that is where R707-T1 cut.
1199#[derive(Debug, Clone, Serialize, Deserialize)]
1200#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
1201pub struct ConnectSpec {
1202 /// Reachable IPv4/host for the box, e.g. `"45.32.194.254"`. Declared: which
1203 /// of a machine's several addresses the camp should use is an operator
1204 /// choice (public IP vs. LAN IP vs. tailnet IP).
1205 pub address: String,
1206 /// SSH target the camp dials for bootstrap + (pre-mesh) tunneled deploys,
1207 /// e.g. `"root@45.32.194.254"` or `"struc@100.64.0.4"`. Declared, whole —
1208 /// see the type doc. Pair with `identity_file` for a copy-pasteable
1209 /// `ssh -i <identity_file> <ssh>`.
1210 pub ssh: String,
1211 /// Private key path the camp uses to authenticate `ssh`, e.g.
1212 /// `"~/.ssh/yah"`. Every node in the fleet uses the same operator key
1213 /// today, but this is declared per-machine rather than assumed globally
1214 /// for the same reason `ssh` is whole rather than decomposed: a future
1215 /// node with a different key should not have to fight a hardcoded
1216 /// default. `~` is not shell-expanded by this crate — callers that shell
1217 /// out to `ssh`/`scp` pass it through `-i`, which expands it itself.
1218 pub identity_file: String,
1219 /// Port yubaba listens on. Declared reach; defaults to 7443 when omitted,
1220 /// which is every machine in the fleet today. Composed with the *observed*
1221 /// [`MachineRegistration::mesh_ipv4`] by [`MachineConfig::yubaba_url`].
1222 #[serde(default, skip_serializing_if = "Option::is_none")]
1223 pub yubaba_port: Option<u16>,
1224 /// Explicit yubaba base URL, overriding the composed form.
1225 ///
1226 /// Two live uses, both genuine declarations: a pre-mesh node saying
1227 /// `"http://127.0.0.1:7443"` — "I have no mesh address; reach me through
1228 /// the SSH tunnel to `ssh`" — and any node whose yubaba is not at
1229 /// `mesh_ipv4:port`. A URL here whose host *is* a mesh IP is the
1230 /// pre-R707-T1 shape; [`MachineConfig::normalize`] lifts it into
1231 /// `[registration].mesh_ipv4` and clears this field so the two cannot
1232 /// drift apart.
1233 #[serde(default, skip_serializing_if = "Option::is_none")]
1234 pub yubaba: Option<String>,
1235}
1236
1237/// Default yubaba listen port, used when `[connect].yubaba_port` is omitted.
1238pub const DEFAULT_YUBABA_PORT: u16 = 7443;
1239
1240impl ConnectSpec {
1241 /// Declared yubaba port, defaulting to [`DEFAULT_YUBABA_PORT`].
1242 pub fn yubaba_port(&self) -> u16 {
1243 self.yubaba_port.unwrap_or(DEFAULT_YUBABA_PORT)
1244 }
1245}
1246
1247#[derive(Debug, Clone, Serialize, Deserialize)]
1248#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
1249pub struct BucketSpec {
1250 pub name: String,
1251 pub public_read: bool,
1252}
1253
1254/// Per-camp mirror declaration from `.yah/cloud/mirrors/<id>/mirror.toml`
1255/// (folder form) or the legacy `.yah/cloud/mirrors/<id>.toml` (flat form).
1256///
1257/// The folder form is preferred for new mirrors so that per-mirror secrets
1258/// and override files can sit next to `mirror.toml` without polluting the
1259/// top-level `mirrors/` directory.
1260#[derive(Debug, Clone, Serialize, Deserialize)]
1261pub struct LegacyMirrorConfig {
1262 /// Logical camp name this mirror hosts, e.g. `"yah"` or `"noisetable"`.
1263 ///
1264 /// Serialised as `camp`; accepts the legacy `rig` spelling for files that
1265 /// predate the R137 rig→camp rename (one-time migration: `sed -i ''
1266 /// 's/^rig = /camp = /' ~/.yah/cloud/mirrors/*.toml`).
1267 #[serde(rename = "camp", alias = "rig")]
1268 pub camp: String,
1269 pub regions: Vec<String>,
1270 /// Workload names deployed as part of this mirror (references `workloads/<name>.toml`).
1271 /// Renamed from `services` in R092-F1; use `yah cloud config migrate-services-to-workloads`
1272 /// on repos that still have the old `services/` layout.
1273 #[serde(alias = "services")]
1274 pub workloads: Vec<String>,
1275 /// Base domain for Cloudflare-fronted services on this mirror's machines.
1276 /// Combined with the machine's `location` to build virtual-host names:
1277 /// e.g. `cloud_domain = "cloud.noisetable.example"` on machine in location
1278 /// `pdx` → Caddyfile site address `pdx.cloud.noisetable.example`.
1279 /// Optional: if unset the Caddyfile falls back to `:port` listeners.
1280 #[serde(default, skip_serializing_if = "Option::is_none")]
1281 pub cloud_domain: Option<String>,
1282}
1283
1284/// Error from loading or validating a single workload TOML file.
1285#[derive(Debug, Error)]
1286pub enum WorkloadConfigError {
1287 #[error("reading {path}: {source}")]
1288 Io {
1289 path: String,
1290 source: std::io::Error,
1291 },
1292 #[error("parsing {path}: {source}")]
1293 Toml {
1294 path: String,
1295 source: toml::de::Error,
1296 },
1297 #[error("invalid WorkloadSpec in {path}: {source}")]
1298 Shape {
1299 path: String,
1300 source: validate::ShapeError,
1301 },
1302}
1303
1304/// A workload declaration loaded from `.yah/cloud/workloads/<name>.toml`.
1305///
1306/// Each file is the human-authored TOML serialization of a [`WorkloadSpec`].
1307/// On load, the spec is validated against the shape layer; failures surface as
1308/// a [`CloudConfigError::Workload`] with the file path and field path.
1309#[derive(Debug, Clone, Serialize, Deserialize)]
1310pub struct WorkloadConfig {
1311 /// The validated spec.
1312 #[serde(flatten)]
1313 pub spec: WorkloadSpec,
1314}
1315
1316impl WorkloadConfig {
1317 /// Persist to `<cloud_dir>/workloads/<name>.toml`, creating the dir if needed.
1318 pub fn save(&self, cloud_dir: &Path) -> Result<()> {
1319 let dir = cloud_dir.join("workloads");
1320 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
1321 let path = dir.join(format!("{}.toml", self.spec.name));
1322 let s = toml::to_string_pretty(self)
1323 .with_context(|| format!("serializing workload {}", self.spec.name))?;
1324 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
1325 }
1326}
1327
1328/// Error surfaced by [`CloudConfig::load`] when a workload TOML fails validation.
1329#[derive(Debug, Error)]
1330pub enum CloudConfigError {
1331 #[error(transparent)]
1332 Anyhow(#[from] anyhow::Error),
1333 #[error("workload validation failed: {0}")]
1334 Workload(WorkloadConfigError),
1335}
1336
1337/// Mirror-to-machine assignment table from `.yah/cloud/topology.toml`.
1338///
1339/// Declares which logical mirror names are assigned to which machines.
1340/// This is the source-canonical placement until yubaba raft observes it
1341/// (per the migration tracker in the arch doc).
1342#[derive(Debug, Clone, Serialize, Deserialize, Default)]
1343pub struct TopologyConfig {
1344 /// Mirror→machine assignments.
1345 #[serde(default)]
1346 pub assignments: Vec<MirrorAssignment>,
1347 /// Declared buckets, logged by `yah cloud bucket create`.
1348 /// Source-canonical until yubaba raft observes actual placement.
1349 #[serde(default, skip_serializing_if = "Vec::is_empty")]
1350 pub buckets: Vec<BucketLogEntry>,
1351}
1352
1353impl TopologyConfig {
1354 /// Load from a `topology.toml` file, returning `Default` when absent.
1355 pub fn load(path: &Path) -> Result<Self> {
1356 if !path.exists() {
1357 return Ok(Self::default());
1358 }
1359 let s =
1360 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
1361 toml::from_str(&s).with_context(|| format!("parsing {}", path.display()))
1362 }
1363
1364 /// Persist to `topology.toml`, creating parent dirs if needed.
1365 pub fn save(&self, path: &Path) -> Result<()> {
1366 if let Some(parent) = path.parent() {
1367 std::fs::create_dir_all(parent)
1368 .with_context(|| format!("creating {}", parent.display()))?;
1369 }
1370 let s = toml::to_string_pretty(self).context("serializing topology")?;
1371 std::fs::write(path, s).with_context(|| format!("writing {}", path.display()))
1372 }
1373
1374 /// Find a declared bucket by name.
1375 pub fn bucket_by_name(&self, name: &str) -> Option<&BucketLogEntry> {
1376 self.buckets.iter().find(|b| b.name == name)
1377 }
1378
1379 /// Find a mutable declared bucket by name.
1380 pub fn bucket_by_name_mut(&mut self, name: &str) -> Option<&mut BucketLogEntry> {
1381 self.buckets.iter_mut().find(|b| b.name == name)
1382 }
1383
1384 /// Returns true if the bucket is declared as cross-machine (no owning machine).
1385 pub fn is_cross_machine_bucket(&self, name: &str) -> bool {
1386 self.buckets
1387 .iter()
1388 .any(|b| b.name == name && b.machine.is_none())
1389 }
1390}
1391
1392/// One mirror→machine placement entry in `topology.toml`.
1393#[derive(Debug, Clone, Serialize, Deserialize)]
1394pub struct MirrorAssignment {
1395 /// Logical mirror name, e.g. `"noisetable-pdx"`.
1396 pub mirror: String,
1397 /// Machine that hosts this mirror, e.g. `"noisetable-pdx-1"`.
1398 pub machine: String,
1399}
1400
1401/// A bucket declaration logged in `topology.toml` by `yah cloud bucket create`.
1402#[derive(Debug, Clone, Serialize, Deserialize)]
1403pub struct BucketLogEntry {
1404 pub name: String,
1405 /// Machine that owns this bucket. `None` marks it as cross-machine
1406 /// (no single-machine ownership; requires an explicit declaration in
1407 /// `topology.toml` before `yah cloud bucket create` will proceed without
1408 /// `--machine`).
1409 #[serde(default, skip_serializing_if = "Option::is_none")]
1410 pub machine: Option<String>,
1411 /// Logical location of the bucket, e.g. `"pdx"`.
1412 pub location: String,
1413 /// Current declared policy: `"private"` | `"public-read"` | `"signed-only"`.
1414 #[serde(default = "default_bucket_policy")]
1415 pub policy: String,
1416}
1417
1418fn default_bucket_policy() -> String {
1419 "private".to_string()
1420}
1421
1422#[derive(Debug, Clone, Serialize, Deserialize)]
1423pub struct PortMapping {
1424 pub host: u16,
1425 pub container: u16,
1426}
1427
1428/// A loaded service plus its per-environment mirrors.
1429///
1430/// Wraps the `service.toml` body and the directory of `mirrors/<env>.toml`
1431/// files that project the service onto concrete infra.
1432#[derive(Debug, Clone, Serialize, Deserialize)]
1433pub struct ServiceWithMirrors {
1434 pub service: ServiceConfig,
1435 /// Mirrors keyed by environment name (file stem of `mirrors/<env>.toml`).
1436 pub mirrors: BTreeMap<String, MirrorConfig>,
1437 /// Transform recipe names keyed by component id. Populated from each
1438 /// static-asset component's `workload.toml` at load time — not stored
1439 /// in service.toml. Only present for components that declare
1440 /// `[asset.derive.transform] recipe = "..."`.
1441 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
1442 pub component_transform_recipes: BTreeMap<String, String>,
1443 /// Nodes each mirror's passway front door is placed on, keyed by env —
1444 /// exactly what [`MirrorConfig::passway_machines`] returns, with the envs
1445 /// that declare no passway edge left out.
1446 ///
1447 /// Derived at load time like `component_transform_recipes` above: it is
1448 /// stored in no TOML file. It exists so that a consumer of this wire type —
1449 /// the desktop `service_list` command, and through it the Services tab's
1450 /// custom-domain panel — never reconciles the two `ingress` spellings
1451 /// itself. An env present here with an **empty** list is a passway edge
1452 /// whose placement is co-located rather than declared; see
1453 /// [`MirrorConfig::passway_machines`] for why that is a different answer
1454 /// from being absent.
1455 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
1456 pub passway_machines: BTreeMap<String, Vec<String>>,
1457}
1458
1459/// All cloud config loaded from a workspace root (the parent of `.yah/`).
1460///
1461/// Reads two trees:
1462/// - `.yah/infra/` — `machines/`, `providers/`
1463/// - `.yah/services/<svc>/` — `service.toml` + `mirrors/<env>.toml`
1464///
1465/// Pre-R215 fields (`legacy_mirrors`, `workloads`, `topology`) are still
1466/// populated from `.yah/cloud/` when present so pre-R215 callers (bucket
1467/// commands) keep compiling — they just see empty collections in a post-B1
1468/// workspace where the legacy data was deleted. These fields are scheduled
1469/// for removal in B3-T3. (`legacy_services` was the last of this group with
1470/// a live renderer — the compose/Caddy generator over it — and was removed
1471/// along with that renderer in R895-T2.)
1472#[derive(Debug)]
1473pub struct CloudConfig {
1474 /// Workspace root that was loaded — useful for path-resolving
1475 /// component references on a [`ServiceComponent`].
1476 pub workspace_root: std::path::PathBuf,
1477
1478 // ─── R215+ tree ────────────────────────────────────────────────────────
1479 /// `.yah/infra/machines/<name>.toml`
1480 pub machines: Vec<MachineConfig>,
1481 /// `.yah/infra/providers/<id>.toml`
1482 pub providers: Vec<ProviderConfig>,
1483 /// Provenance for every entry in `machines` that came from a linked
1484 /// `.yah/infra/sources.toml` source rather than this camp's own
1485 /// `.yah/infra/machines/` (R615-F2 / W274). Keyed by
1486 /// [`MachineConfig::name`]; a name absent here is camp-local. Empty from
1487 /// [`CloudConfig::load_from_config_dir`] — see its doc for why sources
1488 /// don't apply to multi-root sibling trees.
1489 pub machine_origins: BTreeMap<String, InfraOrigin>,
1490 /// Same as [`machine_origins`](Self::machine_origins), keyed by
1491 /// [`ProviderConfig::id`].
1492 pub provider_origins: BTreeMap<String, InfraOrigin>,
1493 /// `.yah/services/<svc>/` — service.toml plus mirrors/<env>.toml.
1494 pub services: BTreeMap<String, ServiceWithMirrors>,
1495 /// Measured restore times replayed from `.yah/cloud/recovery.jsonl`
1496 /// (R850-T2), keyed by workload name and summed across that workload's
1497 /// subjects. Empty in every camp that has never timed a restore — a missing
1498 /// journal is not an error, exactly like an unsynced infra source.
1499 ///
1500 /// This field is what lets [`crate::topology::analyze`] report a measured
1501 /// recovery instead of an extrapolation **while staying pure**: the
1502 /// measurement is read off the local tree here, at load, alongside every
1503 /// other declaration, so the analyzer gains no I/O and no `Path` argument.
1504 /// See [`crate::recovery_journal`].
1505 pub recovery_measurements: BTreeMap<String, crate::recovery_journal::WorkloadRecovery>,
1506 /// `.yah/domains/<name>.toml` — public-facing routing manifests
1507 /// (R347). Single file per domain; no nested per-env tree because
1508 /// domains themselves aren't projected onto infra — they describe
1509 /// how a Worker bundle ingresses requests onto services.
1510 pub domains: BTreeMap<String, DomainConfig>,
1511
1512 // ─── Pre-R215 legacy (slated for removal in B3-T3) ────────────────────
1513 /// Legacy mirrors from `.yah/cloud/mirrors/`.
1514 pub legacy_mirrors: Vec<LegacyMirrorConfig>,
1515 /// Workloads from `.yah/cloud/workloads/*.toml` (R092-F1 schema).
1516 pub workloads: Vec<WorkloadConfig>,
1517 /// Topology from `.yah/cloud/topology.toml` (mirror→machine assignments).
1518 pub topology: TopologyConfig,
1519}
1520
1521impl CloudConfig {
1522 /// Load all cloud config rooted at `workspace_root` (the parent of `.yah/`).
1523 ///
1524 /// Reads the R215+ tree (`.yah/infra/`, `.yah/services/<svc>/`) eagerly
1525 /// and the pre-R215 `.yah/cloud/` tree opportunistically. Returns `Err`
1526 /// immediately if any TOML fails to parse or a workload TOML fails
1527 /// shape validation; the error includes the file path and field path.
1528 ///
1529 /// Cross-ref validation runs after both trees finish loading: every
1530 /// `mirror.providers.X.use = "<id>"` must resolve to a real provider
1531 /// declared under `.yah/infra/providers/`.
1532 ///
1533 /// R844-B7 — **a missing `.yah/` is a wrong-root error, not an empty
1534 /// fleet.** Every sub-loader below tolerates a missing directory by
1535 /// returning empty, so before this check a call against the wrong
1536 /// directory produced a perfectly valid `CloudConfig` with zero machines,
1537 /// zero services and zero providers. Nothing downstream can tell that
1538 /// apart from a camp that genuinely declares nothing, so the failure
1539 /// surfaces as an operation that silently does nothing to nothing: a
1540 /// collate that renders no backends, a fanout that asks no nodes, a
1541 /// rollout that plans against an empty fleet. It was found the hard way —
1542 /// a live-fleet test in `app/yah/cli` called this with `"."`, which under
1543 /// `cargo test` is the *package* root, and passed while measuring nothing.
1544 ///
1545 /// The line is drawn at `.yah/` and only there: a workspace whose
1546 /// `.yah/infra/machines/` is absent or empty is a real, if unusual, camp
1547 /// with an empty fleet and still loads. `unknown` is not `answered with
1548 /// none`.
1549 pub fn load(workspace_root: &Path) -> Result<Self> {
1550 let yah_dir = crate::paths::yah_dir(workspace_root);
1551 if !yah_dir.is_dir() {
1552 anyhow::bail!(
1553 "not a yah workspace: no {} — expected the camp root (the parent \
1554 of `.yah/`), got {}. This is a wrong-root error, not an empty \
1555 fleet; a camp with no machines declared still has a `.yah/`.",
1556 yah_dir.display(),
1557 workspace_root.display(),
1558 );
1559 }
1560
1561 let mut providers = load_providers(&crate::paths::providers_dir(workspace_root))?;
1562 let services = load_services(&crate::paths::services_dir(workspace_root), workspace_root)?;
1563 let domains = load_domains(&crate::paths::domains_dir(workspace_root))?;
1564
1565 Self::cross_ref_validate(&providers, &services, &domains)?;
1566
1567 // Legacy `.yah/cloud/` reads — empty in post-B1 workspaces. Wrapped in
1568 // a helper so a missing tree is silent (no error, no warning).
1569 let cloud_dir = crate::paths::legacy_cloud_dir(workspace_root);
1570 let (legacy_mirrors, legacy_workloads, topology) = if cloud_dir.exists() {
1571 (
1572 load_mirrors(cloud_dir.join("mirrors"))?,
1573 load_workloads(cloud_dir.join("workloads"))?,
1574 load_topology(cloud_dir.join("topology.toml"))?,
1575 )
1576 } else {
1577 Default::default()
1578 };
1579
1580 // Workloads come from `.yah/infra/workloads/` (R215+). R568-T7: before
1581 // that path was read here, this field was populated *only* from the
1582 // legacy tree above — which R222-B1 emptied — so `cfg.workload(name)`
1583 // resolved nothing in every post-R215 camp and `yah cloud workload
1584 // deploy` could not find any declaration at all. The bug survived
1585 // because the only workloads ever deployed were forge/QED runs, which
1586 // build their spec in memory and never come through here. Same
1587 // dedupe-by-name shape as machines below: R215+ wins.
1588 let mut workloads = load_workloads(crate::paths::workloads_dir(workspace_root))?;
1589 let workload_names: std::collections::HashSet<String> =
1590 workloads.iter().map(|w| w.spec.name.clone()).collect();
1591 for w in legacy_workloads {
1592 if !workload_names.contains(&w.spec.name) {
1593 workloads.push(w);
1594 }
1595 }
1596
1597 // R870-B13: machines are resolved by [`resolve_fleet_inventory`] —
1598 // camp-local, the pre-R215 legacy tree, and every machine borrowed
1599 // through `.yah/infra/sources.toml`, in that precedence. This used to
1600 // be spelled out inline here, which made `CloudConfig::load` the only
1601 // reader that saw borrowed machines at all; the two *resolution*
1602 // callers in `validate`/`reconciler::domain` read a camp-local-only
1603 // loader and could not see a borrowing camp's fleet. There is now one
1604 // implementation and three callers.
1605 let fleet = resolve_fleet_inventory(workspace_root)?;
1606
1607 // Providers overlay here rather than inside `resolve_fleet_inventory`:
1608 // that function answers "which machines does this camp have", which is
1609 // the question with three readers. Providers have exactly one reader —
1610 // this load — so hoisting them would build a seam nothing crosses.
1611 let mut provider_origins = BTreeMap::new();
1612 overlay_source_providers(
1613 workspace_root,
1614 &fleet.sources,
1615 &mut providers,
1616 &mut provider_origins,
1617 );
1618
1619 Ok(Self {
1620 workspace_root: workspace_root.to_path_buf(),
1621 machines: fleet.machines,
1622 providers,
1623 machine_origins: fleet.origins,
1624 provider_origins,
1625 // Local, offline, and absent in most camps — the journal replays to
1626 // an empty map when the file isn't there (R850-T2).
1627 recovery_measurements: crate::recovery_journal::RecoveryJournal::at_workspace(
1628 workspace_root,
1629 )
1630 .replay(),
1631 services,
1632 domains,
1633 legacy_mirrors,
1634 workloads,
1635 topology,
1636 })
1637 }
1638
1639 /// Load the R215+ tree (`infra/`, `services/`, `domains/`) rooted at an
1640 /// arbitrary config directory instead of the hardcoded `.yah/`. This is the
1641 /// building block for multi-root deployments (W206 config layout (b), sibling
1642 /// `.noisetable/` trees) — see [`crate::multi_root`]. Part of R558-F4.
1643 ///
1644 /// `config_dir` is the `.X/` directory itself (e.g. `<parent>/.noisetable`);
1645 /// `workspace_root` remains the camp dir (the config dir's parent) so a
1646 /// component's `path` reference resolves against the same tree the classic
1647 /// [`CloudConfig::load`] uses. The legacy `.yah/cloud/` reads are skipped —
1648 /// multi-root deployments are post-R215 by construction — so `legacy_*`,
1649 /// `workloads`, and `topology` come back empty. Machines are read from
1650 /// `config_dir/infra/machines` directly (sibling trees declare their own
1651 /// inventory or none).
1652 ///
1653 /// R615-F2 decision, explicit rather than silent: **sources.toml overlay
1654 /// does NOT apply here.** This function
1655 /// exists specifically because a multi-root sibling tree (W206 layout
1656 /// (b), e.g. `.noisetable/`) is a *second config root inside the same
1657 /// camp*, not a second camp — `config_dir` is already wherever the
1658 /// caller decided this tree's infra lives, and `.yah/infra/sources.toml`
1659 /// (singular, tied to `paths::infra_dir(workspace_root)`) has no
1660 /// well-defined meaning for an arbitrary `config_dir` that isn't that
1661 /// path. A sibling tree that wants borrowed infra declares its own
1662 /// `sources.toml` under whichever root actually calls
1663 /// [`CloudConfig::load`] for it; `machine_origins`/`provider_origins`
1664 /// come back empty here, not wrong — there is nothing to overlay.
1665 pub fn load_from_config_dir(config_dir: &Path, workspace_root: &Path) -> Result<Self> {
1666 let providers = load_providers(&config_dir.join("infra").join("providers"))?;
1667 let services = load_services(&config_dir.join("services"), workspace_root)?;
1668 let domains = load_domains(&config_dir.join("domains"))?;
1669
1670 Self::cross_ref_validate(&providers, &services, &domains)?;
1671
1672 let machines = load_dir::<MachineConfig>(config_dir.join("infra").join("machines"))?;
1673
1674 Ok(Self {
1675 workspace_root: workspace_root.to_path_buf(),
1676 machines,
1677 providers,
1678 machine_origins: BTreeMap::new(),
1679 provider_origins: BTreeMap::new(),
1680 // Same reasoning as the sources overlay above: the recovery journal
1681 // is tied to `paths::recovery_journal(workspace_root)`, which has no
1682 // meaning for an arbitrary sibling config dir — and this loader
1683 // returns no `workloads` for a measurement to attach to anyway.
1684 recovery_measurements: BTreeMap::new(),
1685 services,
1686 domains,
1687 legacy_mirrors: vec![],
1688 workloads: vec![],
1689 topology: TopologyConfig::default(),
1690 })
1691 }
1692
1693 /// Cross-reference validation shared by [`CloudConfig::load`] and
1694 /// [`CloudConfig::load_from_config_dir`]: every mirror `providers.X.use =
1695 /// "<id>"` must resolve to a declared provider, and every domain route's
1696 /// `component = "<service>/<component-id>"` must resolve to a real component.
1697 fn cross_ref_validate(
1698 providers: &[ProviderConfig],
1699 services: &BTreeMap<String, ServiceWithMirrors>,
1700 domains: &BTreeMap<String, DomainConfig>,
1701 ) -> Result<()> {
1702 // Mirror `use = "<id>"` slots must resolve to a declared provider.
1703 let provider_ids: std::collections::HashSet<&str> =
1704 providers.iter().map(|p| p.id.as_str()).collect();
1705 for (svc_name, svc) in services {
1706 // Addressing is checked here rather than at probe time: a
1707 // truncated NodeId or a relative health path otherwise fails
1708 // with an error naming neither the file nor the field.
1709 svc.service.address.validate(svc_name)?;
1710 for (env, mirror) in &svc.mirrors {
1711 for (slot, body) in &mirror.providers {
1712 if let Some(id) = body.provider_id() {
1713 if !provider_ids.contains(id) {
1714 anyhow::bail!(
1715 "services/{svc_name}/mirrors/{env}.toml: \
1716 providers.{slot}.use = \"{id}\" — no such provider; \
1717 declare it at infra/providers/{id}.toml"
1718 );
1719 }
1720 }
1721 }
1722 // An `[[ingress]]` edge's own `use` is the same kind of
1723 // reference (R845) and gets the same check: a typo there is
1724 // otherwise invisible until `yah cloud apply` reaches the
1725 // Cloudflare arm and fails on a missing provider file.
1726 for (idx, edge) in mirror.ingress_edge_slice().iter().enumerate() {
1727 if let Some(id) = edge.provider_id.as_deref() {
1728 if !provider_ids.contains(id) {
1729 anyhow::bail!(
1730 "services/{svc_name}/mirrors/{env}.toml: \
1731 ingress[{idx}].use = \"{id}\" — no such provider; \
1732 declare it at infra/providers/{id}.toml"
1733 );
1734 }
1735 }
1736 }
1737 // R905. A `[build.<id>]` override keyed by a component this
1738 // service does not declare is always a typo, and it is the
1739 // silent kind: nothing reads the table for a component that
1740 // isn't there, so the environment goes on building with the
1741 // command the operator believed they had replaced — which is
1742 // exactly the defect the override exists to fix, wearing a
1743 // config that looks like the fix.
1744 for key in mirror.build.keys() {
1745 if !svc.service.components.iter().any(|c| &c.id == key) {
1746 let declared: Vec<&str> = svc
1747 .service
1748 .components
1749 .iter()
1750 .map(|c| c.id.as_str())
1751 .collect();
1752 anyhow::bail!(
1753 "services/{svc_name}/mirrors/{env}.toml: \
1754 [build.{key}] — no component {key:?} in \
1755 services/{svc_name}/service.toml (declared: {declared:?})"
1756 );
1757 }
1758 }
1759 }
1760 }
1761
1762 // R870-B11. Two bundle-tier components sharing a mount would stage
1763 // into the same `app/dist/<mount>/` prefix inside the service's one
1764 // assembled bundle and silently clobber each other on disk — the
1765 // exact failure class this ticket exists to fix, one level down
1766 // (there it was two components silently overwriting the same
1767 // *workload*; here it would be two components silently overwriting
1768 // the same *path inside* the workload). A mount is owned by exactly
1769 // one component; refuse the config before the clobber happens.
1770 //
1771 // R870-F23 widens the same loop to the workload tier rather than
1772 // adding a parallel one. A mount is owned by exactly one component
1773 // whichever tier serves it: two workload-tier components at one mount
1774 // would hand the inner door two upstream sets for one prefix, and two
1775 // components in *different* tiers at one mount is the same clobber
1776 // read from the routing side — the request reaches whichever of the
1777 // bundle and the workload the mount table happened to name. So the
1778 // rule is now "one component per mount, service-wide", and only the
1779 // explanation branches on tier.
1780 for (svc_name, svc) in services {
1781 let mut owner_by_mount: BTreeMap<String, (&str, DeployTier)> = BTreeMap::new();
1782 for component in &svc.service.components {
1783 let bundle_tier =
1784 component.kind == "mesofact-static" || component.kind == "mesofact-spa";
1785 if !bundle_tier && component.deploy != DeployTier::Workload {
1786 continue;
1787 }
1788 let mount = component
1789 .mount
1790 .as_deref()
1791 .map(normalize_mount)
1792 .unwrap_or_default();
1793 if let Some((existing, existing_tier)) =
1794 owner_by_mount.insert(mount.clone(), (&component.id, component.deploy))
1795 {
1796 let where_ = if mount.is_empty() {
1797 "the service root (no `mount`)".to_string()
1798 } else {
1799 format!("mount = \"/{mount}\"")
1800 };
1801 let why = if existing_tier == component.deploy {
1802 match component.deploy {
1803 DeployTier::Bundle => {
1804 "a bundle-tier component's mount is a storage prefix inside the \
1805 service's single assembled bundle (app/dist/<mount>/), so two \
1806 components at the same mount would stage into the same path and \
1807 silently overwrite each other"
1808 }
1809 DeployTier::Workload => {
1810 "a workload-tier component's mount is its prefix in the service's \
1811 inner-door route table, so two components at the same mount would \
1812 claim one prefix and requests would reach whichever the table \
1813 named"
1814 }
1815 }
1816 } else {
1817 "one is staged into the service bundle and the other deploys as its own \
1818 workload, so the mount names two different things that serve one prefix \
1819 — the inner door can only route it to one of them"
1820 };
1821 anyhow::bail!(
1822 "services/{svc_name}/service.toml: components \"{existing}\" and \
1823 \"{}\" both declare {where_} — {why}. Give one of them a distinct \
1824 `mount`.",
1825 component.id,
1826 );
1827 }
1828 }
1829 }
1830
1831 // Every domain route's `component = "<service>/<component-id>"` must
1832 // resolve to a real component.
1833 for (dom_name, dom) in domains {
1834 for (idx, route) in dom.routes.iter().enumerate() {
1835 let Some(component_ref) = route.mode.component() else {
1836 continue; // redirects don't reference components
1837 };
1838 let Some((svc_name, comp_id)) = split_component_ref(component_ref) else {
1839 anyhow::bail!(
1840 "domains/{dom_name}.toml: routes[{idx}].component = \
1841 \"{component_ref}\" — expected \"<service>/<component-id>\""
1842 );
1843 };
1844 let Some(svc) = services.get(svc_name) else {
1845 anyhow::bail!(
1846 "domains/{dom_name}.toml: routes[{idx}].component = \
1847 \"{component_ref}\" — no such service \"{svc_name}\" \
1848 under services/"
1849 );
1850 };
1851 let Some(component) = svc.service.components.iter().find(|c| c.id == comp_id)
1852 else {
1853 anyhow::bail!(
1854 "domains/{dom_name}.toml: routes[{idx}].component = \
1855 \"{component_ref}\" — service \"{svc_name}\" has no \
1856 component with id \"{comp_id}\""
1857 );
1858 };
1859
1860 // R746 / R931-B2: a mounted component must be routed where it
1861 // publishes. The publisher writes its bundle under the mount
1862 // and the front door looks a request up by its own path, so a
1863 // route path and a mount that disagree produce a 404 with its
1864 // cause two files away. Checked in both directions, since
1865 // either one alone is the same silent miss.
1866 //
1867 // Every mode that names a component is in scope, not Static
1868 // alone: [`RouteMode::component`] already narrowed the loop to
1869 // Static and Backend above (`route.mode.component()` returns
1870 // `None`, and `continue`s, for StaticBucket and Redirect,
1871 // neither of which references a component at all). A prior
1872 // version of this gate additionally required
1873 // `RouteMode::Static` here on the theory that a backend route
1874 // "proxies to an origin that owns its own paths" — but that
1875 // reasoning was about `origin`/`origin_path`, a field this
1876 // check never reads; it was silently skipping Backend
1877 // entirely rather than narrowly exempting the one field that
1878 // legitimately diverges (R931-B2, mutation-proved: a
1879 // Backend-mode component's mount could disagree with its
1880 // route path and `yah cloud validate` still exited 0).
1881 let mount = component.mount.as_deref().map(normalize_mount);
1882 let route_prefix = route_path_prefix(&route.path);
1883 if let Some(mount) = mount {
1884 if mount != route_prefix {
1885 anyhow::bail!(
1886 "domains/{dom_name}.toml: routes[{idx}].path = \
1887 \"{path}\" serves \"{component_ref}\", which \
1888 declares mount = \"/{mount}\" — a mounted \
1889 component publishes under its mount, so the route \
1890 must be \"/{mount}\" or \"/{mount}/*\" (or drop \
1891 the mount to serve from the service root)",
1892 path = route.path,
1893 );
1894 }
1895 } else if !route_prefix.is_empty() {
1896 anyhow::bail!(
1897 "domains/{dom_name}.toml: routes[{idx}].path = \
1898 \"{path}\" serves \"{component_ref}\", which declares \
1899 no `mount` — its bundle publishes at the service root, \
1900 so nothing is stored under \"/{route_prefix}\". Set \
1901 mount = \"/{route_prefix}\" on the component, or route \
1902 it at \"/*\"",
1903 path = route.path,
1904 );
1905 }
1906 }
1907 }
1908 Ok(())
1909 }
1910
1911 /// Look up a domain manifest by name (file stem under `.yah/domains/`).
1912 pub fn domain(&self, name: &str) -> Option<&DomainConfig> {
1913 self.domains.get(name)
1914 }
1915
1916 pub fn machine(&self, name: &str) -> Option<&MachineConfig> {
1917 self.machines.iter().find(|m| m.name == name)
1918 }
1919
1920 /// Look up a provider by id (matches `provider.id`, not the file stem).
1921 pub fn provider(&self, id: &str) -> Option<&ProviderConfig> {
1922 self.providers.iter().find(|p| p.id == id)
1923 }
1924
1925 /// Look up a service by name (matches `service.toml`'s `name` field).
1926 pub fn service(&self, name: &str) -> Option<&ServiceWithMirrors> {
1927 self.services.get(name)
1928 }
1929
1930 /// Look up a legacy mirror by camp name (pre-R215 .yah/cloud/mirrors/).
1931 pub fn legacy_mirror(&self, camp: &str) -> Option<&LegacyMirrorConfig> {
1932 self.legacy_mirrors.iter().find(|m| m.camp == camp)
1933 }
1934
1935 pub fn workload(&self, name: &str) -> Option<&WorkloadConfig> {
1936 self.workloads.iter().find(|w| w.spec.name == name)
1937 }
1938
1939 /// Every machine declaring `sovereign_group == group`, in declaration order.
1940 ///
1941 /// W305/R742-F3. A sovereign group has no file of its own — it exists only
1942 /// as the set of machines that name the same string — so "which boxes are
1943 /// the dev cluster" has to be *derived*, and before this it was not derived
1944 /// anywhere: `yah cloud rollout plan` still takes a hand-listed
1945 /// `--voter us-west-011 --voter us-west-013 …` for a fact the machine TOMLs
1946 /// already state (W314 gap 1).
1947 ///
1948 /// **This is not placement.** Resolving a group to its members is a
1949 /// *lookup*, and it stays outside [`RequiredSpec`] on purpose — see
1950 /// [`MachineConfig::sovereign_group`]. `migrate` calls this to pick the
1951 /// candidate set it then admits a workload against; nothing here filters
1952 /// scheduling, and adding `sovereign_group` to `matches` would still be the
1953 /// category error that doc warns about.
1954 ///
1955 /// An empty result means no machine declares `group`, which is
1956 /// indistinguishable from a typo — callers should say so with
1957 /// [`Self::declared_sovereign_groups`] rather than reporting "no
1958 /// candidates".
1959 pub fn machines_in_group(&self, group: &str) -> Vec<&MachineConfig> {
1960 self.machines
1961 .iter()
1962 .filter(|m| m.sovereign_group.as_deref() == Some(group))
1963 .collect()
1964 }
1965
1966 /// Every distinct `sovereign_group` declared by any machine, sorted.
1967 ///
1968 /// Exists so a bad `--to` names the real vocabulary instead of complaining
1969 /// abstractly — the same fail-loud shape [`taint_effect`]'s legal-key list
1970 /// gives `check_inert_taints`. Standalone machines (`None`) contribute
1971 /// nothing: "in no group" is not a group you can migrate *to*.
1972 pub fn declared_sovereign_groups(&self) -> Vec<&str> {
1973 let mut groups: Vec<&str> = self
1974 .machines
1975 .iter()
1976 .filter_map(|m| m.sovereign_group.as_deref())
1977 .collect();
1978 groups.sort_unstable();
1979 groups.dedup();
1980 groups
1981 }
1982
1983 /// F16 placement v1: the first machine satisfying every hard axis of `req`
1984 /// (region/zone/provider membership + mesh_tags superset). Declaration order
1985 /// in `.yah/infra/machines/` decides ties — deterministic-greedy, no
1986 /// backtracking. A fully-unconstrained `req` matches the first machine.
1987 ///
1988 /// Fails loud with the constraint summary and the candidate machine names
1989 /// when nothing matches, so `yah cloud apply` surfaces *why* placement
1990 /// failed instead of a silent empty set.
1991 pub fn resolve_machine(&self, req: &RequiredSpec) -> Result<&MachineConfig> {
1992 resolve_machine_among(&self.machines, req)
1993 }
1994
1995 /// F16 placement at horizontal scale: the first
1996 /// [`RequiredSpec::replica_count`] machines satisfying every hard axis of
1997 /// `req`, in declaration order (R844-F8).
1998 ///
1999 /// The N-valued form of [`Self::resolve_machine`], which is the N=1 case of
2000 /// this and not a different selector — both land in [`select_matching`].
2001 /// That shared bottom is what makes the deploy resolver
2002 /// (`reconciler::mesofact_bundle::resolve_bundle_machines`, which calls
2003 /// this) and the ingress planner's
2004 /// (`reconciler::ingress::resolve_ingress_placements`, which calls
2005 /// [`resolve_machines_among`] over the same `machines` slice) agree on the
2006 /// same N machines **by construction**. They must agree set-for-set, not
2007 /// merely in count: a front door aimed at nodes the workload was never
2008 /// deployed to renders a *subset* of the backends, which is the failure that
2009 /// looks like it worked.
2010 pub fn resolve_machines(&self, req: &RequiredSpec) -> Result<Vec<&MachineConfig>> {
2011 resolve_machines_among(&self.machines, req)
2012 }
2013
2014 /// F16 placement: first machine whose `mesh_tags` is a superset of
2015 /// `required`. Declaration order in `.yah/infra/machines/` decides ties.
2016 /// Empty `required` matches the first machine; callers should treat
2017 /// empty-required as "no constraint" and skip this lookup.
2018 ///
2019 /// Back-compat thin wrapper over [`CloudConfig::resolve_machine`] for the
2020 /// mesh-tags-only call sites that predate the topology axes.
2021 pub fn resolve_machine_by_mesh_tags(&self, required: &[String]) -> Option<&MachineConfig> {
2022 let req = RequiredSpec {
2023 mesh_tags: required.to_vec(),
2024 ..Default::default()
2025 };
2026 self.resolve_machine(&req).ok()
2027 }
2028
2029 /// Admission: resolve the target machine for a remote [`WorkloadSpec`],
2030 /// honoring the R594 mesh-tag node-selector annotation
2031 /// (`velveteen_exec::remote::NODE_SELECTOR_MESH_TAGS_ANNOTATION` =
2032 /// `yah.node-selector.mesh-tags`, comma-joined).
2033 ///
2034 /// The producer side (`velveteen_exec::remote::build_workload_spec`, R594) writes
2035 /// `TaskLocation::RemoteAny.mesh_tags` — e.g. `[tag:build-worker, arch:x86]`
2036 /// from [`qed::platform::build_worker_mesh_tags`] — into the workload's
2037 /// annotations. This is the consumer: candidates are restricted to machines
2038 /// whose `mesh_tags` are a **superset** of the requested set, so an amd64
2039 /// build lands on the `arch:x86` build-worker (us-west-002) and an arm64
2040 /// build on a `arch:arm` Pi5. Declaration order in `.yah/infra/machines/`
2041 /// breaks ties.
2042 ///
2043 /// An absent or empty annotation means "no mesh-tag constraint" — pre-R594
2044 /// behavior (any node), matching [`RequiredSpec::is_unconstrained`].
2045 ///
2046 /// This is the single admission seam: R572-F5 extends it with the capacity
2047 /// floor (workload request fits node allocatable−committed) and taint
2048 /// repulsion/affinity by enriching [`RequiredSpec::matches`] /
2049 /// [`Self::resolve_machine`]. Do not fork a second selector.
2050 pub fn admit_workload(&self, ws: &WorkloadSpec) -> Result<&MachineConfig> {
2051 self.resolve_machine(&admission_spec(ws, &self.workloads)?)
2052 }
2053
2054 /// Every machine that admits `ws`, in declaration order — the *pool*
2055 /// [`Self::admit_workload`] returns the head of (R605-T14).
2056 ///
2057 /// # Why a pool and not just the winner
2058 ///
2059 /// `tag:build-worker` is a statement that the tagged boxes are
2060 /// **interchangeable**: a build is booked against the tag, not against
2061 /// `us-west-002`. Returning one machine forced every caller to act as if it
2062 /// were booked against a name, and admission has no liveness input — so a
2063 /// tagged box that is asleep won the file-name tie-break and its builds
2064 /// failed rather than landing on the identical box next to it. That is
2065 /// exactly what happened on 2026-09-03 when `us-west-002` regained the tag.
2066 ///
2067 /// The fix is **not** to teach this function about liveness. It stays a pure
2068 /// function of the declared inventory (see `xtask/tests/fleet_build_placement.rs`
2069 /// on why a placement pin that needs the network is a flake). It hands the
2070 /// dispatcher the whole interchangeable set instead, and the dispatcher —
2071 /// which has the network — probes and fails over within it:
2072 /// `app/yah/cli/src/yubaba_client.rs`'s `MeshYubabaClient::deploy`.
2073 ///
2074 /// Order is the declaration order `admit_workload` already used, and callers
2075 /// should preserve it as their preference order rather than load-balancing
2076 /// across it: a retried build wants the node still holding its warm
2077 /// `target/`, which is the same reason [`first_match`] is deliberately
2078 /// first-fit.
2079 ///
2080 /// `Err` — never `Ok(vec![])` — when nothing admits `ws`, carrying the same
2081 /// message [`Self::admit_workload`] would have produced. "No node admits
2082 /// this" and "the pool is empty" are the same failure and must read the same.
2083 pub fn admit_workload_candidates(&self, ws: &WorkloadSpec) -> Result<Vec<&MachineConfig>> {
2084 let req = admission_spec(ws, &self.workloads)?;
2085 let all: Vec<&MachineConfig> = self.machines.iter().collect();
2086 let matched = matching(&all, &req);
2087 if matched.is_empty() {
2088 // Delegate the wording so the two paths cannot drift apart.
2089 return Err(first_match(&all, &req, DECLARED_POOL, EMPTY_DECLARED_POOL)
2090 .expect_err("matching() found nothing, so first_match cannot succeed"));
2091 }
2092 Ok(matched)
2093 }
2094
2095 /// [`Self::admit_workload`] restricted to the machines of one sovereign
2096 /// group (W305/R742-F3, `yah cloud migrate --to <group>`).
2097 ///
2098 /// Same [`RequiredSpec`], same [`RequiredSpec::matches`], same
2099 /// declaration-order tie-break — only the candidate *set* differs. That is
2100 /// the whole reason this is a narrowing of the admission seam rather than a
2101 /// second selector: a workload that cannot be scheduled onto a group's
2102 /// boxes must fail here for exactly the reason it would fail anywhere else,
2103 /// and `no-appliance` on the dev Pis (W305 finding 2) is precisely the case
2104 /// that must not be silently routed around by a migration verb.
2105 ///
2106 /// `Err` when the group has no members *or* when no member admits `ws`; the
2107 /// two are different mistakes, so callers wanting to tell them apart should
2108 /// check [`Self::machines_in_group`] first.
2109 pub fn admit_workload_in_group(
2110 &self,
2111 ws: &WorkloadSpec,
2112 group: &str,
2113 ) -> Result<&MachineConfig> {
2114 let members = self.machines_in_group(group);
2115 let empty_pool = format!(
2116 "(no machine declares sovereign_group = \"{group}\" — declared groups: {})",
2117 match self.declared_sovereign_groups().as_slice() {
2118 [] => "(none)".to_string(),
2119 gs => gs.join(", "),
2120 }
2121 );
2122 first_match(
2123 &members,
2124 &admission_spec(ws, &self.workloads)?,
2125 &format!("machines in sovereign group '{group}'"),
2126 &empty_pool,
2127 )
2128 }
2129}
2130
2131/// **The** placement selector: the first candidate satisfying every axis of
2132/// `req`, declaration order breaking ties, deterministic-greedy with no
2133/// backtracking.
2134///
2135/// Every path that picks a machine goes through here, and the only thing any
2136/// of them varies is *which machines are candidates* — never the predicate.
2137/// [`CloudConfig::resolve_machine`] passes the whole fleet;
2138/// [`CloudConfig::admit_workload_in_group`] passes one sovereign group's
2139/// members. That split is the point: a candidate-set narrowing composes with
2140/// the [`RequiredSpec`] axes for free, whereas expressing the same narrowing
2141/// *as* an axis would put facts like blast radius into a filter they must
2142/// never be in (see [`MachineConfig::sovereign_group`]).
2143///
2144/// So a new placement scope is a new candidate set plus a `pool` label, and a
2145/// new placement *constraint* is a field on [`RequiredSpec`] — those are the
2146/// two extension points, and neither is a second selector. `pool` and
2147/// `empty_pool` exist only so the failure names the set it actually searched;
2148/// a refusal that says "no candidates" without saying *among what* is one the
2149/// operator has to reconstruct by hand.
2150/// F16 placement v1 resolution over an explicit machine list — the
2151/// `.machines`-only half of [`CloudConfig::resolve_machine`], for callers that
2152/// have loaded just the machines tree rather than the whole cross-ref-validated
2153/// config.
2154///
2155/// R772: `resolve_ingress_placements` (`reconciler::ingress`) is the reason
2156/// this is `pub(crate)` rather than staying folded into
2157/// `CloudConfig::resolve_machine` — ingress collation walks every mirror in
2158/// the workspace and has no business hard-failing over an unrelated mirror's
2159/// `providers.X.use = "<id>"` typo, which is what going through
2160/// `CloudConfig::load`'s cross-ref validation would do. "Do not fork a second
2161/// selector" (see the module doc above) still holds: this is the *same*
2162/// [`first_match`], just handed a narrower candidate set than `self.machines`.
2163pub(crate) fn resolve_machine_among<'a>(
2164 machines: &'a [MachineConfig],
2165 req: &RequiredSpec,
2166) -> Result<&'a MachineConfig> {
2167 let all: Vec<&MachineConfig> = machines.iter().collect();
2168 first_match(&all, req, DECLARED_POOL, EMPTY_DECLARED_POOL)
2169}
2170
2171/// R844-F8: [`resolve_machine_among`] widened to the constraint's own replica
2172/// count — the first [`RequiredSpec::replica_count`] matching machines, in the
2173/// same declaration order, from the same candidate slice.
2174///
2175/// **The one entry point both resolvers share.**
2176/// `reconciler::ingress::resolve_ingress_placements` calls this directly and
2177/// `reconciler::mesofact_bundle::resolve_bundle_machines` reaches it through
2178/// [`CloudConfig::resolve_machines`], both over `cfg.machines` — so the ingress
2179/// planner and the deployer cannot pick different subsets. That is a structural
2180/// guarantee, not a tested coincidence, and it has to be: discovery aimed at a
2181/// node the bundle was never placed on publishes a hostname with a dead
2182/// backend behind it, and at scale > 1 the front door still answers from the
2183/// nodes that *did* get it.
2184///
2185/// Determinism is therefore part of correctness here. `machines` arrives in
2186/// file-name order (`load_dir`, pinned by
2187/// `machines_load_in_file_name_order_not_read_dir_order`), and selection is a
2188/// stable prefix of that order — so "the first two matching" is the same two
2189/// on both sides of the same tree.
2190pub(crate) fn resolve_machines_among<'a>(
2191 machines: &'a [MachineConfig],
2192 req: &RequiredSpec,
2193) -> Result<Vec<&'a MachineConfig>> {
2194 let all: Vec<&MachineConfig> = machines.iter().collect();
2195 select_matching(
2196 &all,
2197 req,
2198 req.replica_count(),
2199 DECLARED_POOL,
2200 EMPTY_DECLARED_POOL,
2201 )
2202}
2203
2204const DECLARED_POOL: &str = "declared machines";
2205const EMPTY_DECLARED_POOL: &str = "(no machines declared under .yah/infra/machines/)";
2206
2207fn first_match<'a>(
2208 candidates: &[&'a MachineConfig],
2209 req: &RequiredSpec,
2210 pool: &str,
2211 empty_pool: &str,
2212) -> Result<&'a MachineConfig> {
2213 Ok(select_matching(candidates, req, 1, pool, empty_pool)?
2214 .into_iter()
2215 .next()
2216 .expect("select_matching errors rather than returning short"))
2217}
2218
2219/// The N-selecting core of the placement selector: the first `want` candidates
2220/// satisfying `req`, in candidate order (R844-F8).
2221///
2222/// [`first_match`] is this with `want = 1`, which is why widening a caller to a
2223/// replica count cannot introduce a second selector — the predicate, the
2224/// ordering and the failure vocabulary are all one implementation.
2225///
2226/// **A shortfall is an error.** Matching one machine when two were asked for
2227/// returns `Err` naming both numbers and the pool searched, never a one-element
2228/// vec: a half-placed workload that reports success is worse than a failed
2229/// apply, because the front door then publishes a hostname whose backend set is
2230/// quietly smaller than declared. `want = 0` is the same mistake spelled
2231/// differently and is refused for the same reason.
2232fn select_matching<'a>(
2233 candidates: &[&'a MachineConfig],
2234 req: &RequiredSpec,
2235 want: usize,
2236 pool: &str,
2237 empty_pool: &str,
2238) -> Result<Vec<&'a MachineConfig>> {
2239 let names = || {
2240 if candidates.is_empty() {
2241 empty_pool.to_string()
2242 } else {
2243 candidates
2244 .iter()
2245 .map(|m| m.name.as_str())
2246 .collect::<Vec<_>>()
2247 .join(", ")
2248 }
2249 };
2250
2251 if want == 0 {
2252 anyhow::bail!(
2253 "replicas = 0 places {} on nothing — a placement that deploys to no machine is \
2254 a typo, not a scale-down; remove the slot instead",
2255 req.describe()
2256 );
2257 }
2258
2259 let mut matched = matching(candidates, req);
2260 if matched.len() >= want {
2261 matched.truncate(want);
2262 return Ok(matched);
2263 }
2264
2265 if want == 1 {
2266 anyhow::bail!(
2267 "no candidates matching {} — {pool}: {}",
2268 req.describe(),
2269 names()
2270 );
2271 }
2272 anyhow::bail!(
2273 "only {} of {want} machines match {} — placing fewer than the declared \
2274 `replicas = {want}` would publish a smaller backend set than the mirror asks for; \
2275 {pool}: {}",
2276 matched.len(),
2277 req.describe(),
2278 names()
2279 )
2280}
2281
2282/// The predicate itself, applied to every candidate in order — the one place
2283/// `req.matches` is called on a set.
2284///
2285/// [`select_matching`] takes a prefix of this; [`CloudConfig::admit_workload_candidates`]
2286/// takes all of it. Keeping both on this function is what makes "the pool the
2287/// dispatcher failed over within" and "the machine admission picked" the same
2288/// answer by construction rather than by two filters that happen to agree.
2289fn matching<'a>(candidates: &[&'a MachineConfig], req: &RequiredSpec) -> Vec<&'a MachineConfig> {
2290 candidates
2291 .iter()
2292 .copied()
2293 .filter(|m| req.matches(m))
2294 .collect()
2295}
2296
2297/// The [`RequiredSpec`] a workload is admitted against — the single place the
2298/// axes are derived from a [`WorkloadSpec`].
2299///
2300/// Extracted from [`CloudConfig::admit_workload`] so that
2301/// [`CloudConfig::admit_workload_in_group`] narrows the candidate set without
2302/// restating the axes. Forking that derivation is how the two paths would
2303/// silently disagree about whether a workload fits a node.
2304///
2305/// # It admits a group, not a workload (R860-T4 / W338)
2306///
2307/// The axes come from [`placement_group`] — `ws` plus the transitive closure of
2308/// its `local` requirement edges — because those members are placed together or
2309/// not at all. Capacity is their **sum**, archetype repulsion their **union**,
2310/// and mesh tags their union too. `prefer-local` and `anywhere` edges bind
2311/// nothing: a spec with neither `requires` nor `depends_on` local edges has a
2312/// group of exactly itself and resolves byte-identically to the pre-R860 axes.
2313///
2314/// This is the **only** gate. Node election is CLI-side
2315/// (`MeshYubabaClient::elect_node`, which picks a live member of the pool this
2316/// produces); the yubaba node process accepts whatever it is handed and never
2317/// re-checks placement, so a wrong group here is not caught downstream.
2318///
2319/// @yah:ticket(R860-T4, "Admission: place the transitive closure of `local` edges as one group, not one workload")
2320/// @yah:status(review)
2321/// @yah:phase(P1)
2322/// @yah:at(2026-09-05T18:29:13Z)
2323/// @yah:assignee(agent:bundle-anthropic-ashguard)
2324/// @yah:parent(R860)
2325/// @yah:next("W338 §Placement consequences 1 and 2. `admission_spec()` (config.rs:1974-1998) derives its axes from ONE spec; it must derive them from the group — the transitive closure of `local` requirement edges over `effective_requirements()`. `prefer-local` and `anywhere` edges do NOT bind the group. Three consequences: memory/cpu floor becomes the SUM of the group's requests, not the requirer's alone; `repel_archetype` becomes the union over members (so a group containing an Appliance is repelled by `no-appliance` even if the requirer is a Server); and the group is non-drainable if ANY member is an Appliance, which today is a per-workload check at yubaba/src/lib.rs:3117-3128 and now has to be computed over a set.")
2326/// @yah:verify("cargo test -p cloud --lib config")
2327/// @yah:gotcha("Node election is CLI-side, not cluster-side: `MeshYubabaClient::elect_node` (app/yah/cli/src/yubaba_client.rs:235-268) calls `admit_workload_candidates` (config.rs:1753), picks one node, and POSTs the deploy there. The yubaba node process never decides placement — it accepts whatever it is handed. So group admission has to be right in `config.rs` because there is no second gate downstream to catch it.")
2328/// @arch:see(.yah/docs/working/W338-workload-dependencies-and-appliance-composition.md)
2329/// @yah:depends_on(R860-T1)
2330/// @yah:handoff("ADMISSION NOW PLACES A GROUP, NOT A WORKLOAD. `admission_spec` (oss/yubaba/crates/cloud/src/config.rs:2009) takes `(ws, declared: &[WorkloadConfig])` and derives every axis from `placement_group(ws, declared)` (:2108) — the transitive closure of `local` requirement edges over `effective_requirements()`, traversing `Requirement::provides` where present and resolving by ident against `cfg.workloads` (.yah/infra/workloads/) otherwise, mesh-identity first and workload name second. Capacity is the SUM of the members' `memory_request_mb()` / `resources.cpu_millis` (saturating). Only `local` binds: `prefer-local` and `anywhere` (which every legacy `depends_on` folds into) are skipped, so a spec without local edges has a group of exactly itself and its axes are bit-identical to the pre-R860 derivation.")
2331/// @yah:handoff("REPEL BECAME A SET. `RequiredSpec::repel_archetype: Option<LifecycleArchetype>` is now `repel_archetypes: Vec<LifecycleArchetype>` (config.rs:3878), the union over group members; `matches` (:3971) rejects a node carrying `no-<taint_key()>` for ANY of them, `describe` emits one `not-tainted(...)` part per archetype, `is_unconstrained` tests `is_empty()`. The field is `#[serde(skip)]`, so no wire or schema drift, and grep over app/ crates/ oss/ xtask/ finds no other referent of the old name and no `RequiredSpec { .. }` literal outside config.rs — the rename is contained. `admit_workload` / `admit_workload_candidates` / `admit_workload_in_group` signatures are unchanged; all three now pass `&self.workloads`.")
2332/// @yah:handoff("CYCLE GUARD, AND THE BUG IT TOOK TO GET RIGHT. The walker keeps TWO visited lists: `in_group` (member mesh identities) and `expanded` (requirement idents already resolved). The first version used one list and was silently wrong in the common case — a requirement's ident IS its provider's mesh identity, so marking the ident before resolving made every provider look already-present and `placement_group` returned a group of one. Five of the new tests caught it. If you refactor this, keep the two questions separate.")
2333/// @yah:handoff("ELECT_NODE NEEDS NO CHANGE FOR THIS TICKET — read it (app/yah/cli/src/yubaba_client.rs:235-268). It calls `admit_workload_candidates`, so it now receives a pool already filtered to nodes that can host the WHOLE group, then probes for liveness within it. That is correct for T4 because only the requirer is deployed today. It becomes load-bearing at R860-T6: `supply = \"self\"` provisioning MUST reuse the node URL `elect_node` returned for the requirer and must not re-elect per member — the probe is liveness-sensitive, so a second election can legally return a different member of the same pool and split the group across two nodes.")
2334/// @yah:handoff("DECISIONS THE BRIEF DID NOT COVER, all recorded in doc comments at the site. (1) `mesh_tags` are UNIONED over the group — the axis is already a superset/AND check, so a node that cannot host one member cannot host the group; zero regression risk since nothing in the tree declares `requires` yet. (2) `nodes` (the R833-F8 operator pin) stays REQUIRER-ONLY: it is a membership list, so intersecting two members' pins can yield an empty vec, which the axis reads as no-constraint — the exact inverse of the conflict. (3) `requires_taint` is a single Option: the requirer's wins, else the first member declaring one. Two members demanding DIFFERENT taints is not representable and would be an unplaceable group; widening that axis to a set is a follow-up if a real case appears. (4) An unresolvable `local` ident is SKIPPED, not an error — admission is a pure function of the declared inventory and must not start refusing deploys over a provider a later ticket declares; the cost is that its request does not count toward the floor, which is the exposure `depends_on` has always had.")
2335/// @yah:handoff("DRAINABILITY: placement half landed, node half deliberately NOT touched. `group_is_drainable(members)` (config.rs:2160) is the set-valued predicate W338 §Placement consequences 2 asks for — false as soon as any member is an Appliance — and the `no-appliance` repulsion that follows from it is enforced through `repel_archetypes`. The node-side loop `drain_workloads` (oss/yubaba/crates/yubaba/src/lib.rs, the R572-F4 archetype_registry skip) still decides per workload and knows nothing about requirement edges, so a Server bound to an Appliance by a `local` edge would still be drained alone. Not fixed here for two reasons: that file has three sessions live in it (the brief named them), and the fix needs group edges plumbed to the node process, which is R860-T6's rail rather than a local edit. `yubaba` already depends on `cloud`, so the predicate is directly callable from there when that plumbing exists.")
2336/// @yah:verify("BASELINE recorded before editing, tree anchor 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2: `cargo test -p yah-cloud --lib` (from oss/yubaba) = 1081 passed, 0 failed, 4 ignored, exit 0. AFTER: 1090 passed, 0 failed, 4 ignored, exit 0 — +9, exactly the nine tests added. `cargo check -p yah-cloud --all-targets` exit 0, and `cargo check -p yubaba --all-targets` exit 0 as well (yubaba consumes `cloud`, so it is where the `repel_archetypes` rename would have surfaced). Every exit code echoed explicitly, never inferred from an empty grep.")
2337/// @yah:verify("NEW TESTS (config.rs `mod tests`, R860-T4 section at the end): a_local_edge_binds_the_provider_into_the_placement_group; prefer_local_and_anywhere_edges_do_not_bind_the_group (covers a legacy `depends_on` too); the_group_is_the_transitive_closure_and_traverses_inline_provides; an_ident_cycle_closes_the_group_instead_of_looping_forever; an_unresolvable_local_ident_is_skipped_rather_than_refused; the_capacity_floor_is_the_sum_of_the_group_not_the_requirer_alone (a 300 MiB node refuses two 256 MiB members and the error names memory_mb>=512; a 512 MiB node admits); a_server_requiring_an_appliance_locally_is_repelled_by_no_appliance (same requirer alone still lands on the tainted Pi, so the repulsion provably comes from the edge); a_group_containing_an_appliance_is_not_drainable; a_spec_with_no_local_edges_admits_exactly_as_it_did_before.")
2338/// @yah:gotcha("The camp's `yah build run` rail killed three consecutive verification runs against the shared oss/yubaba/target dir: each ended with only `Blocking waiting for file lock on build directory` in the log and no exit code, after 121s / 720s. The green result above was obtained with `CARGO_TARGET_DIR=/tmp/r860t4-target`, which sidesteps the contended lock at the cost of one cold dep build. Worth reaching for directly when the yubaba target dir is busy rather than burning three cycles discovering it.")
2339/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
2340/// @yah:next("R860-T6 (supply = \"self\"): deploy the group's non-requirer members onto the node `elect_node` already returned for the requirer — do NOT re-elect per member, or a liveness probe can split the group across two nodes. `placement_group` (config.rs:2108) hands you the member specs in traversal order, requirer first.")
2341/// @yah:next("Node-side drain is still per-workload: teach `drain_workloads` (oss/yubaba/crates/yubaba/src/lib.rs) to consult `cloud::config::group_is_drainable` over the requesting workload's placement group once R860-T6 plumbs group membership to the node. Left untouched here on purpose — three sessions were live in that file.")
2342/// @yah:handoff("LEADER RE-VERIFIED (session:69b18855, independent of the courier's self-report). `cargo test -p yah-cloud --lib` from oss/yubaba: 1090 passed / 0 failed / 4 ignored, exit 0, against the courier's recorded 1081/0/4 baseline — +9 = exactly its new tests. Confirmed by content in config.rs: `placement_group` :2120 with the `req.locality != Locality::Local` guard at :2136 (so `prefer-local` and `anywhere` correctly do NOT bind), `group_is_drainable` :2172, and `RequiredSpec::repel_archetype: Option<_>` widened to `repel_archetypes: Vec<_>` at :3890 with the union built at :2039-2050 and enforced at :4004/:4044. The repel rename is `#[serde(skip)]`, so no wire or schema drift.")
2343/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
2344/// @yah:verify("cargo test -p yah-cloud --lib (from oss/yubaba): 1090 passed / 0 failed / 4 ignored, exit 0, vs a 1081/0/4 baseline. Exit codes echoed explicitly throughout rather than inferred from an empty grep — the trap that cost R860-T1 three misses.")
2345/// @yah:gotcha("CORRECTION FROM R860-T6, and the leader propagated the error so it is worth naming: this ticket's handoff asserted \\\"`yubaba` already depends on `cloud`, so the predicate is directly callable from there\\\". THAT IS WRONG. `cloud` is a DEV-dependency of yubaba only — oss/yubaba/crates/yubaba/Cargo.toml:150-152, under the comment \\\"Integration test harness\\\" — and cloud's own Cargo.toml records that the runtime yubaba→cloud edge was DELIBERATELY avoided from R374-F3 onward. The leader repeated the claim verbatim in R860-T6's dispatch brief; T6's courier checked it against the manifest instead of trusting it, which is the only reason it did not become a runtime dependency inversion. Resolution: `group_is_drainable`'s body moved down to `workload_spec::group_is_drainable` (workload-spec/src/lib.rs:2365), the shared home both crates already depend on, and `cloud::config::group_is_drainable` (config.rs:2240) now delegates to it keeping its signature. Verified after the move: yah-cloud still 1093/0/4, yah-workload-spec 171+98/0.")
2346/// @yah:handoff("Tree anchor at handoff: 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2 — the shared tree as I left it. Diff against it (`git diff 0a85122cdb33dbf97ebc04b84e07d9cfc049c0b2..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
2347/// @yah:verify("RE-VERIFIED AT HEAD 00ee20d1 (session:aa5e882d, 2026-09-05). `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` = 1141 passed / 0 failed / 4 ignored, exit 0 (was 1093 at the first leader's check, 1110 at the second; the deltas are peers' tests). Group placement confirmed by content in oss/yubaba/crates/cloud/src/config.rs: `placement_group` derivation at :2073/:2108, `repel_archetypes: Vec<LifecycleArchetype>` at :3878. NOTE FOR ANYONE RE-RUNNING THIS: `cargo test -p yah-cloud --lib` from the repo root FAILS with \"package `yah-cloud` cannot be tested because it requires dev-dependencies and is not a member of the workspace\" — yah-cloud lives in the oss/yubaba workspace, so the invocation needs `--manifest-path oss/yubaba/Cargo.toml`.")
2348fn admission_spec(ws: &WorkloadSpec, declared: &[WorkloadConfig]) -> Result<RequiredSpec> {
2349 let group = placement_group(ws, declared);
2350
2351 // R894-F1: trust is a declared axis, and the substrate floor it implies is
2352 // a REFUSAL here — not a tag, not a downgrade, not a warning.
2353 //
2354 // Every other axis in this function narrows the candidate set: "which nodes
2355 // can host this". Trust is not that question. A `yah.trust = untrusted`
2356 // spec asking for `yah.exec = native` is not unplaceable — it is
2357 // *incoherent*, and there is no fleet on which it becomes coherent. Making
2358 // it a mesh tag would render it as "no node admits this", which reads like
2359 // a capacity problem and sends the operator to look at machines.
2360 //
2361 // It mirrors `wants_microvm`'s no-silent-downgrade semantics from the other
2362 // direction: kamaji refuses to run a microVM-marked spec as a container
2363 // because that delivers less isolation than was asked for; admission
2364 // refuses to place an untrusted spec on a weaker substrate for exactly the
2365 // same reason, one layer earlier, where the operator can still read why.
2366 //
2367 // Checked per member over the whole `placement_group`, not just `ws`.
2368 // R860-T4 made a `local` edge co-place its provider, so an untrusted member
2369 // reaching a node is reaching it whether or not the requirer is the
2370 // untrusted one — and each member carries its own pair, so the check is
2371 // per-member rather than over a group-wide maximum.
2372 check_trust_substrate(&group)?;
2373
2374 // Capacity is the group's demand, not the requirer's (W338 §Placement
2375 // consequences 1). Saturating rather than wrapping: an absurd declared
2376 // request must read as "nothing is big enough", never as a small number.
2377 //
2378 // `memory_request_mb()` and NOT `resources.memory_mb`: the latter is a
2379 // cgroup ceiling, and reading a ceiling as a floor made `for_forge`'s
2380 // deliberately-roomy 32 GiB limit mean "only place me on a 32 GiB node".
2381 // That excluded every build-worker in the fleet but one. The accessor falls
2382 // back to `resources.memory_mb` when no request is declared, so specs that
2383 // never set one are admitted exactly as before.
2384 let mut memory_mb: u32 = 0;
2385 let mut cpu_millis: u32 = 0;
2386 // R572-F5 taint repulsion, unioned over the group (W338 §Placement
2387 // consequences 2): a group is non-drainable — and `no-appliance`-repelled —
2388 // if *any* member is an Appliance, even when the requirer is a Server.
2389 //
2390 // R876-B7 inverted the sense. Repulsion is now unconditional in `matches`,
2391 // so what this loop collects is still the group's archetype union, but it is
2392 // converted below into the complementary TOLERATION set. Same predicate,
2393 // stated from the other side.
2394 let mut group_archetypes: Vec<LifecycleArchetype> = Vec::new();
2395 // Mesh tags are already AND-ed (a machine must be a superset), so unioning
2396 // them over the group is the same predicate applied to every member: a node
2397 // that cannot host one member cannot host the group.
2398 let mut mesh_tags = node_selector_mesh_tags(ws);
2399
2400 for member in &group {
2401 memory_mb = memory_mb.saturating_add(member.memory_request_mb());
2402 cpu_millis = cpu_millis.saturating_add(member.resources.cpu_millis);
2403 let arch = member.effective_archetype();
2404 if !group_archetypes.contains(&arch) {
2405 group_archetypes.push(arch);
2406 }
2407 for tag in node_selector_mesh_tags(member) {
2408 if !mesh_tags.contains(&tag) {
2409 mesh_tags.push(tag);
2410 }
2411 }
2412 }
2413
2414 // R860-T5 / W338 §Placement consequences 3: per-node native-exec
2415 // capability. Computed over the group for the same reason every other axis
2416 // is — a `local` edge to a native provider makes the *requirer* unplaceable
2417 // on a node without the backend, even when the requirer is an ordinary
2418 // container workload. This is the `supply = "self"` precondition W338 names:
2419 // a self-supplied native provider has to be placeable where its requirer
2420 // lands, and until now nothing upstream could see whether it was.
2421 //
2422 // Appended to `mesh_tags` rather than given its own field: the axis is
2423 // already an AND-ed superset check against `machine.mesh_tags`, `describe`
2424 // already renders it, and `RequiredSpec` needs no new shape. See
2425 // [`NATIVE_EXEC_MESH_TAG`] for why a tag and not a taint.
2426 if group.iter().any(WorkloadSpec::wants_native_exec)
2427 && !mesh_tags.iter().any(|t| t == NATIVE_EXEC_MESH_TAG)
2428 {
2429 mesh_tags.push(NATIVE_EXEC_MESH_TAG.to_string());
2430 }
2431
2432 // R894-F1 / R860-T5's deferred second half: the identical axis for the
2433 // microVM backend. Same `if`, same group-wide reasoning, same reason it is a
2434 // tag and not a taint — see [`MICROVM_MESH_TAG`] for why the precondition
2435 // R860-T5 was waiting on (a node that can actually boot a guest) is now met.
2436 //
2437 // This is what makes the trust floor above land somewhere real: without it,
2438 // an untrusted workload passes the coherence check and is then placed on a
2439 // node whose kamaji has no microVM backend, which refuses at dispatch.
2440 if group.iter().any(WorkloadSpec::wants_microvm)
2441 && !mesh_tags.iter().any(|t| t == MICROVM_MESH_TAG)
2442 {
2443 mesh_tags.push(MICROVM_MESH_TAG.to_string());
2444 }
2445
2446 Ok(RequiredSpec {
2447 mesh_tags,
2448 // R833-F8: imperative node pin. Derived here alongside the inferred
2449 // mesh tags rather than short-circuiting the resolver, so a pinned
2450 // workload is still checked against capacity and taints.
2451 //
2452 // Requirer-only on purpose: the pin is what the operator typed on
2453 // *this* deploy, and `nodes` is a membership list, so intersecting two
2454 // members' pins could yield an empty vec — which this axis reads as "no
2455 // constraint", i.e. the exact opposite of the conflict it represents.
2456 nodes: node_selector_node(ws).into_iter().collect(),
2457 memory_mb,
2458 cpu_millis,
2459 // R876-B7: the archetype union, restated as tolerations — every
2460 // repelling key that is NOT this group's own class. A Server group
2461 // tolerates `no-appliance` and `no-job` and is still blocked by
2462 // `no-server`, which is precisely what the pre-B7 `repel_archetypes`
2463 // axis computed. That equivalence is the migration: the `admit_workload`
2464 // path's placement answers are unchanged for every fleet machine, while
2465 // the mirror-declared path — which could never populate an archetype set
2466 // and so read no taints at all — becomes repel-by-default.
2467 //
2468 // Derived from `LifecycleArchetype::ALL` rather than a literal list, so
2469 // a fourth archetype is tolerated by unrelated groups automatically,
2470 // exactly as `taint_effect` already derives the repulsion half.
2471 tolerates: LifecycleArchetype::ALL
2472 .into_iter()
2473 .filter(|a| !group_archetypes.contains(a))
2474 .map(|a| format!("no-{}", a.taint_key()))
2475 .collect(),
2476 // R572-F5: taint affinity from the requires-taint annotation. The
2477 // requirer's wins; otherwise the first member that declares one, since
2478 // the group shares a node and this axis holds a single key. Two members
2479 // demanding *different* taints is not representable here and would be
2480 // an unplaceable group anyway — see the R860-T4 handoff.
2481 requires_taint: group
2482 .iter()
2483 .find_map(|m| m.requires_taint().map(str::to_owned)),
2484 ..Default::default()
2485 })
2486}
2487
2488/// Refuse any group member whose declared trust exceeds what its requested
2489/// [`workload_spec::ExecSubstrate`] provides (R894-F1).
2490///
2491/// The rule in one line: **a caller may request a stricter substrate than its
2492/// trust level requires, never a looser one.** `Trusted` floors at
2493/// [`workload_spec::ExecSubstrate::Native`] — the bottom of the ordering, i.e. no constraint,
2494/// which is what every workload in the fleet has today — and `Untrusted` floors
2495/// at [`workload_spec::ExecSubstrate::MicroVm`], which is the operator's 2026-09-11 call that
2496/// untrusted code never shares a kernel with the fleet.
2497///
2498/// A malformed [`workload_spec::TRUST_ANNOTATION`] is refused too, rather than
2499/// resolving to either side. See [`workload_spec::TrustDeclError`] for why
2500/// guessing in either direction is worse than a named refusal.
2501///
2502/// # Why here and not in `RequiredSpec::matches`
2503///
2504/// `matches` answers "can this node host this group". Trust coherence is a
2505/// property of the *spec alone* — no node makes an untrusted-and-native spec
2506/// legal — so it belongs on the path in, where it can fail with its own
2507/// sentence. Putting it in the predicate would spend a fleet scan to conclude
2508/// "no candidates", which names the wrong thing.
2509///
2510/// It is sited inside [`admission_spec`] and not in each `admit_*` method for
2511/// the reason that function's own docs give: `admission_spec` is the one place
2512/// the three admission entry points share, so a fourth entry point cannot be
2513/// added that skips this. Returning `Result` from it is what makes that
2514/// structural rather than a convention.
2515fn check_trust_substrate(group: &[WorkloadSpec]) -> Result<()> {
2516 for member in group {
2517 let trust = member.trust().map_err(|e| {
2518 anyhow::anyhow!(
2519 "workload '{}' has an unreadable trust declaration: {e}",
2520 member.name
2521 )
2522 })?;
2523 let floor = trust.minimum_substrate();
2524 let requested = member.exec_substrate();
2525 if requested < floor {
2526 bail!(
2527 "workload '{}' declares {}={} but requests the {} substrate, which is weaker than \
2528 the {} minimum that trust level requires — untrusted code does not share a kernel \
2529 with the fleet, so set {}={} on this spec (a stricter substrate is always allowed, \
2530 a weaker one never is)",
2531 member.name,
2532 workload_spec::TRUST_ANNOTATION,
2533 trust.as_str(),
2534 requested.as_str(),
2535 floor.as_str(),
2536 workload_spec::NATIVE_EXEC_ANNOTATION,
2537 floor.annotation_value().unwrap_or("<container>"),
2538 );
2539 }
2540 }
2541 Ok(())
2542}
2543
2544/// The workloads that must be placed together with `ws`: the transitive closure
2545/// of `local` requirement edges over [`WorkloadSpec::effective_requirements`],
2546/// starting at the requirer (R860-T4 / W338 §"Each member keeps its own mesh
2547/// identity").
2548///
2549/// **Only `local` binds.** `prefer-local` explicitly "never blocks placement"
2550/// (W338's locality table) and `anywhere` is an ordinary service dependency —
2551/// treating either as a co-scheduling constraint would turn every `depends_on`
2552/// in the tree into one, since the legacy field folds in as `anywhere` + `wait`.
2553///
2554/// A group is **not** a new addressable object: every member keeps its own mesh
2555/// identity, spec and healthcheck (W338). This function returns the members'
2556/// specs so admission can take the sum / union over them, and nothing here
2557/// deploys, provisions or tears anything down — `supply = "self"` provisioning
2558/// is R860-T6 and per-node native-exec capability is R860-T5.
2559///
2560/// Two ways a member is reached, in this order:
2561/// - [`Requirement::provides`], the inline spec a `supply = "self"` requirement
2562/// carries;
2563/// - otherwise an ident lookup against `declared` (`.yah/infra/workloads/`),
2564/// matched on mesh identity first and on workload name second, because those
2565/// coincide for every spec in the tree today but the requirement is written in
2566/// the mesh-identity currency.
2567///
2568/// An ident that resolves to neither is **skipped**, not an error: admission is
2569/// a pure function of the declared inventory and must not start failing deploys
2570/// over a provider that a not-yet-written ticket will declare. The cost is that
2571/// its request does not count toward the floor, which is the same exposure
2572/// `depends_on` has always had.
2573///
2574/// **Cycle-guarded.** `validate::check_requires` bounds `provides` *nesting* to
2575/// depth 1 but nothing stops two separately-declared specs from requiring each
2576/// other, and this closure would otherwise not terminate. Each requirement ident
2577/// is resolved at most once and each member joins the group at most once, so a
2578/// cycle simply closes the group.
2579pub fn placement_group(ws: &WorkloadSpec, declared: &[WorkloadConfig]) -> Vec<WorkloadSpec> {
2580 let mut members = vec![ws.clone()];
2581 // Two separate visited sets, because the two questions differ: `in_group`
2582 // stops a workload being added twice, `expanded` stops an ident being
2583 // resolved twice. Folding them into one list makes the ident of a member
2584 // already in the group indistinguishable from the member itself — and since
2585 // a requirement's ident *is* its provider's mesh identity, that reads every
2586 // provider as already-present and silently returns a group of one.
2587 let mut in_group: Vec<String> = vec![group_key(ws)];
2588 let mut expanded: Vec<String> = Vec::new();
2589 let mut next = 0;
2590
2591 while next < members.len() {
2592 let requirements = members[next].effective_requirements();
2593 next += 1;
2594 for req in requirements {
2595 if req.locality != Locality::Local {
2596 continue;
2597 }
2598 if expanded.contains(&req.ident.0) {
2599 continue;
2600 }
2601 expanded.push(req.ident.0.clone());
2602
2603 let provider = match req.provides.as_deref() {
2604 Some(spec) => spec.clone(),
2605 None => match resolve_requirement_ident(&req.ident, declared) {
2606 Some(spec) => spec,
2607 None => continue,
2608 },
2609 };
2610 let key = group_key(&provider);
2611 if in_group.contains(&key) {
2612 continue;
2613 }
2614 in_group.push(key);
2615 members.push(provider);
2616 }
2617 }
2618
2619 members
2620}
2621
2622/// Whether a placement group may be drained off its node (W338 §Placement
2623/// consequences 2): false as soon as **any** member is an Appliance.
2624///
2625/// The set-valued form of the per-workload check the node itself makes in
2626/// `drain_workloads` (`oss/yubaba/crates/yubaba/src/lib.rs`), which skips an
2627/// Appliance by its own archetype and knows nothing about requirement edges. A
2628/// `Server` bound to an Appliance by a `local` edge has to move with it or not
2629/// at all, so draining it alone breaks the group the same way placing it alone
2630/// would.
2631///
2632/// R860-T6 moved the body to [`workload_spec::group_is_drainable`] and left this
2633/// signature untouched. The node's `drain_workloads` needs the identical
2634/// predicate, and yubaba has no runtime dependency on this crate by design
2635/// (R374-F3) — so the one implementation now lives in the crate both sides
2636/// already depend on, rather than being copied into the second caller.
2637pub fn group_is_drainable(members: &[WorkloadSpec]) -> bool {
2638 workload_spec::group_is_drainable(members)
2639}
2640
2641/// Identity a placement-group member is deduplicated by — its mesh identity,
2642/// which is the currency [`Requirement::ident`] is written in.
2643fn group_key(ws: &WorkloadSpec) -> String {
2644 ws.expose.mesh.identity.0.clone()
2645}
2646
2647/// Resolve a requirement's ident to a separately-declared spec: mesh identity
2648/// first, workload file name second.
2649fn resolve_requirement_ident(
2650 ident: &workload_spec::MeshIdent,
2651 declared: &[WorkloadConfig],
2652) -> Option<WorkloadSpec> {
2653 declared
2654 .iter()
2655 .find(|w| w.spec.expose.mesh.identity == *ident)
2656 .or_else(|| declared.iter().find(|w| w.spec.name == ident.0))
2657 .map(|w| w.spec.clone())
2658}
2659
2660/// The mesh tag a node declares to advertise that its kamaji can run **native**
2661/// (fork+exec) workloads — R860-T5 / W338 §"Placement consequences" 3.
2662///
2663/// A workload marked `yah.exec = native` ([`WorkloadSpec::wants_native_exec`])
2664/// is not containerized: kamaji fork+execs it on the node's own userland. That
2665/// backend only exists when the node's kamaji was **built** with the
2666/// `native-exec` cargo feature and **started** with `--native-exec-dir`
2667/// (`oss/kamaji/crates/kamaji-bin/src/main.rs`). Both are node-local startup
2668/// decisions, invisible to everything upstream — so before this tag, placement
2669/// happily elected a node whose kamaji then refused the deploy with
2670/// `BackendRefused: ... no native backend is available (native backend not
2671/// configured — start kamaji with --native-exec-dir)`. That is exactly how the
2672/// mesh lost its coordination server for 25 hours on 2026-09-03 (R858: raft
2673/// leadership moved headscale, a native workload, to `us-south-001`, which has
2674/// no such kamaji). [`admission_spec`] now requires this tag whenever any
2675/// placement-group member is native, which turns that dispatch-time surprise
2676/// into a placement precondition.
2677///
2678/// # Why a mesh tag and not a taint
2679///
2680/// The two vocabularies on [`MachineConfig`] mean opposite things. `mesh_tags`
2681/// are **positive capability** matched as a superset — "this node CAN" — which
2682/// is precisely the claim being made, and an extra tag on a machine can only
2683/// ever make it match *more* requirement sets, so declaring it is regression-
2684/// free. `taints` are **repulsion** — "keep this class off" — and would have to
2685/// be inverted (`no-native-exec` on every node lacking the backend, i.e. the
2686/// declaration burden falls on the majority) *and* taught to
2687/// [`taint_effect`], or [`crate::validate::check_inert_taints`] would correctly
2688/// lint the key dead.
2689///
2690/// # The `cap:` namespace
2691///
2692/// New here. The live prefixes are `tag:` (role — `tag:build-worker`,
2693/// `tag:qed`, `tag:cloud-runner`), `arch:` and `os:` (facts about the silicon
2694/// and userland), and `tier:` is reserved for the environment axis (R763, see
2695/// [`crate::validate::check_retired_arch_tags`]). A *capability the daemon was
2696/// configured with* is none of those: it is not a role an operator assigns and
2697/// not a property of the hardware, it is a fact about how kamaji was started,
2698/// and it changes when the node is rolled. Nothing validates tag prefixes, so
2699/// this costs no wiring.
2700///
2701/// # Fails closed
2702///
2703/// A node that does not declare it is not a candidate. An undeclared fleet
2704/// therefore reports "no node admits" at election time rather than dispatching
2705/// to a node that will refuse — the refusal moves earlier and names the
2706/// constraint, which is the whole point. Declared today (from readings recorded
2707/// in-repo, not inferred) on `us-west-001` and `us-west-003`; see those
2708/// machines' TOMLs for the evidence and the date.
2709///
2710/// # What it does NOT cover — the reading that looks like an inversion
2711///
2712/// This tag gates exactly one backend: [`WorkloadSpec::wants_native_exec`] on a
2713/// `Workload::Container` spec, whose only two in-tree producers are
2714/// `yubaba::headscale_appliance::appliance_spec` and
2715/// `velveteen_exec::remote::mark_native_exec` (forge build steps).
2716///
2717/// kamaji has **other** ways to fork a process onto the host userland, and none
2718/// of them are `yah.exec = native`: a `Workload::MesofactStatic` carrying a
2719/// `serve_bundle` is served by `BundleBackend` (cargo feature `bundle-serving`
2720/// + `--bundle-cache-dir` + `--bundle-origin`), and `Workload::TenantPassway`
2721/// by `kamaji::jit::JitRuntime` (cargo feature `tenant-passway` +
2722/// `--tenant-passway-dir`). Those are separate node-local startup decisions.
2723/// The bundle one is now modelled — see [`BUNDLE_SERVING_MESH_TAG`], R885-T14.
2724/// The JIT one deliberately is **not**; that const's docs say why.
2725///
2726/// The practical consequence, because it has already misled one reader: a node
2727/// can run several kamaji-forked host processes and correctly carry no
2728/// `cap:native-exec`. On 2026-09-11 `us-east-001` ran four (two bundle servers,
2729/// a revalidate receiver, an almanac feed) with no such tag while `us-west-001`
2730/// carried the tag and ran none, which reads as an inversion and is not one:
2731/// none of those four is a native-exec workload, and the tag never claimed
2732/// them.
2733pub const NATIVE_EXEC_MESH_TAG: &str = "cap:native-exec";
2734
2735/// The mesh tag a node declares to advertise that its kamaji can **serve W272
2736/// bundles** — R885-T14, the same axis [`NATIVE_EXEC_MESH_TAG`] models for the
2737/// native-exec backend.
2738///
2739/// A `Workload::MesofactStatic` carrying a `serve_bundle` is materialized and
2740/// supervised by `kamaji_bin::BundleBackend`, which only exists when the node's
2741/// kamaji was **built** with the `bundle-serving` cargo feature and **started**
2742/// with `--bundle-cache-dir` *and* `--bundle-origin` (or `$KAMAJI_BUNDLE_ORIGIN`)
2743/// — `attach_bundle_backend` in `oss/kamaji/crates/kamaji-bin/src/main.rs`
2744/// returns the context untouched without the cache dir, logging
2745/// "Deploy { MesofactStatic + serve_bundle } will refuse with BackendRefused".
2746/// All three are node-local startup decisions, invisible to everything upstream.
2747///
2748/// # The gap this closes
2749///
2750/// Bundle placement does not go through [`admission_spec`] at all — a bundle is
2751/// placed by `reconciler::mesofact_bundle::resolve_bundle_machines`, off the
2752/// mirror's `providers.bundle` declaration (`machines = [...]` or `required =
2753/// { … }`), which the operator writes and which knows nothing about backends.
2754/// So before this tag, a mirror whose constraint matched a node without the
2755/// bundle backend deployed there and was refused at dispatch, exactly the way
2756/// R858 lost headscale for 25 hours on the native axis. It stayed invisible
2757/// because one node (`us-east-001`) serves every bundle in the fleet — the
2758/// condition under which an unmodelled constraint costs nothing right up until
2759/// the second node appears.
2760///
2761/// Both arms of `resolve_bundle_machines` enforce it, including the literal
2762/// `machines = [...]` pin: an operator naming a node by hand is making exactly
2763/// the claim this tag exists to check, and a pin is where the mistake is most
2764/// likely, not least.
2765///
2766/// # Fails closed, like the native tag
2767///
2768/// A node that does not declare it cannot serve a bundle. Declared today on
2769/// `us-east-001` only; see that machine's TOML for the evidence and the date.
2770///
2771/// # Why there is no `cap:tenant-passway` beside this
2772///
2773/// Checked rather than assumed, and it is a real asymmetry.
2774/// `Workload::TenantPassway` has **no placement decision to gate**:
2775/// `yubaba::tenant_passway::reconcile_once` runs *inside the node's own yubaba*
2776/// and drives *that node's own* kamaji over the local UDS. Which node arms a
2777/// domain is decided by which node was started with
2778/// `YUBABA_TENANT_PASSWAY_STATE_DIR`, not by any selector — so a capability tag
2779/// would have no consumer, and R852-B4 records that the tier is enabled on no
2780/// fleet machine today. Declaring it now would be the same wrong fact
2781/// [`NATIVE_EXEC_MESH_TAG`]'s notes refuse to state for `cap:microvm`. Model it
2782/// when a *selector* exists to read it.
2783///
2784/// `Workload::Almanac` needs no tag either, and the reason is stronger: kamaji
2785/// refuses it outright (`server.rs`, "almanac and static-asset live in yubaba's
2786/// reconcilers"), so it is never node-placed. An "almanac feed" observed
2787/// forked on `us-east-001` is a bundle-staged feed fetcher running under
2788/// `BundleBackend`, covered by *this* tag — not a `Workload::Almanac`.
2789pub const BUNDLE_SERVING_MESH_TAG: &str = "cap:bundle-serving";
2790
2791/// The mesh tag a node declares to advertise that its kamaji can **boot a
2792/// Firecracker microVM** — the third instance of the axis
2793/// [`NATIVE_EXEC_MESH_TAG`] and [`BUNDLE_SERVING_MESH_TAG`] model, and the one
2794/// R860-T5 deliberately left open.
2795///
2796/// # Why it was deferred, and why it is no longer
2797///
2798/// R860-T5's `@yah:cleanup` says this tag is "one line from done in the same
2799/// `if` in [`admission_spec`]" and refuses to take it, because **no node in the
2800/// fleet could host a microVM**: the guest kernel and rootfs were gated on
2801/// R605-F14, and declaring a capability nothing has is asserting a false fact.
2802/// That precondition is met. `us-west-003` has had the backend attached since
2803/// 2026-09-10T23:27Z (its TOML records the drop-in, the journal line and the
2804/// staged guest material), and on 2026-09-11 R605-T24's
2805/// `.yah/qed/microvm-dispatch-smoke.toml` drove a forge through the entire
2806/// chain — qed → velveteen-exec → yubaba admission → kamaji → `MicroVmRuntime`
2807/// — asserting on `/proc/cmdline` tokens a container cannot produce.
2808///
2809/// R894-F1 is what made taking it *necessary* rather than merely available: an
2810/// untrusted workload now has [`workload_spec::ExecSubstrate::MicroVm`] as a hard floor, so
2811/// without this axis every untrusted workload would be admitted onto whichever
2812/// node won the tie-break and refused at dispatch — the exact 25-hour-outage
2813/// shape R858 paid for on the native axis.
2814///
2815/// # Fails closed, and the node-side fact is in a drop-in, not ExecStart
2816///
2817/// A node that does not declare it cannot host a microVM workload. Declared
2818/// today on `us-west-003` only.
2819///
2820/// Reading `ExecStart` **cannot** answer whether a node has this backend, and
2821/// that trap is written into `us-west-003`'s own TOML: the backend is enabled by
2822/// `Environment=KAMAJI_MICROVM_DIR=…` in
2823/// `/etc/systemd/system/kamaji.service.d/10-microvm.conf`, so the unit's
2824/// `ExecStart` carries no `--microvm-dir` while `MicroVmRuntime` constructs
2825/// anyway. The same TOML records that rolling the node to a published version
2826/// reverts the staged guest material — so this tag, like the other two, must be
2827/// re-checked after any roll.
2828pub const MICROVM_MESH_TAG: &str = "cap:microvm";
2829
2830/// Parse the R594 mesh-tag node-selector off a workload's annotations into the
2831/// requested tag set. Absent annotation or empty value ⇒ empty vec ("no
2832/// constraint"). Whitespace around each comma-separated tag is trimmed and
2833/// empty segments are dropped, so `"tag:build-worker, arch:x86"` and
2834/// `"tag:build-worker,arch:x86"` parse identically.
2835pub fn node_selector_mesh_tags(ws: &WorkloadSpec) -> Vec<String> {
2836 ws.annotations
2837 .get(velveteen_exec::remote::NODE_SELECTOR_MESH_TAGS_ANNOTATION)
2838 .map(|v| {
2839 v.split(',')
2840 .map(str::trim)
2841 .filter(|s| !s.is_empty())
2842 .map(String::from)
2843 .collect()
2844 })
2845 .unwrap_or_default()
2846}
2847
2848/// Parse the R833-F8 imperative node-selector off a workload's annotations —
2849/// the single machine `name` the operator pinned the run to
2850/// (`--where=node:us-west-003`). Absent or blank ⇒ `None` ("no constraint"),
2851/// which is every workload built before this axis existed.
2852///
2853/// One node, not a list: the annotation exists to express "run it *there*", and
2854/// a comma-joined set would be a worse spelling of the mesh-tag selector that
2855/// already handles "any of these".
2856pub fn node_selector_node(ws: &WorkloadSpec) -> Option<String> {
2857 ws.annotations
2858 .get(velveteen_exec::remote::NODE_SELECTOR_NODE_ANNOTATION)
2859 .map(|v| v.trim())
2860 .filter(|v| !v.is_empty())
2861 .map(String::from)
2862}
2863
2864/// Load every `.yah/infra/providers/*.toml` into a [`ProviderConfig`] list.
2865/// Missing directory → empty list.
2866fn load_providers(dir: &Path) -> Result<Vec<ProviderConfig>> {
2867 if !dir.exists() {
2868 return Ok(vec![]);
2869 }
2870 let mut items = vec![];
2871 let mut entries: Vec<_> = std::fs::read_dir(dir)
2872 .with_context(|| format!("reading {}", dir.display()))?
2873 .filter_map(|e| e.ok())
2874 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
2875 .collect();
2876 entries.sort_by_key(|e| e.file_name());
2877 for entry in entries {
2878 items.push(ProviderConfig::load(&entry.path())?);
2879 }
2880 Ok(items)
2881}
2882
2883/// Map legacy mirror file stems to their canonical tier names.
2884///
2885/// Canonical tiers: `dev` / `pond` / `prod` / `ha`.
2886/// Legacy stems pre-R362: `local` (dev tier), `local-sim` / `sim` (pond tier).
2887/// Legacy stem `cloud` (prod tier) — renamed 2026-09-14: "cloud" named the
2888/// deployment mechanism, not the environment, and every operator-facing
2889/// surface already said "prod" (`yah cloud apply --env prod`, `mirror up ...
2890/// for prod`, the mirror files themselves are `prod.toml`) while only this
2891/// loader's internal key disagreed.
2892/// Both forms are accepted; canonical names are preferred for new files.
2893pub fn canonical_tier(stem: &str) -> &str {
2894 match stem {
2895 "local" => "dev",
2896 "local-sim" | "sim" => "pond",
2897 "cloud" => "prod",
2898 other => other,
2899 }
2900}
2901
2902/// Walk `.yah/services/<svc>/` for every service and its mirrors.
2903/// Missing directory → empty map. Mirror file stems are normalized to canonical
2904/// tier names via [`canonical_tier`] so callers always see `dev/pond/prod/ha`.
2905fn load_services(
2906 dir: &Path,
2907 workspace_root: &Path,
2908) -> Result<BTreeMap<String, ServiceWithMirrors>> {
2909 if !dir.exists() {
2910 return Ok(BTreeMap::new());
2911 }
2912 let mut out = BTreeMap::new();
2913 let mut entries: Vec<_> = std::fs::read_dir(dir)
2914 .with_context(|| format!("reading {}", dir.display()))?
2915 .filter_map(|e| e.ok())
2916 .filter(|e| e.path().is_dir())
2917 .collect();
2918 entries.sort_by_key(|e| e.file_name());
2919
2920 for entry in entries {
2921 let svc_dir = entry.path();
2922 let service_toml = svc_dir.join("service.toml");
2923 if !service_toml.exists() {
2924 // Skip directories without a service.toml — leaves room for
2925 // future siblings (e.g. `secrets/`, `README.md`) without
2926 // triggering false-positive parse errors.
2927 continue;
2928 }
2929 let service = ServiceConfig::load(&service_toml)?;
2930 let mut mirrors = BTreeMap::new();
2931 let mirrors_dir = svc_dir.join("mirrors");
2932 if mirrors_dir.exists() {
2933 let mut menv: Vec<_> = std::fs::read_dir(&mirrors_dir)
2934 .with_context(|| format!("reading {}", mirrors_dir.display()))?
2935 .filter_map(|e| e.ok())
2936 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
2937 .collect();
2938 menv.sort_by_key(|e| e.file_name());
2939 for m in menv {
2940 let path = m.path();
2941 let stem = path
2942 .file_stem()
2943 .and_then(|s| s.to_str())
2944 .unwrap_or("")
2945 .to_string();
2946 let tier = canonical_tier(&stem).to_string();
2947 // Last-write wins if both legacy and canonical forms coexist
2948 // (e.g. local-sim.toml + pond.toml). Sort order ensures the
2949 // canonical file (pond.toml) wins because 'p' > 'l'.
2950 mirrors.insert(tier, MirrorConfig::load(&path)?);
2951 }
2952 }
2953 let mut component_transform_recipes = BTreeMap::new();
2954 for component in &service.components {
2955 if component.kind == "static-asset" {
2956 if let Some(recipe) =
2957 read_component_transform_recipe(workspace_root, &component.path)
2958 {
2959 component_transform_recipes.insert(component.id.clone(), recipe);
2960 }
2961 }
2962 }
2963 let passway_machines = mirrors
2964 .iter()
2965 .filter_map(|(env, m)| m.passway_machines().map(|ms| (env.clone(), ms)))
2966 .collect();
2967 out.insert(
2968 service.name.clone(),
2969 ServiceWithMirrors {
2970 service,
2971 mirrors,
2972 component_transform_recipes,
2973 passway_machines,
2974 },
2975 );
2976 }
2977 Ok(out)
2978}
2979
2980/// Read the first transform recipe name from a component's `workload.toml`.
2981/// Returns `None` when the file is absent or has no `[asset.derive.transform]`
2982/// section. Best-effort — parse failures are silently ignored so a malformed
2983/// workload.toml doesn't abort the entire service catalog load.
2984fn read_component_transform_recipe(workspace_root: &Path, component_path: &str) -> Option<String> {
2985 let workload_path = workspace_root.join(component_path).join("workload.toml");
2986 let text = std::fs::read_to_string(&workload_path).ok()?;
2987 let value: toml::Value = toml::from_str(&text).ok()?;
2988 let assets = value.get("asset")?.as_array()?;
2989 for asset in assets {
2990 if let Some(recipe) = asset
2991 .get("derive")
2992 .and_then(|d| d.get("transform"))
2993 .and_then(|t| t.get("recipe"))
2994 .and_then(|r| r.as_str())
2995 {
2996 return Some(recipe.to_string());
2997 }
2998 }
2999 None
3000}
3001
3002/// Load every `.yah/domains/*.toml` into a [`DomainConfig`] map keyed by
3003/// file stem. Missing directory → empty map.
3004pub(crate) fn load_domains(dir: &Path) -> Result<BTreeMap<String, DomainConfig>> {
3005 if !dir.exists() {
3006 return Ok(BTreeMap::new());
3007 }
3008 let mut out = BTreeMap::new();
3009 let mut entries: Vec<_> = std::fs::read_dir(dir)
3010 .with_context(|| format!("reading {}", dir.display()))?
3011 .filter_map(|e| e.ok())
3012 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
3013 .collect();
3014 entries.sort_by_key(|e| e.file_name());
3015 for entry in entries {
3016 let path = entry.path();
3017 let stem = path
3018 .file_stem()
3019 .and_then(|s| s.to_str())
3020 .unwrap_or("")
3021 .to_string();
3022 let dom = DomainConfig::load(&path)?;
3023 if dom.name != stem {
3024 anyhow::bail!(
3025 "domains/{}.toml: name = \"{}\" must match the file stem",
3026 stem,
3027 dom.name
3028 );
3029 }
3030 out.insert(dom.name.clone(), dom);
3031 }
3032 Ok(out)
3033}
3034
3035/// Load and shape-validate all `*.toml` files in `dir` as [`WorkloadConfig`].
3036fn load_workloads(dir: std::path::PathBuf) -> Result<Vec<WorkloadConfig>> {
3037 if !dir.exists() {
3038 return Ok(vec![]);
3039 }
3040 let mut items = vec![];
3041 let mut entries: Vec<_> = std::fs::read_dir(&dir)
3042 .with_context(|| format!("reading {}", dir.display()))?
3043 .filter_map(|e| e.ok())
3044 .filter(|e| e.path().extension().map_or(false, |x| x == "toml"))
3045 .collect();
3046 entries.sort_by_key(|e| e.file_name());
3047
3048 for entry in entries {
3049 let path = entry.path();
3050 let path_str = path.display().to_string();
3051 let src =
3052 std::fs::read_to_string(&path).with_context(|| format!("reading {}", path_str))?;
3053 let spec: WorkloadSpec =
3054 toml::from_str(&src).with_context(|| format!("parsing {}", path_str))?;
3055
3056 // R892-B1: refuse a file whose keys the parser silently threw away.
3057 refuse_dropped_keys(&src, &spec, &path_str)?;
3058
3059 // Shape-validate before accepting into the loaded config.
3060 validate::shape(&spec)
3061 .map_err(|e| anyhow::anyhow!("workload {} failed shape validation: {e}", path_str))?;
3062
3063 items.push(WorkloadConfig { spec });
3064 }
3065 Ok(items)
3066}
3067
3068/// Refuse a workload file that declares keys the parser did not keep (R892-B1).
3069///
3070/// `WorkloadSpec` deliberately does **not** carry `deny_unknown_fields`, and
3071/// must not: the JSON leg of the deploy wire relies on an un-rolled node
3072/// ignoring a field it has never heard of, which is what lets a fleet cross a
3073/// schema change one node at a time. That forgiveness is right on the wire and
3074/// wrong in a hand-authored file — there, an ignored key is an operator's
3075/// declared intent evaporating between parse and serialise, with no diagnostic.
3076///
3077/// On 2026-09-11 that cost a production outage: `[resources]
3078/// ephemeral_storage_mb = 256` in noisetable's `noisetable-account.toml` had
3079/// been deleted from `ResourceLimits` by R885-T6, so the CLI read the file, drop
3080/// the value, and sent a spec the (older) node could not parse — after it had
3081/// destroyed the incumbent. The file said the right thing the whole time.
3082///
3083/// The check is a round trip rather than a key whitelist, so it needs no list to
3084/// maintain and catches every renamed, removed or misspelled key at once: parse
3085/// the file, re-serialise the parsed spec, and report any key the source
3086/// declared that the re-serialisation does not carry. Value *representation* may
3087/// legitimately change across that trip (an enum canonicalising its spelling),
3088/// so only missing KEYS are reported, never differing values.
3089fn refuse_dropped_keys(src: &str, spec: &WorkloadSpec, path_str: &str) -> Result<()> {
3090 let declared: toml::Value = match toml::from_str(src) {
3091 Ok(v) => v,
3092 // Unreachable: the caller just parsed this same text into a typed spec.
3093 Err(_) => return Ok(()),
3094 };
3095 let kept = match toml::Value::try_from(spec) {
3096 Ok(v) => v,
3097 // A spec that cannot be re-serialised is a bug in the schema, not in the
3098 // operator's file — say so rather than blaming their config, and let the
3099 // load proceed exactly as it did before this check existed.
3100 Err(e) => {
3101 eprintln!(
3102 "warning: {path_str} could not be checked for silently-dropped keys \
3103 (re-serialising the parsed spec failed: {e})"
3104 );
3105 return Ok(());
3106 }
3107 };
3108
3109 let mut dropped = Vec::new();
3110 collect_dropped_keys(&declared, &kept, "", &mut dropped);
3111
3112 // R896-B5: a key whose field was deleted as inert is ignored, not refused —
3113 // otherwise every field deletion forces a same-day sweep of every camp's
3114 // committed TOML. Say so, so the dead key still gets cleaned up eventually.
3115 dropped.retain(|path| {
3116 let Some(retired) = workload_spec::RETIRED_KEYS.iter().find(|r| r.path == path) else {
3117 return true;
3118 };
3119 eprintln!(
3120 "warning: {path_str} declares `{path}`, retired by {} and ignored; delete it",
3121 retired.retired_by
3122 );
3123 false
3124 });
3125 if dropped.is_empty() {
3126 return Ok(());
3127 }
3128
3129 anyhow::bail!(
3130 "{path_str} declares {} the workload schema does not have, and their values were being \
3131 discarded in silence:\n\
3132 \x20 {}\n\
3133 \n\
3134 Delete them, or correct the spelling. A key that is present in the file and absent \
3135 from the deployed spec is exactly the failure that destroyed a live workload on \
3136 2026-09-11 (R892): the declaration reads as honoured and is not.",
3137 if dropped.len() == 1 { "a key" } else { "keys" },
3138 dropped.join("\n ")
3139 );
3140}
3141
3142/// Recursive half of [`refuse_dropped_keys`] — keys in `declared` with no
3143/// counterpart in `kept`, reported as dotted paths.
3144fn collect_dropped_keys(
3145 declared: &toml::Value,
3146 kept: &toml::Value,
3147 prefix: &str,
3148 out: &mut Vec<String>,
3149) {
3150 match (declared, kept) {
3151 (toml::Value::Table(d), toml::Value::Table(k)) => {
3152 for (key, value) in d {
3153 match k.get(key) {
3154 Some(kept_value) => {
3155 collect_dropped_keys(value, kept_value, &format!("{prefix}{key}."), out)
3156 }
3157 None => out.push(format!("{prefix}{key}")),
3158 }
3159 }
3160 }
3161 // Element-wise only when nothing was added or removed. A length change
3162 // means the serialiser reshaped the list (materialisation appends mounts,
3163 // for one), and pairing across that would report nonsense.
3164 (toml::Value::Array(d), toml::Value::Array(k)) if d.len() == k.len() => {
3165 for (i, (dv, kv)) in d.iter().zip(k).enumerate() {
3166 collect_dropped_keys(dv, kv, &format!("{prefix}{i}."), out);
3167 }
3168 }
3169 _ => {}
3170 }
3171}
3172
3173/// Load all mirror configs from the `mirrors/` directory.
3174///
3175/// Handles two layouts that may coexist:
3176/// - **Folder**: `mirrors/<id>/mirror.toml` — preferred; allows secrets and
3177/// per-mirror overrides to live next to the config file.
3178/// - **Flat**: `mirrors/<id>.toml` — legacy; still supported.
3179///
3180/// Each file is parsed as [`LegacyMirrorConfig`]. A malformed file returns an error
3181/// that includes the file path and the TOML field path + line/column, so the
3182/// caller can surface it to the user directly.
3183fn load_mirrors(dir: std::path::PathBuf) -> Result<Vec<LegacyMirrorConfig>> {
3184 if !dir.exists() {
3185 return Ok(vec![]);
3186 }
3187 let mut mirrors = vec![];
3188 let mut entries: Vec<_> = std::fs::read_dir(&dir)
3189 .with_context(|| format!("reading {}", dir.display()))?
3190 .filter_map(|e| e.ok())
3191 .collect();
3192 entries.sort_by_key(|e| e.file_name());
3193
3194 for entry in entries {
3195 let path = entry.path();
3196 if path.is_dir() {
3197 // Folder layout: mirrors/<id>/mirror.toml
3198 let mirror_toml = path.join("mirror.toml");
3199 if mirror_toml.exists() {
3200 let src = std::fs::read_to_string(&mirror_toml)
3201 .with_context(|| format!("reading {}", mirror_toml.display()))?;
3202 let cfg: LegacyMirrorConfig = toml::from_str(&src)
3203 .with_context(|| format!("parsing {}", mirror_toml.display()))?;
3204 mirrors.push(cfg);
3205 }
3206 } else if path.extension().map_or(false, |e| e == "toml") {
3207 // Flat layout: mirrors/<id>.toml
3208 let src = std::fs::read_to_string(&path)
3209 .with_context(|| format!("reading {}", path.display()))?;
3210 let cfg: LegacyMirrorConfig =
3211 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
3212 mirrors.push(cfg);
3213 }
3214 }
3215 Ok(mirrors)
3216}
3217
3218/// Load `topology.toml` if it exists; return a default (empty) topology otherwise.
3219fn load_topology(path: std::path::PathBuf) -> Result<TopologyConfig> {
3220 if !path.exists() {
3221 return Ok(TopologyConfig::default());
3222 }
3223 let src =
3224 std::fs::read_to_string(&path).with_context(|| format!("reading {}", path.display()))?;
3225 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
3226}
3227
3228/// R555-S1: entries are sorted by file name before parsing, so "declaration
3229/// order in `.yah/infra/machines/` breaks ties" — the contract
3230/// [`CloudConfig::admit_workload`] documents — is actually true. `read_dir`
3231/// yields filesystem order, which is unspecified and differs between APFS and
3232/// a hashed-dir ext4; without the sort, *which* of two equally-matching nodes a
3233/// workload admits to could change when an unrelated file is added to the
3234/// directory. That was latent while each tag set had one match and became
3235/// observable the day us-west-003 joined us-west-002 on
3236/// `[tag:build-worker, arch:x86, os:linux]`. Same sort `load_providers` has
3237/// always done.
3238fn load_dir<T: for<'de> Deserialize<'de>>(dir: std::path::PathBuf) -> Result<Vec<T>> {
3239 if !dir.exists() {
3240 return Ok(vec![]);
3241 }
3242 let mut entries: Vec<_> = std::fs::read_dir(&dir)
3243 .with_context(|| format!("reading {}", dir.display()))?
3244 .collect::<std::io::Result<Vec<_>>>()
3245 .with_context(|| format!("reading {}", dir.display()))?;
3246 entries.sort_by_key(|e| e.file_name());
3247
3248 let mut items = vec![];
3249 for entry in entries {
3250 let path = entry.path();
3251 if path.extension().map_or(false, |e| e == "toml") {
3252 let src = std::fs::read_to_string(&path)
3253 .with_context(|| format!("reading {}", path.display()))?;
3254 let item: T =
3255 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
3256 items.push(item);
3257 }
3258 }
3259 Ok(items)
3260}
3261
3262// ─── New manifest shapes (R222 B2) ───────────────────────────────────────────
3263//
3264// The post-R215 layout splits substrate from service declarations:
3265//
3266// .yah/infra/providers/<id>.toml → ProviderConfig
3267// .yah/services/<svc>/service.toml → ServiceConfig
3268// .yah/services/<svc>/mirrors/<env>.toml → MirrorConfig
3269//
3270// CloudConfig::load still reads the legacy layout — B3 swaps in these types
3271// and removes the Legacy* shapes plus TopologyConfig.
3272
3273/// Tag for the infrastructure provider kind. Drives which fields are valid in
3274/// a [`ProviderConfig`] body or a [`MirrorProviderSlot::Inline`] block.
3275///
3276/// Two flavors:
3277/// - **Account/runtime providers** (`cloudflare`, `hetzner`, `local-container`)
3278/// live as files under `.yah/infra/providers/<id>.toml` and are referenced
3279/// from a mirror via `use = "<id>"`.
3280/// - **Inline-only providers** (`miniflare-native`, `miniflare-container`,
3281/// `minio-container`) declare an operator-local stand-in directly inside a
3282/// mirror via `kind = "..."`. They carry no credentials and have no provider
3283/// file. The container-backed kinds ride on top of whichever
3284/// `local-container` runtime is declared in infra (orbstack/colima/docker);
3285/// the reconciler resolves the runtime at up-time.
3286#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
3287#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3288#[serde(rename_all = "kebab-case")]
3289pub enum Provider {
3290 /// Cloudflare account: R2 buckets, DNS, Workers, Tunnels.
3291 Cloudflare,
3292 /// Hetzner Cloud + Object Storage account.
3293 Hetzner,
3294 /// Vultr cloud VPS — auto-provisioned via the `cloud.vps.*` Envoy
3295 /// (`VultrEnvoy`), the burst/scaling counterpart to Hetzner. Driver-backed.
3296 Vultr,
3297 /// BYO bare/static node (OVH, on-prem, anything we did NOT provision via a
3298 /// cloud API). Brought up over SSH (`stand-up-yubaba.sh` / `yah cloud
3299 /// machine bootstrap`); reach is declared in the machine's `[connect]`
3300 /// block. No create/destroy driver — placement-only.
3301 Static,
3302 /// Dev-tier static surface: miniflare (workerd) running **natively** — no
3303 /// container — in front of whatever the mirror binds to the `s3`
3304 /// capability (W265, R584-F4).
3305 ///
3306 /// The door at every tier is the same compiled Worker bundle
3307 /// (`worker/router.bundle.js`); only the object store underneath differs,
3308 /// and that difference is now declared rather than branched on:
3309 ///
3310 /// ```toml
3311 /// [providers.static]
3312 /// kind = "miniflare-native"
3313 /// port = 4321
3314 ///
3315 /// [drivers.s3]
3316 /// kind = "local-s3-fs"
3317 /// ```
3318 ///
3319 /// This replaces `local-static`, which served a workload's `dist/` off
3320 /// disk and so gave the dev tier a storage interface no other tier had —
3321 /// the fork W265 exists to delete. Inline-only; it carries no credentials
3322 /// (the store is loopback, the Worker runs on this machine).
3323 MiniflareNative,
3324 /// Local container runtime (orbstack/colima/docker). Configured by a
3325 /// provider file under `.yah/infra/providers/` so the discovery hints +
3326 /// runtime override sit in one place.
3327 LocalContainer,
3328 /// Dev-tier compute: the component runs as a kamaji-supervised host
3329 /// process against the operator's real workspace, no container and no
3330 /// build step per edit. Inline-only — it carries no credentials, and
3331 /// "the machine you are sitting at" is not an account to point at.
3332 /// See `reconciler::local_process`.
3333 LocalProcess,
3334 /// Dev-tier compute on an attached iOS device or simulator, or Android
3335 /// device or emulator (R941). The slot's `target` names which
3336 /// (`ios-simulator | ios-device | android-emulator | android-device`) and
3337 /// optional `device` pins one by id or name; the component's
3338 /// `workload.toml` `[device]` section names the app. Launched and
3339 /// supervised through the host's own `simctl` / `devicectl` / `adb`.
3340 /// Inline-only, like `local-process`. See `reconciler::device`.
3341 Device,
3342 /// Containerized miniflare (workerd subprocess) fronting MinIO — the
3343 /// pond-tier stand-in for a CF Worker + R2 static surface. Inline-only;
3344 /// the reconciler spawns miniflare via the JS runtime and starts a MinIO
3345 /// container on the local-container runtime.
3346 MiniflareContainer,
3347 /// Containerized MinIO providing an S3-compatible API — the pond-tier
3348 /// stand-in for Cloudflare R2. Inline-only; the reconciler spins up the
3349 /// container on the local-container runtime and auto-creates the declared
3350 /// bucket on first up.
3351 MinioContainer,
3352 /// Dev-tier PostgreSQL — a real server speaking real pgwire on loopback,
3353 /// supervised by kamaji as the `yah-pg-dev` workload (W265, R584-F1). No
3354 /// docker daemon: the driver fetches a per-arch PostgreSQL tarball on first
3355 /// run and `initdb`s a cluster under `.yah/infra/state/dev/pg/`.
3356 ///
3357 /// Inline-only — it carries no credentials worth a provider file (the
3358 /// cluster is loopback-bound with a fixed dev password). Declared under
3359 /// [`MirrorConfig::drivers`], not `providers`:
3360 ///
3361 /// ```toml
3362 /// [drivers.pg]
3363 /// kind = "local-pg-dev"
3364 /// ```
3365 LocalPgDev,
3366 /// Dev/pond-tier SMTP — [mailcrab] supervised by kamaji as the
3367 /// `yah-smtp-dev` workload (W265, R584-F2). A real SMTP listener that
3368 /// accepts every message and delivers none of them, plus a web inbox to
3369 /// read what was sent. No docker daemon: the driver fetches the per-arch
3370 /// mailcrab release binary on first run and caches it under
3371 /// `.yah/cache/mailcrab/`.
3372 ///
3373 /// Inline-only — it carries no credentials at all (the listener is
3374 /// loopback-bound and unauthenticated, which is the point: a catcher that
3375 /// refused unauthenticated mail would not catch the mail your app sends).
3376 /// Declared under [`MirrorConfig::drivers`], not `providers`:
3377 ///
3378 /// ```toml
3379 /// [drivers.smtp]
3380 /// kind = "local-mailcrab"
3381 /// ```
3382 ///
3383 /// [mailcrab]: https://github.com/tweedegolf/mailcrab
3384 LocalMailcrab,
3385 /// Dev-tier S3 — a filesystem-backed, path-style S3 surface supervised by
3386 /// kamaji as the `yah-s3-fs` workload (W265, R584-F3). No docker daemon
3387 /// and no download: the driver *is* the server, and objects live under
3388 /// `.yah/infra/state/dev/s3/data/<bucket>/`.
3389 ///
3390 /// Inline-only — the credentials are fixed dev strings on a loopback
3391 /// listener, which is not an account to point a provider file at.
3392 /// Declared under [`MirrorConfig::drivers`], not `providers`:
3393 ///
3394 /// ```toml
3395 /// [drivers.s3]
3396 /// kind = "local-s3-fs"
3397 /// ```
3398 ///
3399 /// Unlike its two siblings the binding is **optional**: the camp brings
3400 /// this driver up for any camp with a dev mirror whether or not a stanza
3401 /// says so, because every static-asset component needs object storage and
3402 /// [`crate::capability::Capability::for_component_kind`] already says as
3403 /// much. Declaring it is documentation, not activation — see
3404 /// `crate::reconciler::s3_driver::camp_needs_s3_driver`.
3405 LocalS3Fs,
3406}
3407
3408/// A provider account/runtime binding from `.yah/infra/providers/<id>.toml`.
3409///
3410/// The `kind` discriminator picks the schema for the remaining fields. Strict
3411/// on `kind` (unknown values are a parse error); permissive on per-kind fields
3412/// (carried as a free-form map so this loader stays stable as new fields land).
3413/// B3/B4 will tighten by introducing typed variants alongside JSON Schema.
3414#[derive(Debug, Clone, Serialize, Deserialize)]
3415#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3416pub struct ProviderConfig {
3417 pub schema_version: u32,
3418 pub id: String,
3419 pub kind: Provider,
3420 /// Reference into the OS keystore for live credentials (e.g.
3421 /// `"keystore://cloudflare/yah"`). `None` for providers that don't need
3422 /// creds (miniflare-native, optionally local-container).
3423 #[serde(default, skip_serializing_if = "Option::is_none")]
3424 pub credentials: Option<String>,
3425 /// Kind-specific fields. Examples:
3426 /// - cloudflare: `default_zone`
3427 /// - hetzner: `default_location`, `default_server_type`, `ssh_keys`
3428 /// - local-container: `runtime`, `discovery`
3429 #[serde(flatten)]
3430 #[cfg_attr(
3431 feature = "json-schema",
3432 schemars(with = "std::collections::BTreeMap<String, serde_json::Value>")
3433 )]
3434 pub fields: BTreeMap<String, toml::Value>,
3435}
3436
3437impl ProviderConfig {
3438 /// Parse a single `providers/<id>.toml` file.
3439 pub fn load(path: &Path) -> Result<Self> {
3440 let src =
3441 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
3442 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
3443 }
3444}
3445
3446/// Where a service answers, and therefore what a liveness probe asks
3447/// (R926-F1).
3448///
3449/// # Why this replaced a bare `domain: String`
3450///
3451/// `domain` used to be required, and two of this camp's services have no
3452/// honest value for it. `yah-cloud` is publish-only and carries
3453/// `unset.yah-cloud.invalid` — RFC 2606's reserved TLD, chosen precisely
3454/// because the field demanded a string and there was none (the R546-S2
3455/// decision, recorded in that file's own header). The push relay is worse
3456/// than awkward: it has no HTTP surface at all. It binds an iroh endpoint,
3457/// serves ALPN `yah/push-relay/1`, and is addressed by hex `NodeId` over
3458/// QUIC — so a `.invalid` domain there would not merely look wrong, it
3459/// would leave the service permanently unprobeable, which is the exact
3460/// hole R926 exists to close.
3461///
3462/// The alternative was a second health mechanism beside `health_path`.
3463/// This repo is below v1.0.0 and its standing rule is to change the one
3464/// mechanism rather than grow a parallel one, so addressing became a sum
3465/// type: a service declares exactly one way to be reached, and the prober
3466/// switches on it. A service cannot accidentally declare both, and the
3467/// enum is what makes that unrepresentable rather than merely validated.
3468///
3469/// # TOML
3470///
3471/// ```toml
3472/// [address]
3473/// kind = "front-door"
3474/// domain = "cloud.mesh.yah.dev"
3475/// health_path = "/key?v=138"
3476/// ```
3477///
3478/// ```toml
3479/// [address]
3480/// kind = "node"
3481/// node_id = "8f2c…" # 64 hex chars
3482/// alpn = "yah/push-relay/1"
3483/// ```
3484///
3485/// Tagged rather than untagged: an operator edits this file by hand, and
3486/// serde's untagged errors name none of the variants it tried.
3487#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3488#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3489#[serde(tag = "kind", rename_all = "kebab-case")]
3490pub enum ServiceAddress {
3491 /// An HTTPS front door on a public or mesh domain. What all ten
3492 /// services registered in this camp today are.
3493 FrontDoor {
3494 domain: String,
3495 /// Path the probe requests, relative to `domain`. `None` means
3496 /// `/`.
3497 ///
3498 /// `/` is the right question for a service whose root serves a
3499 /// site, and the wrong one for an API. An account/RPC origin that
3500 /// versions its surface answers only under its prefix and 404s
3501 /// everything else *on purpose* — so probing the root reported a
3502 /// broken cell for a service that was behaving exactly as
3503 /// designed, which is the failure mode the probe exists to remove
3504 /// rather than add to. Naming the path here makes the probe ask a
3505 /// question the service has agreed to answer.
3506 ///
3507 /// Service-scoped rather than per-component or per-mirror: one
3508 /// domain has one front door and therefore one canonical liveness
3509 /// URL, and that URL is a property of the service's own router —
3510 /// it does not vary by environment, so repeating it per mirror
3511 /// would only let the copies drift.
3512 ///
3513 /// Must be absolute (leading `/`); the loader refuses anything
3514 /// else rather than silently joining it onto the origin.
3515 #[serde(default, skip_serializing_if = "Option::is_none")]
3516 health_path: Option<String>,
3517 },
3518 /// An iroh endpoint: a stable `NodeId` speaking one ALPN. Probed by
3519 /// dialling that ALPN and expecting an answer.
3520 ///
3521 /// There is no path and no port because there is no URL — QUIC
3522 /// multiplexes on the ALPN, and the `NodeId` is the address. A
3523 /// service reached this way is unreachable from a browser tab, so the
3524 /// Services grid links nothing and renders the node id instead.
3525 Node {
3526 /// Hex-encoded `NodeId`, 64 characters. Not parsed here — this
3527 /// crate does not depend on mshr — but the length and alphabet
3528 /// are checked at load, because a truncated paste otherwise fails
3529 /// at first dial with an error naming neither the file nor the
3530 /// field.
3531 node_id: String,
3532 /// The ALPN to negotiate, e.g. `yah/push-relay/1`. Declared
3533 /// rather than inferred from the service name: the protocol is
3534 /// the thing being probed and it is versioned independently of
3535 /// whatever the service is called.
3536 alpn: String,
3537 },
3538}
3539
3540impl ServiceAddress {
3541 /// Shorthand for the common case.
3542 pub fn front_door(domain: impl Into<String>) -> Self {
3543 Self::FrontDoor {
3544 domain: domain.into(),
3545 health_path: None,
3546 }
3547 }
3548
3549 /// A front door with a declared probe path.
3550 pub fn front_door_at(domain: impl Into<String>, health_path: impl Into<String>) -> Self {
3551 Self::FrontDoor {
3552 domain: domain.into(),
3553 health_path: Some(health_path.into()),
3554 }
3555 }
3556
3557 pub fn node(node_id: impl Into<String>, alpn: impl Into<String>) -> Self {
3558 Self::Node {
3559 node_id: node_id.into(),
3560 alpn: alpn.into(),
3561 }
3562 }
3563
3564 /// The domain, for the callers that genuinely need one (DNS, zone
3565 /// selection, a public URL). `None` for a node-addressed service —
3566 /// and those callers must say so rather than substitute a
3567 /// placeholder, which is how `unset.yah-cloud.invalid` happened.
3568 pub fn domain(&self) -> Option<&str> {
3569 match self {
3570 Self::FrontDoor { domain, .. } => Some(domain),
3571 Self::Node { .. } => None,
3572 }
3573 }
3574
3575 pub fn health_path(&self) -> Option<&str> {
3576 match self {
3577 Self::FrontDoor { health_path, .. } => health_path.as_deref(),
3578 Self::Node { .. } => None,
3579 }
3580 }
3581
3582 /// One line for a status table or a log: the domain, or `node:<8 hex
3583 /// prefix>/<alpn>`. Never a placeholder.
3584 pub fn label(&self) -> String {
3585 match self {
3586 Self::FrontDoor { domain, .. } => domain.clone(),
3587 Self::Node { node_id, alpn } => {
3588 let short: String = node_id.chars().take(8).collect();
3589 format!("node:{short}/{alpn}")
3590 }
3591 }
3592 }
3593
3594 /// Reject what would otherwise fail at probe time with an error
3595 /// naming neither the file nor the field.
3596 fn validate(&self, svc_name: &str) -> Result<()> {
3597 match self {
3598 Self::FrontDoor { domain, health_path } => {
3599 if domain.trim().is_empty() {
3600 anyhow::bail!(
3601 "services/{svc_name}/service.toml: address.domain must not be empty"
3602 );
3603 }
3604 // A relative health path would be joined onto the origin as
3605 // if it were absolute by one URL builder and dropped by the
3606 // next, so the probe would silently ask a different question
3607 // than the file reads. Refuse it at load instead.
3608 if let Some(p) = health_path {
3609 if !p.starts_with('/') {
3610 anyhow::bail!(
3611 "services/{svc_name}/service.toml: health_path = \"{p}\" \
3612 must be absolute — write \"/{p}\""
3613 );
3614 }
3615 }
3616 }
3617 Self::Node { node_id, alpn } => {
3618 let id = node_id.trim();
3619 if id.len() != 64 || !id.chars().all(|c| c.is_ascii_hexdigit()) {
3620 anyhow::bail!(
3621 "services/{svc_name}/service.toml: address.node_id = \"{node_id}\" \
3622 is not a NodeId — want 64 hex characters, got {}",
3623 id.len()
3624 );
3625 }
3626 if alpn.trim().is_empty() {
3627 anyhow::bail!(
3628 "services/{svc_name}/service.toml: address.alpn must not be empty — \
3629 a NodeId with no ALPN names a process, not a service"
3630 );
3631 }
3632 }
3633 }
3634 Ok(())
3635 }
3636}
3637
3638/// An operator-facing service declaration from
3639/// `.yah/services/<svc>/service.toml`.
3640///
3641/// A service groups one or more components (a static surface, a containerized
3642/// API, an almanac…) under a single [`address`](ServiceConfig::address).
3643/// Mirrors project the service onto concrete infra; see [`MirrorConfig`].
3644#[derive(Debug, Clone, Serialize, Deserialize)]
3645#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3646pub struct ServiceConfig {
3647 pub schema_version: u32,
3648 pub name: String,
3649 /// One-line operator-facing answer to "what is this and why does the
3650 /// camp run it?", rendered beside the service wherever it is listed.
3651 ///
3652 /// New surface in R926, and it exists because the registry had no
3653 /// place to say what a service IS. A name and a domain identify a
3654 /// service to someone who already knows the fleet; they tell a reader
3655 /// who does not nothing at all, so the knowledge lived in whichever
3656 /// working doc or board annotation last touched the thing. That is
3657 /// exactly the knowledge a health check makes actionable — a red cell
3658 /// is only useful next to what went red.
3659 ///
3660 /// Deliberately not a free-form `[meta]` table: one field with one
3661 /// meaning cannot accumulate a second half-owner, which is the
3662 /// failure the headscale appliance already demonstrated four times
3663 /// over (see this repo's CLAUDE.md on giving a thing ONE owner).
3664 ///
3665 /// Optional, because nine services predate it and a required field
3666 /// would make every one of them fail to load. `None` renders as no
3667 /// description rather than as an empty one.
3668 #[serde(default, skip_serializing_if = "Option::is_none")]
3669 pub description: Option<String>,
3670 /// How this service is reached, and therefore how its liveness is
3671 /// probed. See [`ServiceAddress`].
3672 pub address: ServiceAddress,
3673 #[serde(default, skip_serializing_if = "Vec::is_empty")]
3674 pub components: Vec<ServiceComponent>,
3675 /// Databases this service exposes, grouped by environment (W241). Every
3676 /// entry becomes a data-workbench / `sql_*` catalog id of the shape
3677 /// `<env>:<service>:<name>` (e.g. `pond:scrabcake:main`). Optional and
3678 /// default-empty — services without databases omit the `[db]` table
3679 /// entirely.
3680 #[serde(default, skip_serializing_if = "DbCatalog::is_empty")]
3681 pub db: DbCatalog,
3682}
3683
3684impl ServiceConfig {
3685 /// The service's domain, or `None` when it is node-addressed
3686 /// (R926-F1). Callers that cannot proceed without one must say which
3687 /// service and why — see [`ServiceAddress::domain`].
3688 pub fn domain(&self) -> Option<&str> {
3689 self.address.domain()
3690 }
3691
3692 /// The domain, or an error naming this service and what the caller
3693 /// wanted it for. The shape every `yah cloud apply` reader wants: a
3694 /// node-addressed service is not deployed by the reconcilers at all,
3695 /// so reaching one of them with a `Node` address is a config bug and
3696 /// deserves a sentence, not an `unwrap`.
3697 pub fn require_domain(&self, wanted_for: &str) -> Result<&str> {
3698 self.address.domain().ok_or_else(|| {
3699 anyhow::anyhow!(
3700 "service {} is node-addressed ({}) and has no domain, but {wanted_for} needs one",
3701 self.name,
3702 self.address.label()
3703 )
3704 })
3705 }
3706
3707 pub fn health_path(&self) -> Option<&str> {
3708 self.address.health_path()
3709 }
3710
3711 /// Parse a single `services/<svc>/service.toml` file.
3712 pub fn load(path: &Path) -> Result<Self> {
3713 let src =
3714 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
3715 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
3716 }
3717
3718 /// Persist to `.yah/services/<name>/service.toml`, creating the service
3719 /// directory if needed. Create-or-overwrite — the canonical replacement
3720 /// for the legacy `sites.json` write path. `workspace_root` is the camp
3721 /// dir (the parent of `.yah/`).
3722 pub fn save(&self, workspace_root: &Path) -> Result<()> {
3723 let dir = crate::paths::service_dir(workspace_root, &self.name);
3724 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
3725 let path = crate::paths::service_toml(workspace_root, &self.name);
3726 let s = toml::to_string_pretty(self)
3727 .with_context(|| format!("serializing service {}", self.name))?;
3728 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
3729 }
3730
3731 /// Remove `.yah/services/<name>/` and everything under it (service.toml
3732 /// plus its `mirrors/`). Returns `false` when the directory was already
3733 /// absent, so callers can distinguish "deleted" from "no-op".
3734 pub fn delete(workspace_root: &Path, name: &str) -> Result<bool> {
3735 let dir = crate::paths::service_dir(workspace_root, name);
3736 if !dir.exists() {
3737 return Ok(false);
3738 }
3739 std::fs::remove_dir_all(&dir).with_context(|| format!("removing {}", dir.display()))?;
3740 Ok(true)
3741 }
3742}
3743
3744/// A git source for a component (R561-F1, "BYO git").
3745///
3746/// When a [`ServiceComponent`] sets `git`, the component's code is NOT in this
3747/// workspace — it lives in an external repo that the reconciler shallow-clones
3748/// into a source cache before build (approach A: clone-at-reconcile, so config
3749/// load + validation stay offline). The component's `path` is then interpreted
3750/// relative to `<checkout>/<subdir>` instead of the workspace root.
3751#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3752#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3753pub struct GitSource {
3754 /// Clone URL (https or ssh) of the tenant repo.
3755 pub repo: String,
3756 /// Branch, tag, or commit SHA to check out. Defaults to `"main"`.
3757 #[serde(default = "default_git_ref")]
3758 pub r#ref: String,
3759 /// Optional sub-directory within the repo that the workspace is rooted at
3760 /// (e.g. a monorepo's `site/`). `path` is resolved relative to this.
3761 #[serde(default, skip_serializing_if = "Option::is_none")]
3762 pub subdir: Option<String>,
3763}
3764
3765fn default_git_ref() -> String {
3766 "main".to_string()
3767}
3768
3769/// How to reach an external infra root (R615-F1 / W274, "linked infra
3770/// sources"): a filesystem link to a sibling camp's live tree, or a git
3771/// checkout of an extracted infra repo.
3772#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3773#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3774#[serde(tag = "kind", rename_all = "kebab-case")]
3775pub enum InfraSourceKind {
3776 /// Filesystem link — reads the owner's live tree. The dev-loop shortcut,
3777 /// and the whole story until W274's "infra as its own repo" end-state.
3778 /// `path` is relative to *this* camp's root; infra is read from
3779 /// `<path>/.yah/infra/`.
3780 Path {
3781 path: String,
3782 },
3783 /// Git link — reused verbatim from [`GitSource`] (R561, "BYO git"),
3784 /// lifted here from "a component's code" to "a camp's infra registry."
3785 /// Loading stays offline (W274 §3): `yah infra sync` (R615-T3) is what
3786 /// clones/pulls this into `.yah/cache/infra/<owner>/`; `CloudConfig::load`
3787 /// only ever reads that cache, never the network.
3788 Git(GitSource),
3789}
3790
3791/// Write-gate for a linked [`InfraSource`] (R615-F1 / W274).
3792///
3793/// An enum, not a bool: the two states today are "borrower renders/plans but
3794/// cannot reconcile" and "this camp genuinely co-administers the shared
3795/// root," and a future read-write-with-approval tier is a third variant, not
3796/// a renamed boolean.
3797#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq, Eq)]
3798#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3799#[serde(rename_all = "kebab-case")]
3800pub enum SourceMode {
3801 /// Borrower can render and plan against the linked entries but cannot
3802 /// reconcile/mutate them — the owner remains the single manager. Default:
3803 /// a borrower is opt-in to write access, never opt-out of the safe state.
3804 #[default]
3805 ReadOnly,
3806 /// Escape hatch for a camp that genuinely co-administers a shared root.
3807 Manage,
3808}
3809
3810/// One `[[source]]` entry in `.yah/infra/sources.toml` (R615-F1 / W274) — an
3811/// external infra root this camp borrows machines/providers from.
3812#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3813#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3814pub struct InfraSource {
3815 /// Logical owner name, badged in the Infra tab (e.g. `"yah"`). Distinct
3816 /// from any camp/repo name the `kind` resolves through — this is what an
3817 /// operator sees on a borrowed row, not a path.
3818 pub owner: String,
3819 #[serde(flatten)]
3820 pub kind: InfraSourceKind,
3821 #[serde(default)]
3822 pub mode: SourceMode,
3823 /// Optional filter — name globs or mesh-tag selectors — to borrow a
3824 /// subset of the source root rather than everything it declares. Empty
3825 /// (the default) borrows everything.
3826 #[serde(default)]
3827 pub select: Vec<String>,
3828}
3829
3830fn default_sources_schema_version() -> u32 {
3831 1
3832}
3833
3834/// `.yah/infra/sources.toml` — the ordered list of external infra roots this
3835/// camp borrows from (R615-F1 / W274).
3836#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3837#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3838pub struct SourcesConfig {
3839 #[serde(default = "default_sources_schema_version")]
3840 pub schema_version: u32,
3841 /// `[[source]]` entries, in declaration order — overlay order matters
3842 /// when two linked sources both name the same machine (R615-F2).
3843 #[serde(default, rename = "source")]
3844 pub source: Vec<InfraSource>,
3845}
3846
3847impl Default for SourcesConfig {
3848 fn default() -> Self {
3849 Self {
3850 schema_version: default_sources_schema_version(),
3851 source: Vec::new(),
3852 }
3853 }
3854}
3855
3856impl SourcesConfig {
3857 /// Load `<infra_dir>/sources.toml`. A missing file is not an error —
3858 /// every camp without linked infra has none, which today is every camp —
3859 /// and yields an empty source list rather than `Err`.
3860 pub fn load(infra_dir: &Path) -> Result<Self> {
3861 let path = infra_dir.join("sources.toml");
3862 if !path.exists() {
3863 return Ok(Self::default());
3864 }
3865 let src =
3866 std::fs::read_to_string(&path).with_context(|| format!("reading {}", path.display()))?;
3867 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
3868 }
3869}
3870
3871impl InfraSource {
3872 /// Human-readable descriptor of *which* source this is, for
3873 /// [`InfraOrigin::source`] — distinguishes two linked sources from the
3874 /// same owner. Never includes credentials: `GitSource.repo` is a clone
3875 /// URL (https/ssh), the same thing R561 already treats as safe to log,
3876 /// with any real secret resolved separately via `keystore://` (W274's
3877 /// own precedent).
3878 fn describe(&self) -> String {
3879 match &self.kind {
3880 InfraSourceKind::Path { path } => format!("path:{path}"),
3881 InfraSourceKind::Git(g) => format!("git:{}@{}", g.repo, g.r#ref),
3882 }
3883 }
3884
3885 /// Resolve this source to an infra root directory (R615-F2 / W274 §3).
3886 /// Does no I/O and touches no network: `path` sources read the owner's
3887 /// live tree directly; `git` sources read wherever `yah infra sync`
3888 /// (R615-T3) last synced to, which may not exist yet (an unsynced git
3889 /// source overlays nothing, not an error — see [`load_dir_tolerant`]).
3890 ///
3891 /// `git.subdir` (reused verbatim from [`GitSource`]/R561) is honoured
3892 /// exactly like the component case: the checkout root when unset, or
3893 /// `<checkout>/<subdir>` when set — e.g. `subdir = "infra"` for a
3894 /// monorepo whose infra registry lives under `infra/` rather than at the
3895 /// clone's root. `yah infra sync` (R615-T3) clones into the *checkout*
3896 /// root ([`crate::paths::infra_source_cache_dir`]), never into a
3897 /// subdir-suffixed path, so this is the one place that appends `subdir`.
3898 fn infra_root(&self, workspace_root: &Path) -> std::path::PathBuf {
3899 match &self.kind {
3900 InfraSourceKind::Path { path } => workspace_root.join(path).join(".yah").join("infra"),
3901 InfraSourceKind::Git(g) => {
3902 let checkout = crate::paths::infra_source_cache_dir(workspace_root, &self.owner);
3903 match g.subdir.as_deref() {
3904 Some(subdir) => checkout.join(subdir),
3905 None => checkout,
3906 }
3907 }
3908 }
3909 }
3910}
3911
3912/// Provenance for a [`MachineConfig`] or [`ProviderConfig`] pulled in from a
3913/// linked `.yah/infra/sources.toml` entry, rather than declared in this
3914/// camp's own `.yah/infra/` (R615-F2 / W274).
3915///
3916/// Lives in [`CloudConfig::machine_origins`] / `provider_origins`, keyed by
3917/// name/id, rather than as a field on `MachineConfig`/`ProviderConfig`
3918/// themselves: those two types are constructed by struct literal in test
3919/// helpers across several crates (including ones this ticket has no reason to
3920/// touch), so widening either shape would ripple out past this crate for no
3921/// semantic gain — origin is a property of *this load*, not an inherent
3922/// property of the machine/provider. A name absent from the map is
3923/// camp-local; present means borrowed, and the Infra tab (R615-F4) / reconcile
3924/// gating (`InfraSource::mode`, copied onto `mode` below) read it from here.
3925#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
3926#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
3927pub struct InfraOrigin {
3928 /// The [`InfraSource::owner`] that supplied this entry, e.g. `"yah"`.
3929 pub owner: String,
3930 /// Which source, rendered — see [`InfraSource::describe`].
3931 pub source: String,
3932 /// The write-gate that applied when this entry was overlaid — copied
3933 /// from [`InfraSource::mode`] so a caller holding just the machine/
3934 /// provider doesn't need the source list in hand to know it's borrowed
3935 /// read-only.
3936 pub mode: SourceMode,
3937}
3938
3939/// Like [`load_dir`], but tolerant **per file**: a foreign infra root (an
3940/// owner's live tree, or a synced git checkout) can carry entries this
3941/// binary's `T` predates — noisetable's pre-migration machines used an older
3942/// schema than yah's, and the reverse will happen too as each side evolves
3943/// independently. One unparseable file on a source this camp doesn't own must
3944/// never sink every other entry in the same directory, let alone this camp's
3945/// own load (R615-F2 gotcha). Contrast [`load_dir`], which stays strict for
3946/// camp-local files, where a malformed TOML genuinely should be a hard error.
3947///
3948/// Returns the entries that parsed, plus `(path, error)` for every file that
3949/// didn't — the caller logs those, it doesn't drop them silently. A missing
3950/// or unreadable directory yields `(vec![], vec![])`, same "no entries" as
3951/// `load_dir`'s `!dir.exists()` case (an unsynced git source, or a source
3952/// root with no `providers/` at all, are both normal, not warnings).
3953fn load_dir_tolerant<T: for<'de> Deserialize<'de>>(
3954 dir: &Path,
3955) -> (Vec<T>, Vec<(std::path::PathBuf, anyhow::Error)>) {
3956 let Ok(read_dir) = std::fs::read_dir(dir) else {
3957 return (Vec::new(), Vec::new());
3958 };
3959 let mut entries: Vec<_> = read_dir.filter_map(|e| e.ok()).collect();
3960 entries.sort_by_key(|e| e.file_name());
3961
3962 let mut items = Vec::new();
3963 let mut skipped = Vec::new();
3964 for entry in entries {
3965 let path = entry.path();
3966 if path.extension().map_or(true, |e| e != "toml") {
3967 continue;
3968 }
3969 let parsed = std::fs::read_to_string(&path)
3970 .with_context(|| format!("reading {}", path.display()))
3971 .and_then(|src| {
3972 toml::from_str::<T>(&src).with_context(|| format!("parsing {}", path.display()))
3973 });
3974 match parsed {
3975 Ok(item) => items.push(item),
3976 Err(e) => skipped.push((path, e)),
3977 }
3978 }
3979 (items, skipped)
3980}
3981
3982/// Whether a borrowed machine passes an [`InfraSource::select`] filter
3983/// (R615-F2 / W274). Empty `select` borrows everything. A non-empty `select`
3984/// entry matches either the machine's exact `name` or literal membership in
3985/// its `mesh_tags` — the one shape W274's own example uses
3986/// (`select = ["tag:cloud-runner"]`). Not a glob engine: mesh tags are
3987/// already flat strings compared for exact equality everywhere else in this
3988/// crate (see `resolve_machine_by_mesh_tags`), so a select entry is that same
3989/// comparison, not a new pattern language.
3990fn machine_matches_select(machine: &MachineConfig, select: &[String]) -> bool {
3991 select.is_empty()
3992 || select
3993 .iter()
3994 .any(|s| *s == machine.name || machine.mesh_tags.contains(s))
3995}
3996
3997/// What one `[[source]]` in `.yah/infra/sources.toml` actually contributed to
3998/// [`FleetInventory`] on this load (R870-B13).
3999///
4000/// Recorded because a link that resolves to *nothing* is indistinguishable, at
4001/// every downstream use site, from a camp that declared no link at all — and
4002/// that is precisely the failure this ticket exists to fix. A source that
4003/// contributes zero machines is not an error here (an unsynced `kind = "git"`
4004/// source is legitimately empty, and `load()` must stay offline), so instead
4005/// the fact is *carried* to whoever fails for want of a machine. See
4006/// [`FleetInventory::describe_sources`].
4007#[derive(Debug, Clone, PartialEq, Eq)]
4008pub struct SourceContribution {
4009 /// [`InfraSource::owner`] — the name this camp knows the fleet by.
4010 pub owner: String,
4011 /// The source, rendered — see [`InfraSource::describe`].
4012 pub source: String,
4013 /// Where the link resolved to, i.e. the foreign `.yah/infra/`.
4014 pub root: std::path::PathBuf,
4015 /// Whether `root` exists on disk. `false` for a `kind = "path"` link
4016 /// aimed at a directory that is not a camp, and for a `kind = "git"`
4017 /// source that `yah infra sync` has never fetched.
4018 pub root_exists: bool,
4019 /// How many machines this source actually added to the inventory — after
4020 /// [`InfraSource::select`] filtering and after losing every name a
4021 /// camp-local entry or an earlier source already claimed.
4022 pub machines: usize,
4023}
4024
4025/// A camp's resolved machine inventory: **the** answer to "which machines does
4026/// this camp have", with exactly one implementation
4027/// ([`resolve_fleet_inventory`]) behind it (R870-B13).
4028///
4029/// A borrowing camp — one whose own `.yah/infra/machines/` is empty and which
4030/// declares `[[source]]` links to another camp's fleet in
4031/// `.yah/infra/sources.toml` — is the case this type exists for. Before it,
4032/// the overlay was applied inline inside [`CloudConfig::load`], so the two
4033/// callers that resolve a *machine name to a machine* (ingress collation and
4034/// the sovereign apex render) read a camp-local-only loader and saw an empty
4035/// fleet. There was no bug in either of them; the inventory simply had two
4036/// readers that disagreed about what the inventory was.
4037#[derive(Debug)]
4038pub struct FleetInventory {
4039 /// Camp-local machines first, then each source's contribution in
4040 /// declaration order. Camp-local wins any name collision; among sources,
4041 /// the earlier-declared one wins.
4042 pub machines: Vec<MachineConfig>,
4043 /// Provenance for the borrowed entries, keyed by [`MachineConfig::name`].
4044 /// A name absent here is camp-local. Same shape and meaning as
4045 /// [`CloudConfig::machine_origins`], which is populated from this.
4046 pub origins: BTreeMap<String, InfraOrigin>,
4047 /// The parsed `.yah/infra/sources.toml`, kept so a caller that already has
4048 /// an inventory in hand does not re-read it (`CloudConfig::load` overlays
4049 /// providers from the same list).
4050 pub sources: SourcesConfig,
4051 /// Per-source accounting — see [`SourceContribution`].
4052 pub contributions: Vec<SourceContribution>,
4053}
4054
4055impl FleetInventory {
4056 /// One line per declared `[[source]]`, for attaching to the error a caller
4057 /// raises when a machine name does not resolve (R870-B13).
4058 ///
4059 /// The failure being diagnosed is always "I was told about machine X and
4060 /// cannot find it", and the three ways a borrowing camp gets there — no
4061 /// link declared, a link pointing somewhere that is not a camp, a link
4062 /// whose `select` filtered X out — are indistinguishable from the name
4063 /// alone. Empty string when the camp declares no sources, so the caller
4064 /// can append it unconditionally without emitting a dangling header.
4065 pub fn describe_sources(&self) -> String {
4066 if self.contributions.is_empty() {
4067 return String::new();
4068 }
4069 let mut out = String::from("linked infra sources consulted:");
4070 for c in &self.contributions {
4071 out.push_str(&format!(
4072 "\n {} ({}) -> {}{} — contributed {} machine(s)",
4073 c.owner,
4074 c.source,
4075 c.root.display(),
4076 if c.root_exists {
4077 ""
4078 } else {
4079 " [ABSENT: not a camp, or an unsynced git source]"
4080 },
4081 c.machines,
4082 ));
4083 }
4084 out
4085 }
4086}
4087
4088/// Resolve a camp's machine inventory: camp-local `.yah/infra/machines/`, the
4089/// pre-R215 `.yah/cloud/machines/` tree, then every machine borrowed through
4090/// `.yah/infra/sources.toml` (R870-B13, on R615-F2's mechanism).
4091///
4092/// **How a camp names another camp's fleet**, decided here rather than
4093/// invented: through the `[[source]]` entry R615-F1 already defines — `owner`
4094/// is the logical name an operator sees, `kind = "path"` resolves against the
4095/// borrowing camp's own root and `kind = "git"` against `yah infra sync`'s
4096/// cache. There is deliberately no second naming scheme: a camp that could
4097/// name a foreign fleet two ways would be a camp whose inventory can drift
4098/// from itself, which is the thing this ticket rejected.
4099///
4100/// **There is exactly one copy.** A `kind = "path"` source reads the owner's
4101/// live tree at `<path>/.yah/infra/` on every load — the borrowing camp
4102/// persists nothing, so the two can never disagree. `kind = "git"` reads a
4103/// synced checkout, which *is* a copy, but an explicit one with a named
4104/// refresh verb (`yah infra sync`) and a pinned `ref`; that is the cache with
4105/// an invalidation story, as against a hand-maintained second inventory.
4106///
4107/// Camp-local files are strict (a malformed TOML this camp owns is a hard
4108/// error) and foreign files are tolerant per-file (R615-F2: a foreign entry
4109/// whose schema this binary predates must not sink the load). A foreign
4110/// machine skipped that way is not silently lost — it fails loudly at the
4111/// point some caller needs it, with [`FleetInventory::describe_sources`]
4112/// naming the link it should have come from.
4113///
4114/// Deliberately *without* [`CloudConfig::load`]'s R844-B7 wrong-root guard: a
4115/// missing `.yah/infra/machines/` is an empty inventory here, because the
4116/// callers that resolve against it (ingress collation, apex render) are handed
4117/// a root that a `CloudConfig::load` already accepted.
4118pub fn resolve_fleet_inventory(workspace_root: &Path) -> Result<FleetInventory> {
4119 let mut machines = load_dir::<MachineConfig>(crate::paths::machines_dir(workspace_root))?;
4120
4121 // Pre-R215 `.yah/cloud/machines/`. Shouldn't have anything since R215-B1
4122 // moved them, but if it does we dedupe by name — R215+ wins.
4123 let cloud_dir = crate::paths::legacy_cloud_dir(workspace_root);
4124 if cloud_dir.exists() {
4125 let names: std::collections::HashSet<String> =
4126 machines.iter().map(|m| m.name.clone()).collect();
4127 for m in load_dir::<MachineConfig>(cloud_dir.join("machines"))? {
4128 if !names.contains(&m.name) {
4129 machines.push(m);
4130 }
4131 }
4132 }
4133
4134 // `SourcesConfig::load` never touches the network — git sources are read
4135 // from `yah infra sync`'s cache (R615-T3) — so this keeps the whole
4136 // offline contract `CloudConfig::load` has always had.
4137 let sources = SourcesConfig::load(&crate::paths::infra_dir(workspace_root))?;
4138 let mut origins = BTreeMap::new();
4139 let contributions = overlay_source_machines(workspace_root, &sources, &mut machines, &mut origins);
4140
4141 Ok(FleetInventory {
4142 machines,
4143 origins,
4144 sources,
4145 contributions,
4146 })
4147}
4148
4149/// Overlay every linked `.yah/infra/sources.toml` source's machines into
4150/// `machines`, recording provenance into `machine_origins` (R615-F2 / W274).
4151/// Must be called AFTER camp-local entries are already in the vector:
4152/// collision resolution is "first writer wins," so seeding with camp-local
4153/// first is what makes camp-local win over every source, and an earlier source
4154/// win over a later one.
4155///
4156/// `select` filters which machines a source contributes. Returns one
4157/// [`SourceContribution`] per declared source, in declaration order.
4158fn overlay_source_machines(
4159 workspace_root: &Path,
4160 sources: &SourcesConfig,
4161 machines: &mut Vec<MachineConfig>,
4162 machine_origins: &mut BTreeMap<String, InfraOrigin>,
4163) -> Vec<SourceContribution> {
4164 let mut seen_machine_names: std::collections::HashSet<String> =
4165 machines.iter().map(|m| m.name.clone()).collect();
4166 let mut contributions = Vec::with_capacity(sources.source.len());
4167
4168 for source in &sources.source {
4169 let root = source.infra_root(workspace_root);
4170 let origin = InfraOrigin {
4171 owner: source.owner.clone(),
4172 source: source.describe(),
4173 mode: source.mode,
4174 };
4175
4176 let (foreign_machines, skipped) = load_dir_tolerant::<MachineConfig>(&root.join("machines"));
4177 for (path, e) in skipped {
4178 tracing::warn!(
4179 "infra source {:?} ({}): skipping unparseable machine {}: {e:#}",
4180 source.owner,
4181 root.display(),
4182 path.display()
4183 );
4184 }
4185 let mut added = 0usize;
4186 for m in foreign_machines {
4187 if seen_machine_names.contains(&m.name) {
4188 continue; // camp-local, or an earlier source, already claimed this name
4189 }
4190 if !machine_matches_select(&m, &source.select) {
4191 continue;
4192 }
4193 seen_machine_names.insert(m.name.clone());
4194 machine_origins.insert(m.name.clone(), origin.clone());
4195 machines.push(m);
4196 added += 1;
4197 }
4198
4199 contributions.push(SourceContribution {
4200 owner: source.owner.clone(),
4201 source: source.describe(),
4202 root_exists: root.is_dir(),
4203 root,
4204 machines: added,
4205 });
4206 }
4207
4208 contributions
4209}
4210
4211/// Overlay every linked source's providers into `providers`, recording
4212/// provenance into `provider_origins` (R615-F2 / W274). Same first-writer-wins
4213/// rule as [`overlay_source_machines`], and the same requirement that
4214/// camp-local entries already be in the vector.
4215///
4216/// [`InfraSource::select`] deliberately does not apply: nothing in W274 or
4217/// R615-F1 describes a provider-scoped filter — every provider a source
4218/// declares either overlays whole or, on an id collision, doesn't.
4219fn overlay_source_providers(
4220 workspace_root: &Path,
4221 sources: &SourcesConfig,
4222 providers: &mut Vec<ProviderConfig>,
4223 provider_origins: &mut BTreeMap<String, InfraOrigin>,
4224) {
4225 let mut seen_provider_ids: std::collections::HashSet<String> =
4226 providers.iter().map(|p| p.id.clone()).collect();
4227
4228 for source in &sources.source {
4229 let root = source.infra_root(workspace_root);
4230 let origin = InfraOrigin {
4231 owner: source.owner.clone(),
4232 source: source.describe(),
4233 mode: source.mode,
4234 };
4235
4236 let (foreign_providers, skipped) =
4237 load_dir_tolerant::<ProviderConfig>(&root.join("providers"));
4238 for (path, e) in skipped {
4239 tracing::warn!(
4240 "infra source {:?} ({}): skipping unparseable provider {}: {e:#}",
4241 source.owner,
4242 root.display(),
4243 path.display()
4244 );
4245 }
4246 for p in foreign_providers {
4247 if seen_provider_ids.contains(&p.id) {
4248 continue;
4249 }
4250 seen_provider_ids.insert(p.id.clone());
4251 provider_origins.insert(p.id.clone(), origin.clone());
4252 providers.push(p);
4253 }
4254 }
4255}
4256
4257/// One component of a [`ServiceConfig`]. The `kind` (e.g. `"mesofact-static"`,
4258/// `"almanac"`, `"container"`) selects which reconciler runs against the
4259/// pointed-at workload manifest.
4260#[derive(Debug, Clone, Serialize, Deserialize)]
4261#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4262pub struct ServiceComponent {
4263 pub id: String,
4264 pub kind: String,
4265 /// Path of the directory holding this component's `workload.toml`. Relative
4266 /// to the workspace root for in-tree components, or to the materialized
4267 /// `<checkout>/<subdir>` when [`git`](Self::git) is set.
4268 pub path: String,
4269 /// Optional external git source (R561-F1). When set, the component's code
4270 /// is materialized by shallow-clone before build; see [`GitSource`].
4271 #[serde(default, skip_serializing_if = "Option::is_none")]
4272 pub git: Option<GitSource>,
4273 /// Operator-facing role label, e.g. `"static"`, `"dynamic"`, `"compute"`.
4274 pub role: String,
4275 /// Optional artifact kind this component publishes (`"static"`,
4276 /// `"container-image"`, …). Drives mirror provider-slot routing.
4277 #[serde(default, skip_serializing_if = "Option::is_none")]
4278 pub publishes: Option<String>,
4279 /// URL sub-path a static component's build output is published under,
4280 /// relative to the service's publish prefix (R746). `None` = the service
4281 /// root, which is what every pre-R746 component means.
4282 ///
4283 /// Static publishers lay a component's `out_dir` down at
4284 /// `<bucket>/<service>/<env>/…` and the front door fetches
4285 /// `${ASSET_ORIGIN}/<request path>` — the request path *is* the key. So a
4286 /// service with two static components had them overwrite each other at
4287 /// one prefix, and there was no way to say "this bundle serves under
4288 /// /app". `mount` is that: it appends to the publish prefix, which makes
4289 /// the URL sub-path and the storage sub-path the same string by
4290 /// construction rather than by two manifests agreeing.
4291 ///
4292 /// Cross-checked against the domain route that names the component
4293 /// ([`CloudConfig::cross_ref_validate`]): a component mounted at `/app`
4294 /// must be routed at `/app` or `/app/*`, because a disagreement means
4295 /// requests land on a prefix nothing published to — a 404 whose cause is
4296 /// two files apart.
4297 #[serde(default, skip_serializing_if = "Option::is_none")]
4298 pub mount: Option<String>,
4299 /// Sync-wave index (0-based). Components in wave 0 roll out in parallel
4300 /// first; the reconciler waits for all wave-N components to become healthy
4301 /// before starting wave N+1. Defaults to 0 (all components in one wave).
4302 #[serde(default, skip_serializing_if = "is_zero_u32")]
4303 pub wave: u32,
4304
4305 /// Whether this component ships inside the service's one assembled bundle
4306 /// or as a deployed unit of its own (R870-F23).
4307 ///
4308 /// This is the vocabulary R870-F15's design needed and the config did not
4309 /// have. `[providers.bundle]` is a per-**mirror** slot, so before this
4310 /// there was no way to say "give this one component its own workload" at
4311 /// all — the whole service was one bundle or it was nothing, and a service
4312 /// whose components genuinely release on different cadences had no shape
4313 /// to declare.
4314 ///
4315 /// It is one field rather than a pair of flags on purpose: a component
4316 /// being both bundle-staged and its own workload is the second admission
4317 /// rule R870-F23 was asked to enforce, and an enum makes it unrepresentable
4318 /// instead of merely refused.
4319 #[serde(default, skip_serializing_if = "DeployTier::is_default")]
4320 pub deploy: DeployTier,
4321}
4322
4323/// How one [`ServiceComponent`] reaches a node — see
4324/// [`ServiceComponent::deploy`].
4325#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
4326#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4327#[serde(rename_all = "kebab-case")]
4328pub enum DeployTier {
4329 /// Staged into the service's single assembled W272 bundle under
4330 /// `app/dist/<mount>/` and served by the one bundle workload (R870-B11).
4331 /// The default, and what every component in the tree means today.
4332 #[default]
4333 Bundle,
4334 /// Deployed as its own workload, with its own release cadence, its own
4335 /// address, and its own place in the inner door's mount table.
4336 Workload,
4337}
4338
4339impl DeployTier {
4340 /// Skip serializing the default so existing `service.toml` files
4341 /// round-trip byte-identically.
4342 fn is_default(&self) -> bool {
4343 matches!(self, DeployTier::Bundle)
4344 }
4345}
4346
4347#[inline]
4348fn is_zero_u32(n: &u32) -> bool {
4349 *n == 0
4350}
4351
4352/// A service's declared databases, grouped by environment (W241 §Sections).
4353/// Parsed from the `[db]` table of `service.toml`; each `[[db.<env>]]` array
4354/// entry names one database. The environment tag drives backend selection at
4355/// query time (see the data-workbench's `db.query` / the `sql_*` MCP tools):
4356/// `dev` = local file, `pond` = a DB inside the running pond container stack
4357/// (reached on a declared localhost port), `cloud` = a remote libSQL/Turso or
4358/// Postgres endpoint whose auth comes from an env var (never stored in TOML).
4359#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq, Eq)]
4360#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4361pub struct DbCatalog {
4362 /// Local-file SQLite databases used in dev mode.
4363 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4364 pub dev: Vec<DevDb>,
4365 /// Databases running inside the pond container stack.
4366 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4367 pub pond: Vec<PondDb>,
4368 /// Remote cloud databases (Turso, Postgres).
4369 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4370 pub cloud: Vec<CloudDb>,
4371}
4372
4373impl DbCatalog {
4374 /// True when no database is declared in any environment. Lets
4375 /// [`ServiceConfig`] skip serializing an empty `[db]` table.
4376 pub fn is_empty(&self) -> bool {
4377 self.dev.is_empty() && self.pond.is_empty() && self.cloud.is_empty()
4378 }
4379}
4380
4381/// A dev-mode local SQLite database (`[[db.dev]]`). `path` is resolved
4382/// relative to the workspace root and opened as a local file — read/write, no
4383/// network, no auth.
4384#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
4385#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4386pub struct DevDb {
4387 /// Logical name, unique within the service's `dev` list. Forms the `name`
4388 /// segment of the catalog id `dev:<service>:<name>`.
4389 pub name: String,
4390 /// On-disk SQLite path, relative to the workspace root (or absolute).
4391 pub path: String,
4392}
4393
4394/// A database running inside the pond container stack (`[[db.pond]]`). The
4395/// pond publishes the DB on a localhost TCP port; the hub connects to
4396/// `127.0.0.1:<port>` when the pond is up and returns a clear error when it is
4397/// not. Either `port` (defaulting to a libSQL/`sqld` HTTP endpoint) or a full
4398/// `url` must be given.
4399#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
4400#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4401pub struct PondDb {
4402 /// Logical name, unique within the service's `pond` list.
4403 pub name: String,
4404 /// Localhost TCP port the pond publishes the DB on. Interpreted per
4405 /// [`kind`](Self::kind). Mutually complete with `url` (provide one).
4406 #[serde(default, skip_serializing_if = "Option::is_none")]
4407 pub port: Option<u16>,
4408 /// Full connection URL, overriding `port` when set (e.g. a non-localhost
4409 /// host or an explicit scheme).
4410 #[serde(default, skip_serializing_if = "Option::is_none")]
4411 pub url: Option<String>,
4412 /// Wire protocol the pond DB speaks. Selects how a bare `port` becomes a
4413 /// URL: `turso` → `http://127.0.0.1:<port>` (libSQL/`sqld` over Hrana),
4414 /// `postgres` → `postgres://127.0.0.1:<port>`.
4415 #[serde(default)]
4416 pub kind: PondDbKind,
4417}
4418
4419/// Wire protocol of a [`PondDb`].
4420#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq, Eq)]
4421#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4422#[serde(rename_all = "kebab-case")]
4423pub enum PondDbKind {
4424 /// libSQL / `sqld` over Hrana HTTP — the default.
4425 #[default]
4426 Turso,
4427 /// PostgreSQL wire protocol.
4428 Postgres,
4429}
4430
4431/// A remote cloud database (`[[db.cloud]]`). The connection `url` is stored in
4432/// TOML but the credential never is — `auth_token_env` names an environment
4433/// variable the daemon reads at connect time, so the same declaration works
4434/// whether the token is provisioned service-locally or camp-shared (W241;
4435/// operator confirmed both scopes are needed). A camp-wide cloud DB not owned
4436/// by any single service is declared identically in `.yah/db/cloud.toml`.
4437#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
4438#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4439pub struct CloudDb {
4440 /// Logical name, unique within its `cloud` list.
4441 pub name: String,
4442 /// Connection URL: `libsql://…` / `http(s)://…` (Turso, `sqld`) or
4443 /// `postgres://…`.
4444 pub url: String,
4445 /// Name of the environment variable holding the auth token. Resolved in
4446 /// the daemon at connect time (value never stored on disk). For a libSQL
4447 /// URL the token is threaded as `?auth_token=…`.
4448 #[serde(default, skip_serializing_if = "Option::is_none")]
4449 pub auth_token_env: Option<String>,
4450}
4451
4452/// A camp-shared cloud database catalog, parsed from `.yah/db/cloud.toml`.
4453/// These are cloud DBs not owned by any single service — declared once at camp
4454/// scope and addressed as `cloud:<name>` (two-segment id), distinct from a
4455/// service-local `cloud:<service>:<name>`.
4456#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq, Eq)]
4457#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4458pub struct CampCloudDbs {
4459 #[serde(default, rename = "cloud", skip_serializing_if = "Vec::is_empty")]
4460 pub cloud: Vec<CloudDb>,
4461}
4462
4463impl CampCloudDbs {
4464 /// Load `<camp_root>/.yah/db/cloud.toml`, or an empty catalog if the file
4465 /// is absent (the common case — most camps declare no shared cloud DBs).
4466 pub fn load(camp_root: &Path) -> Result<Self> {
4467 let path = camp_root.join(".yah/db/cloud.toml");
4468 if !path.exists() {
4469 return Ok(Self::default());
4470 }
4471 let src = std::fs::read_to_string(&path)
4472 .with_context(|| format!("reading {}", path.display()))?;
4473 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
4474 }
4475}
4476
4477/// Topological shape of a mirror — how its providers sit relative to each other.
4478#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
4479#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4480#[serde(rename_all = "kebab-case")]
4481pub enum MirrorShape {
4482 /// Single machine hosts compute (and any non-Cloudflare-fronted static).
4483 SingleMachine,
4484 /// Operator-local dev mirror — static via built-in file server, compute
4485 /// via the local container runtime.
4486 Local,
4487 /// Multi-machine deployment (machines listed per provider slot).
4488 MultiMachine,
4489}
4490
4491/// Which public-ingress provider fronts this mirror's compute (W267, R594-F11).
4492///
4493/// Both arms answer exactly one question — *given these local workload ports,
4494/// make them publicly reachable at these hostnames* — and they differ only in
4495/// where the ingress rules live and who supervises the front door:
4496///
4497/// | | [`CloudflareTunnel`](Self::CloudflareTunnel) | [`Passway`](Self::Passway) |
4498/// |---|---|---|
4499/// | Ingress rules live | Cloudflare's API (token-form tunnels are remotely-managed) | the pingora `Backends` set in the proxy process |
4500/// | How they get there | an API call per deployed workload | passway polls `GET /service-records?ready=true` |
4501/// | Front door lifecycle | a kamaji-supervised `cloudflared` appliance | a kamaji-supervised passway appliance |
4502///
4503/// Flipping this field is the whole tier ladder: rented edge → sovereign edge
4504/// is a one-line mirror edit, not a rewrite. The provider owns **addressing**
4505/// and never **rendering** — the W173 render cube stays in mesofact's manifest.
4506#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
4507#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4508#[serde(rename_all = "kebab-case")]
4509pub enum IngressProvider {
4510 /// No public front door for this mirror. The default: a mirror that
4511 /// publishes to R2 behind a Worker, or a mesh-only compute tier, has no
4512 /// ingress provider to reconcile.
4513 #[default]
4514 None,
4515 /// Rented edge — `cloudflared` dials *out* from the node to Cloudflare's
4516 /// edge. Zero inbound ports, no TLS to manage on the box, hostname rules
4517 /// held in Cloudflare's API.
4518 CloudflareTunnel,
4519 /// Sovereign edge — passway terminates TLS on the node and load-balances
4520 /// an upstream set discovered from yubaba's service records.
4521 Passway,
4522}
4523
4524impl IngressProvider {
4525 /// `true` when this mirror declares a front door that has to be reconciled.
4526 pub fn is_declared(self) -> bool {
4527 !matches!(self, Self::None)
4528 }
4529
4530 /// Kebab-case wire name, as it appears in `mirrors/<env>.toml`.
4531 pub fn as_str(self) -> &'static str {
4532 match self {
4533 Self::None => "none",
4534 Self::CloudflareTunnel => "cloudflare-tunnel",
4535 Self::Passway => "passway",
4536 }
4537 }
4538}
4539
4540/// What a `cloudflare-tunnel` edge dials instead of the fronted workload —
4541/// W348 §1.3's **stacked** shape (R910).
4542///
4543/// `via = "passway"` puts the tunnel in front of the same node's passway door
4544/// for the same hostnames: cloudflared → the node's sni-demux on loopback
4545/// `:443` → the per-tenant passway → the workload. Passway keeps everything it
4546/// does on a public door — origin TLS, host routing, `[ingress.auth]`, ACME
4547/// (by DNS-01, the one challenge that reaches a NAT'd node) — and Cloudflare
4548/// owns only the browser-facing handshake. Without it a tunnel edge dials the
4549/// workload directly, and a tunnel edge and a passway edge claiming one
4550/// hostname is a partition conflict.
4551///
4552/// An enum rather than a bool because what it names is a front door, and
4553/// passway is simply the only one a tunnel can stack in front of today.
4554#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
4555#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4556#[serde(rename_all = "kebab-case")]
4557pub enum IngressVia {
4558 /// The node's own passway door, reached through its sni-demux.
4559 Passway,
4560}
4561
4562impl IngressVia {
4563 /// Kebab-case wire name, as it appears in `mirrors/<env>.toml`.
4564 pub fn as_str(self) -> &'static str {
4565 match self {
4566 Self::Passway => "passway",
4567 }
4568 }
4569
4570 /// The provider of the edge a tunnel carrying this `via` stacks in front
4571 /// of — the edge that has to exist, on the same machines, for the pair to
4572 /// plan.
4573 pub fn provider(self) -> IngressProvider {
4574 match self {
4575 Self::Passway => IngressProvider::Passway,
4576 }
4577 }
4578}
4579
4580/// One declared **edge**: a front door, the slots it fronts, and the nodes it
4581/// is placed on (W305 F2).
4582///
4583/// A mirror declares a *list* of these, which is what lets one service mix
4584/// front doors — cloudflare for the public web tier, passway for an internal or
4585/// high-throughput one. Before this, [`MirrorConfig::ingress`] was a single
4586/// [`IngressProvider`], so a mirror could **swap** front doors but never mix
4587/// them.
4588///
4589/// ```toml
4590/// [[ingress]]
4591/// provider = "passway"
4592/// machines = ["us-east-001", "us-south-001"]
4593/// slots = ["bundle"]
4594///
4595/// [[ingress]]
4596/// provider = "cloudflare-tunnel"
4597/// hostnames = ["issues.yah.dev"]
4598/// ```
4599///
4600/// **The per-node appliance is derived from this, never declared beside it.**
4601/// An edge does invoke a cloudflared or passway process on a box, but that is a
4602/// *consequence* of the service's declaration:
4603/// [`collate_front_doors`](crate::reconciler::collate_front_doors) walks every
4604/// service and derives what each node must run. Declaring it node-side too is
4605/// what produces two sources of truth for one fact.
4606#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
4607#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4608pub struct IngressEdge {
4609 /// Which front door this edge is. [`IngressProvider::None`] is rejected at
4610 /// plan time — an edge that fronts with nothing is always a typo, never an
4611 /// intent (write no edge instead).
4612 pub provider: IngressProvider,
4613 /// Nodes this front door is placed on — **independent of where the fronted
4614 /// workload runs** (R330-F37).
4615 ///
4616 /// Empty falls back to the fronted slot's own `machine` / `machines`, which
4617 /// is the co-located shape every mirror had before front-door placement was
4618 /// expressible. Listing several is what lets the ingress tier and the
4619 /// service tier scale independently: **N front doors over ONE deployment**,
4620 /// one rendered copy, so no cache coherence to settle.
4621 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4622 pub machines: Vec<String>,
4623 /// Provider slot roles this edge fronts (`"bundle"`, `"compute"`, …).
4624 ///
4625 /// One of the two selectors. With a single edge both may be empty, meaning
4626 /// "every fronted slot" — the legacy shape. With **several** edges a
4627 /// selector is mandatory on each, and the partition must be total and
4628 /// disjoint: a slot claimed by no edge, or by two, is an error naming it.
4629 /// An implicit catch-all across mixed front doors would silently publish a
4630 /// service through the wrong one.
4631 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4632 pub slots: Vec<String>,
4633 /// Public hostnames this edge fronts — the other selector, for partitioning
4634 /// by what the world dials rather than by which slot serves it.
4635 #[serde(default, skip_serializing_if = "Vec::is_empty")]
4636 pub hostnames: Vec<String>,
4637 /// Cloudflare Tunnel id this edge publishes through, overriding the
4638 /// fronting machine's [`MachineConfig::cloudflared`].
4639 ///
4640 /// This is W267 Gap 3's real fix, and it is the *service* side of it: a node
4641 /// can join two cohorts' orange networks, and since §Granularity argues the
4642 /// tunnel credential **is** the isolation boundary, which cohort a given
4643 /// service fronts through is a property of the service, not of the box.
4644 /// `MachineConfig.cloudflared` stays as the per-node default (one tunnel is
4645 /// the common case, and the credential does live on the node), but it is no
4646 /// longer the only way to say it — so the node never has to enumerate
4647 /// cohorts.
4648 #[serde(default, skip_serializing_if = "Option::is_none")]
4649 pub tunnel_id: Option<String>,
4650 /// Infra provider id whose credentials this edge's front door authenticates
4651 /// with — `use = "cloudflare"`, resolved through
4652 /// `.yah/infra/providers/<id>.toml` exactly as a slot's `use` is.
4653 ///
4654 /// Same split as [`tunnel_id`](Self::tunnel_id), one field over: whose
4655 /// Cloudflare account holds the tunnel is a property of the **front door**,
4656 /// not of the box that runs the compute. Without this the account was read
4657 /// off the fronted slot's own `use`, which conflates two unrelated facts —
4658 /// and is unwritable for a slot whose compute provider is `kind = "static"`
4659 /// (a borrowed bare box: placement only, no credentials). Such a mirror had
4660 /// no way to name a Cloudflare account at all, short of writing
4661 /// `use = "cloudflare"` on the compute slot and lying about what runs it
4662 /// (R845).
4663 ///
4664 /// `None` falls back to the fronted slot's `use`, which is what every
4665 /// mirror written before this field meant.
4666 #[serde(default, rename = "use", skip_serializing_if = "Option::is_none")]
4667 pub provider_id: Option<String>,
4668 /// Digest-pinned image reference for this edge's front-door appliance,
4669 /// e.g. `localhost/passway:tag@sha256:<hex>` (R870-F16).
4670 ///
4671 /// `None` is the state of every mirror on disk today: the passway arm of
4672 /// `yah cloud apply` cannot deploy an appliance the mirror doesn't name an
4673 /// image for, so it renders the manual `yah cloud ingress deploy …
4674 /// --image <passway-ref>` step instead of running it. Declaring this field
4675 /// is what makes the arm self-sufficient, matching the CloudflareTunnel
4676 /// arm's real-API-call shape rather than only printing for an operator to
4677 /// copy by hand.
4678 #[serde(default, skip_serializing_if = "Option::is_none")]
4679 pub image: Option<String>,
4680 /// Cheers bearer-auth for this edge's door, spelled as an `[ingress.auth]`
4681 /// table under the `[[ingress]]` entry (R870-F26).
4682 ///
4683 /// ```toml
4684 /// [[ingress]]
4685 /// provider = "passway"
4686 /// image = "localhost/passway:v1@sha256:…"
4687 ///
4688 /// [ingress.auth]
4689 /// key_secret = "cheers/yah-camp/verify"
4690 /// kid = "YOHV4Riq-g8fX4uYl8rTjQ"
4691 /// iss = "yah-camp"
4692 /// aud = "analytics.yah.dev"
4693 /// require_prefixes = ["/"]
4694 /// ```
4695 ///
4696 /// **This is what makes an apply-driven push FAITHFUL rather than merely
4697 /// blocked.** Before it, `yah cloud apply`'s Passway arm rebuilt the door's
4698 /// spec with `auth: None` because a mirror had no way to say otherwise, and
4699 /// `/workloads/deploy` is a full replace — so pushing at a door someone had
4700 /// deployed with `--auth-key-secret …` took its auth away and brought it
4701 /// back anonymous (R870-B24). That strip is guarded by a read-back in
4702 /// `push_passway_ingress`, and the guard STAYS: it covers a door that
4703 /// acquired auth in a way no mirror can see. This field is what lets the
4704 /// common case sail past that guard by carrying the auth instead of losing
4705 /// it — the guard early-returns on any push that carries auth of its own.
4706 ///
4707 /// [`PasswayAuth`] verbatim, not a config-side copy of its five fields: all
4708 /// five are required by `Deserialize`, so a half-written table is refused
4709 /// by serde naming the missing field, and the renderer that emits the
4710 /// `PASSWAY_AUTH_*` variables reads the very same struct.
4711 ///
4712 /// Only meaningful on a `provider = "passway"` edge — a cloudflare-tunnel
4713 /// edge carrying one is refused by [`MirrorConfig::ingress_edges`] rather
4714 /// than silently ignored, since ignoring it yields exactly the
4715 /// believed-protected-but-public door this vocabulary exists to prevent.
4716 #[serde(default, skip_serializing_if = "Option::is_none")]
4717 pub auth: Option<PasswayAuth>,
4718 /// Stack this edge in front of another front door on the same node instead
4719 /// of dialing the workload (R910) — see [`IngressVia`].
4720 ///
4721 /// ```toml
4722 /// [[ingress]]
4723 /// provider = "passway"
4724 /// machines = ["us-west-011"]
4725 /// hostnames = ["api-staging.noisetable.com"]
4726 ///
4727 /// [[ingress]]
4728 /// provider = "cloudflare-tunnel"
4729 /// via = "passway"
4730 /// use = "cloudflare-tunnel-staging"
4731 /// machines = ["us-west-011"]
4732 /// hostnames = ["api-staging.noisetable.com"]
4733 /// ```
4734 ///
4735 /// Only a `cloudflare-tunnel` edge may carry it, and the mirror must also
4736 /// declare the edge it names, claiming the **same hostnames on the same
4737 /// machines** — the tunnel dials its own node's loopback demux, so a pair
4738 /// split across nodes routes to nothing. Refused otherwise by
4739 /// [`plan_ingress`](crate::reconciler::plan_ingress), naming both edges.
4740 ///
4741 /// `None` is every mirror written before R910.
4742 #[serde(default, skip_serializing_if = "Option::is_none")]
4743 pub via: Option<IngressVia>,
4744 /// The door behind a tunnel, spelled `[ingress.tunnel_door]` on the
4745 /// **passway** edge a `via = "passway"` tunnel stacks in front of (R910-F2).
4746 ///
4747 /// ```toml
4748 /// [ingress.tunnel_door]
4749 /// contact_email = "ops@example.com"
4750 /// zone_id = "<cloudflare zone id>"
4751 /// token_secret = "example/staging/cf-dns-token"
4752 /// ports = { "staging.example.com" = 8445 }
4753 /// ```
4754 ///
4755 /// Required exactly when the edge is derived behind a tunnel, refused
4756 /// otherwise — see `partition`. `yah cloud apply` turns it into one scoped
4757 /// enrollment per hostname, which the tunnel's machines route, arm and
4758 /// issue from; see [`TunnelDoor`].
4759 #[serde(default, skip_serializing_if = "Option::is_none")]
4760 pub tunnel_door: Option<TunnelDoor>,
4761}
4762
4763/// What a passway door behind a cloudflare tunnel needs that a public door
4764/// does not (R910-F2): where its per-hostname loopback listeners are, and the
4765/// inputs to issue its own certificates by DNS-01 — the only ACME challenge
4766/// that works when nothing public reaches the node.
4767///
4768/// One door (one cold passway, one certificate) per hostname, which is the
4769/// per-tenant shape yubaba's tenant tier arms — so each hostname names its own
4770/// port rather than sharing one listener.
4771#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
4772#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4773#[serde(deny_unknown_fields)]
4774pub struct TunnelDoor {
4775 /// ACME account contact.
4776 pub contact_email: String,
4777 /// Cloudflare zone id the `_acme-challenge` TXT records are written into.
4778 pub zone_id: String,
4779 /// Cluster secret holding a `Zone:DNS:Edit` token for that zone, declared
4780 /// in the machines' sovereign group with an access rule naming each
4781 /// hostname's door (`passway.<hostname>`).
4782 pub token_secret: String,
4783 /// Loopback port each hostname's door listens on — the demux backend the
4784 /// tunnel's SNI is spliced to. Every fronted hostname needs one.
4785 pub ports: std::collections::BTreeMap<String, u16>,
4786}
4787
4788impl IngressEdge {
4789 /// An edge with no selector — fronts every fronted slot, legal only when it
4790 /// is the mirror's only edge.
4791 pub fn all_slots(provider: IngressProvider, machines: Vec<String>) -> Self {
4792 Self {
4793 provider,
4794 machines,
4795 slots: Vec::new(),
4796 hostnames: Vec::new(),
4797 tunnel_id: None,
4798 provider_id: None,
4799 image: None,
4800 auth: None,
4801 via: None,
4802 tunnel_door: None,
4803 }
4804 }
4805
4806 /// Refuse `via` on an edge that cannot stack (R910). A passway edge
4807 /// terminates the connection itself, so `via` there would be ignored — and
4808 /// an ignored `via` is a mirror that reads as tunnel-fronted while
4809 /// publishing the door's own address.
4810 pub fn validate_via(&self) -> Result<()> {
4811 if self.via.is_some() && self.provider != IngressProvider::CloudflareTunnel {
4812 bail!(
4813 "{}: declares `via`, but only a `provider = \"cloudflare-tunnel\"` edge can stack \
4814 in front of another front door — a {:?} edge terminates the connection itself. \
4815 Move `via` to the tunnel edge, or drop it.",
4816 self.label(),
4817 self.provider.as_str()
4818 );
4819 }
4820 Ok(())
4821 }
4822
4823 /// Refuse an `[ingress.auth]` table that cannot produce a protected door
4824 /// (R870-F26). Called from [`MirrorConfig::ingress_edges`], so every reader
4825 /// of a mirror — plan, collate, `yah cloud validate`, apply — gets it.
4826 ///
4827 /// Two failures, and they fail in opposite directions, which is why both
4828 /// are here rather than left to the deploy:
4829 ///
4830 /// - **Auth on a non-passway edge.** Nothing downstream would read it, so
4831 /// the operator gets a door they believe is protected and is not. Only
4832 /// passway renders `PASSWAY_AUTH_*`; a cloudflare-tunnel edge publishes
4833 /// through Cloudflare Access instead and has no place to put these.
4834 /// - **A present-but-empty field.** `Deserialize` already refuses a
4835 /// *missing* one by name; [`PasswayAuth::validate`] covers the rest, and
4836 /// is the same implementation `yah cloud ingress deploy` runs on its
4837 /// flags — so the two doors cannot diverge on what counts as configured.
4838 pub fn validate_auth(&self) -> Result<()> {
4839 let Some(auth) = &self.auth else {
4840 return Ok(());
4841 };
4842 if !matches!(self.provider, IngressProvider::Passway) {
4843 bail!(
4844 "{}: declares `[ingress.auth]`, but only `provider = \"passway\"` renders the \
4845 PASSWAY_AUTH_* variables — this edge would deploy a door with NO bearer auth \
4846 while the mirror says otherwise. Move the auth to the passway edge, or drop it.",
4847 self.label()
4848 );
4849 }
4850 auth.validate(AuthSpelling::MirrorTable)
4851 .map_err(|why| anyhow::anyhow!("{}: {why}", self.label()))
4852 }
4853
4854 /// Refuse an `[ingress.tunnel_door]` that cannot produce a working door
4855 /// (R910-F2). Whether the edge is actually behind a tunnel is a property
4856 /// of the pair, so `partition` checks that half.
4857 pub fn validate_tunnel_door(&self) -> Result<()> {
4858 let Some(door) = &self.tunnel_door else {
4859 return Ok(());
4860 };
4861 if self.provider != IngressProvider::Passway {
4862 bail!(
4863 "{}: declares `[ingress.tunnel_door]`, but only the `provider = \"passway\"` edge a \
4864 tunnel stacks in front of has a door to describe. Move it to that edge, or drop it.",
4865 self.label()
4866 );
4867 }
4868 for (field, value) in [
4869 ("contact_email", &door.contact_email),
4870 ("zone_id", &door.zone_id),
4871 ("token_secret", &door.token_secret),
4872 ] {
4873 if value.trim().is_empty() {
4874 bail!("{}: `[ingress.tunnel_door].{field}` is empty", self.label());
4875 }
4876 }
4877 if door.ports.is_empty() {
4878 bail!(
4879 "{}: `[ingress.tunnel_door].ports` is empty — each fronted hostname needs the \
4880 loopback port its door listens on",
4881 self.label()
4882 );
4883 }
4884 let mut seen: std::collections::BTreeMap<u16, &str> = std::collections::BTreeMap::new();
4885 for (host, port) in &door.ports {
4886 if *port == 0 {
4887 bail!("{}: `[ingress.tunnel_door].ports.{host:?}` is 0", self.label());
4888 }
4889 if let Some(other) = seen.insert(*port, host) {
4890 bail!(
4891 "{}: `[ingress.tunnel_door].ports` gives {other:?} and {host:?} the same port \
4892 {port} — one held socket cannot be two hostnames' door",
4893 self.label()
4894 );
4895 }
4896 }
4897 Ok(())
4898 }
4899
4900 /// `true` when this edge names which slots/hostnames it fronts.
4901 pub fn has_selector(&self) -> bool {
4902 !self.slots.is_empty() || !self.hostnames.is_empty()
4903 }
4904
4905 /// Does this edge claim the rule derived from `slot` publishing `hostname`?
4906 ///
4907 /// A selectorless edge claims everything; that is checked to be
4908 /// unambiguous (one edge only) before this is consulted.
4909 pub fn claims(&self, slot: &str, hostname: &str) -> bool {
4910 if !self.has_selector() {
4911 return true;
4912 }
4913 self.slots.iter().any(|s| s == slot) || self.hostnames.iter().any(|h| h == hostname)
4914 }
4915
4916 /// Human-readable identity for an error message — the provider plus
4917 /// whichever selector was written.
4918 pub fn label(&self) -> String {
4919 let sel = match (self.slots.is_empty(), self.hostnames.is_empty()) {
4920 (true, true) => "no selector".to_string(),
4921 (false, true) => format!("slots = {:?}", self.slots),
4922 (true, false) => format!("hostnames = {:?}", self.hostnames),
4923 (false, false) => format!("slots = {:?} + hostnames = {:?}", self.slots, self.hostnames),
4924 };
4925 format!("[[ingress]] provider = {:?} ({sel})", self.provider.as_str())
4926 }
4927}
4928
4929/// A mirror's `ingress` declaration, in either spelling.
4930///
4931/// The list is the general form; the bare provider is shorthand for the single
4932/// edge fronting everything, and is kept rather than migrated because it is the
4933/// honest spelling for the common case — one service, one front door. Both
4934/// normalize to the same `Vec<IngressEdge>` through
4935/// [`MirrorConfig::ingress_edges`], so nothing downstream branches on which was
4936/// written.
4937#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
4938#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
4939#[serde(untagged)]
4940pub enum IngressDecl {
4941 /// `ingress = "passway"` — one edge fronting every fronted slot, placed by
4942 /// the sibling [`MirrorConfig::ingress_machines`].
4943 Provider(IngressProvider),
4944 /// `[[ingress]]` — one entry per declared edge.
4945 Edges(Vec<IngressEdge>),
4946}
4947
4948/// Hand-written because `#[serde(untagged)]` throws the real error away.
4949///
4950/// A derived untagged `Deserialize` tries each variant and, on failure, reports
4951/// only `data did not match any variant of untagged enum IngressDecl` — so a
4952/// misspelled `provider = "passwya"` says nothing about providers, nothing about
4953/// the legal values, and points at the `[[ingress]]` header rather than the
4954/// field. Dispatching on the input shape first means each arm's own error
4955/// survives: a bad string names the legal provider vocabulary, a bad edge table
4956/// names the offending field.
4957impl<'de> Deserialize<'de> for IngressDecl {
4958 fn deserialize<D: serde::Deserializer<'de>>(d: D) -> std::result::Result<Self, D::Error> {
4959 struct DeclVisitor;
4960
4961 impl<'de> serde::de::Visitor<'de> for DeclVisitor {
4962 type Value = IngressDecl;
4963
4964 fn expecting(&self, f: &mut std::fmt::Formatter) -> std::fmt::Result {
4965 f.write_str(
4966 "a provider name (`ingress = \"passway\"`) or a list of edge tables \
4967 (`[[ingress]]`)",
4968 )
4969 }
4970
4971 fn visit_str<E: serde::de::Error>(self, v: &str) -> std::result::Result<Self::Value, E> {
4972 IngressProvider::deserialize(serde::de::value::StrDeserializer::new(v))
4973 .map(IngressDecl::Provider)
4974 }
4975
4976 fn visit_seq<A: serde::de::SeqAccess<'de>>(
4977 self,
4978 seq: A,
4979 ) -> std::result::Result<Self::Value, A::Error> {
4980 Vec::<IngressEdge>::deserialize(serde::de::value::SeqAccessDeserializer::new(seq))
4981 .map(IngressDecl::Edges)
4982 }
4983 }
4984
4985 d.deserialize_any(DeclVisitor)
4986 }
4987}
4988
4989/// No front door — the shape of every mirror that publishes to R2 behind a
4990/// Worker, or runs a mesh-only compute tier.
4991impl Default for IngressDecl {
4992 fn default() -> Self {
4993 Self::Provider(IngressProvider::None)
4994 }
4995}
4996
4997impl IngressDecl {
4998 /// `true` when this mirror declares no front door at all.
4999 pub fn is_absent(&self) -> bool {
5000 match self {
5001 Self::Provider(p) => !p.is_declared(),
5002 Self::Edges(e) => e.is_empty(),
5003 }
5004 }
5005}
5006
5007impl From<IngressProvider> for IngressDecl {
5008 fn from(p: IngressProvider) -> Self {
5009 Self::Provider(p)
5010 }
5011}
5012
5013/// A service mirror — the projection of a [`ServiceConfig`] onto concrete
5014/// infra. Lives at `.yah/services/<svc>/mirrors/<env>.toml`.
5015#[derive(Debug, Clone, Serialize, Deserialize)]
5016#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5017pub struct MirrorConfig {
5018 pub schema_version: u32,
5019 pub shape: MirrorShape,
5020 /// Public-ingress edges fronting this mirror (W267, W305 F2). Defaults to
5021 /// none.
5022 ///
5023 /// Two spellings, one meaning — see [`IngressDecl`]. `ingress = "passway"`
5024 /// is one edge fronting everything; `[[ingress]]` entries declare several,
5025 /// each naming its provider plus the slots or hostnames it fronts. Read it
5026 /// through [`ingress_edges`](Self::ingress_edges), never by matching on the
5027 /// enum, so the two spellings cannot drift apart.
5028 ///
5029 /// Declared at mirror scope rather than per provider slot because a front
5030 /// door does **fan-in**: one `cloudflared` (or one passway) on a node
5031 /// multiplexes every hostname→port rule it fronts, so pinning one to a
5032 /// single slot would mint one edge connection per slot for no gain. An
5033 /// edge's `slots` selector is the general form of that — it groups slots
5034 /// behind one front door, it does not split a front door per slot.
5035 #[serde(default, skip_serializing_if = "IngressDecl::is_absent")]
5036 pub ingress: IngressDecl,
5037 /// Machines the front door is placed on — **independent of where the
5038 /// fronted workload runs** (R330-F37).
5039 ///
5040 /// The single-edge spelling of [`IngressEdge::machines`]: it applies to the
5041 /// one edge `ingress = "<provider>"` declares, and combining it with
5042 /// `[[ingress]]` entries is an error rather than a silent precedence rule.
5043 ///
5044 /// Empty (the default) keeps the pre-existing behaviour: the front door is
5045 /// co-located with the fronted slot's own `machine` / `machines`. That was
5046 /// never a design choice, it was an artifact of bundles binding
5047 /// `127.0.0.1` — nothing off-node could reach a workload, so a proxy had to
5048 /// sit on top of it. R599-F12 landed mesh binding, which removes the
5049 /// constraint: passway is a reverse proxy, and a valid front door needs a
5050 /// cert and an upstream it can *reach*, not a local copy of the service.
5051 ///
5052 /// Listing several machines is what lets the ingress tier and the service
5053 /// tier scale independently — **N front doors over ONE deployment**. There
5054 /// is still exactly one rendered copy of the site, so fanning the front door
5055 /// out introduces no cache-coherence problem; that only appears if you
5056 /// deploy the *workload* to every node instead.
5057 ///
5058 /// ```toml
5059 /// ingress = "passway"
5060 /// ingress_machines = ["us-east-001", "us-west-001"]
5061 /// ```
5062 ///
5063 /// Declaring this without [`ingress`](Self::ingress) is an error, not a
5064 /// no-op — it always means the operator expected a front door somewhere.
5065 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5066 pub ingress_machines: Vec<String>,
5067 /// Provider slots, keyed by role (`"static"`, `"compute"`, …). Each value
5068 /// either references a provider declared under `.yah/infra/providers/` or
5069 /// inlines a local-only provider (no creds, no infra file).
5070 ///
5071 /// A role is normally service-wide — one slot serves every component that
5072 /// shares it — but [`ReconcileCtx::slot`](crate::reconciler::ReconcileCtx::slot)
5073 /// looks up the component-qualified key `"<role>:<component id>"` first.
5074 /// A service with two components of the same role (e.g. two
5075 /// `mesofact-static` components under one mirror) declares
5076 /// `providers."static:<id>"` per component to give each its own port;
5077 /// omitting the qualifier keeps the pre-existing single-slot behavior.
5078 #[serde(default)]
5079 pub providers: BTreeMap<String, MirrorProviderSlot>,
5080 /// Capability→driver bindings, keyed by **capability** (`"pg"`, `"s3"`, …)
5081 /// rather than by slot role (W265 §Drivers).
5082 ///
5083 /// This is the generalization of [`Self::providers`]: `providers.static` /
5084 /// `providers.object_store` are the special case where the slot name and
5085 /// the capability happen to coincide, and keying by capability is what stops
5086 /// the slot enum growing one arm per tier-specific implementation. A service
5087 /// says "I need pg"; the mirror says which implementation of pg *this tier*
5088 /// uses; the app talks the same wire protocol either way and never forks.
5089 ///
5090 /// ```toml
5091 /// [drivers.pg]
5092 /// kind = "local-pg-dev" # dev — kamaji-supervised loopback postgres
5093 ///
5094 /// [drivers.smtp]
5095 /// kind = "local-mailcrab" # dev/pond — a catcher with a browsable inbox
5096 /// ```
5097 ///
5098 /// Additive in P1: `drivers` lands *alongside* `providers`, and migrating
5099 /// the existing `providers.static` / `providers.object_store` declarations
5100 /// over is a separate pass (W265 §"Open follow-ups"). A mirror that declares
5101 /// neither is unchanged.
5102 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
5103 pub drivers: BTreeMap<String, MirrorProviderSlot>,
5104 /// Per-environment alias overrides for `kind = "static-asset"` components.
5105 ///
5106 /// Keys are logical names (e.g. `"whisper-default"`); values must be
5107 /// filenames present in the component's `workload.toml` catalog.
5108 /// **Resolution only** — this table may never introduce a filename absent
5109 /// from the catalog. Validated against the workload catalog at sync time.
5110 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
5111 pub asset_aliases: BTreeMap<String, String>,
5112 /// Per-environment build overrides, keyed by **component id** (R905).
5113 ///
5114 /// A `[build]` block lives on the component's `workload.toml`, and a
5115 /// component is declared exactly once in `service.toml` — so without this
5116 /// table a service that deploys the same component to two environments
5117 /// builds it identically for both. That is wrong for any bundle whose
5118 /// contents depend on the tier it is being built *for*: noisetable's
5119 /// landing site bakes `NOISETABLE_API_ORIGIN` into the shipped JS, so the
5120 /// staging site was served a production API origin and every call from it
5121 /// was blocked by production CORS.
5122 ///
5123 /// ```toml
5124 /// # .yah/services/noisetable-marketing/mirrors/staging.toml
5125 /// [build.site]
5126 /// command = "bun run build:staging"
5127 ///
5128 /// # or, without a sibling script per environment:
5129 /// [build.site.env]
5130 /// NOISETABLE_API_ORIGIN = "https://api-staging.noisetable.com"
5131 /// ```
5132 ///
5133 /// The environment axis stays on the mirror, where `providers`, `drivers`
5134 /// and `ingress` already live, rather than growing an `env`-keyed table on
5135 /// the component's own `BuildConfig` — a per-component struct is the wrong
5136 /// place to enumerate environments, and doing it there would have made the
5137 /// mirror the *second* per-environment surface instead of the only one.
5138 ///
5139 /// Deliberately NOT overridable here: `out_dir`. Where a bundler writes is
5140 /// a property of the project's own toolchain, not of the tier it is built
5141 /// for, and it is read independently of `[build]` by the publish path
5142 /// (`read_workload_out_dir`, `collect_component_files`) — making it
5143 /// per-environment would mean threading the mirror into every one of those
5144 /// readers to buy a knob no tier needs.
5145 ///
5146 /// Keys are validated against the service's declared component ids at
5147 /// config load (`cross_ref_validate`), so a typo is a refusal rather than
5148 /// an override that silently never fires.
5149 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
5150 pub build: BTreeMap<String, MirrorBuildOverride>,
5151}
5152
5153/// One component's per-environment build override — the value type of
5154/// [`MirrorConfig::build`] (R905).
5155///
5156/// Every field is additive-or-replacing against the component's own
5157/// `workload.toml [build]` table; an empty override is indistinguishable from
5158/// declaring none.
5159#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
5160#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5161#[serde(deny_unknown_fields)]
5162pub struct MirrorBuildOverride {
5163 /// Replaces `build.command` for this environment.
5164 ///
5165 /// Absent leaves the workload's own command in place — including absent,
5166 /// which means "this project has no external bundler step" and must keep
5167 /// meaning that (R838-B1). An override may therefore *introduce* a command
5168 /// where the workload's `[build]` table declares none — a deliberate
5169 /// per-tier opt-in to a bundler, since the only way to write it is to name
5170 /// one. A workload with no `[build]` table at all is still skipped
5171 /// wholesale: there is no `out_dir` to publish from, so an override there
5172 /// would have nothing to hand the publish step.
5173 #[serde(default, skip_serializing_if = "Option::is_none")]
5174 pub command: Option<String>,
5175 /// Replaces `build.render_command` — the data-only re-render
5176 /// (`revalidate_static`). Overridden separately from `command` because the
5177 /// two run at different times against different inputs; a tier that needs
5178 /// a different bundler command usually needs the same renderer.
5179 #[serde(default, skip_serializing_if = "Option::is_none")]
5180 pub render_command: Option<String>,
5181 /// Environment variables exported to the build (and render) subprocess,
5182 /// applied on top of the inherited environment.
5183 ///
5184 /// This is the knob for the common case — the command is the same, only a
5185 /// baked-in origin/flag differs — and it is what keeps a project from
5186 /// having to pre-declare one `build:<env>` script per environment before
5187 /// any environment can exist.
5188 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
5189 pub env: BTreeMap<String, String>,
5190}
5191
5192impl MirrorBuildOverride {
5193 /// `true` when this override would change nothing.
5194 pub fn is_empty(&self) -> bool {
5195 self.command.is_none() && self.render_command.is_none() && self.env.is_empty()
5196 }
5197
5198 /// The override's `env` table as the `Vec<(String, String)>` that both
5199 /// `ExecContext::with_env` and `std::process::Command::envs` want.
5200 pub fn env_pairs(&self) -> Vec<(String, String)> {
5201 self.env
5202 .iter()
5203 .map(|(k, v)| (k.clone(), v.clone()))
5204 .collect()
5205 }
5206}
5207
5208impl MirrorConfig {
5209 /// This environment's build override for `component_id`, if any.
5210 ///
5211 /// Returns `None` for an override that exists but changes nothing, so
5212 /// callers can treat `Some(_)` as "something differs here" without
5213 /// re-checking each field.
5214 pub fn build_override(&self, component_id: &str) -> Option<&MirrorBuildOverride> {
5215 self.build.get(component_id).filter(|o| !o.is_empty())
5216 }
5217}
5218
5219impl MirrorConfig {
5220 /// This mirror's declared edges, with both spellings normalized (W305 F2).
5221 ///
5222 /// The single place `ingress` + `ingress_machines` are reconciled, so no
5223 /// consumer has to know which spelling was written. Returns an empty vec
5224 /// when the mirror declares no front door.
5225 ///
5226 /// Errors are the declarations that cannot mean anything:
5227 ///
5228 /// - `ingress_machines` with no `ingress` — front-door placement with no
5229 /// front door to place, always a typo (R330-F37);
5230 /// - `ingress_machines` alongside `[[ingress]]` — placement declared twice,
5231 /// in a form where one silently wins;
5232 /// - `provider = "none"` on an edge — an edge that fronts with nothing.
5233 /// The `[[ingress]]` entries exactly as written, without normalizing the
5234 /// scalar spelling or validating anything.
5235 ///
5236 /// [`ingress_edges`](Self::ingress_edges) is the one to reach for; this
5237 /// exists for the checks that must run *before* a mirror is known to be
5238 /// well-formed — cross-reference validation walks every mirror in the
5239 /// workspace, and hard-failing there on an unrelated mirror's shape error
5240 /// would report the wrong file. Empty for the scalar spelling, which has no
5241 /// edge table to carry per-edge fields.
5242 pub fn ingress_edge_slice(&self) -> &[IngressEdge] {
5243 match &self.ingress {
5244 IngressDecl::Edges(edges) => edges,
5245 IngressDecl::Provider(_) => &[],
5246 }
5247 }
5248
5249 pub fn ingress_edges(&self) -> Result<Vec<IngressEdge>> {
5250 match &self.ingress {
5251 IngressDecl::Provider(p) if !p.is_declared() => {
5252 if !self.ingress_machines.is_empty() {
5253 bail!(
5254 "mirror declares `ingress_machines = {:?}` but no `ingress` provider — \
5255 front-door placement with no front door to place. Add \
5256 `ingress = \"passway\"` (or \"cloudflare-tunnel\"), or drop \
5257 `ingress_machines`.",
5258 self.ingress_machines
5259 );
5260 }
5261 Ok(Vec::new())
5262 }
5263 IngressDecl::Provider(p) => Ok(vec![IngressEdge::all_slots(
5264 *p,
5265 self.ingress_machines.clone(),
5266 )]),
5267 IngressDecl::Edges(edges) => {
5268 if !self.ingress_machines.is_empty() {
5269 bail!(
5270 "mirror declares both `[[ingress]]` edges and the single-edge \
5271 `ingress_machines = {:?}` — front-door placement stated twice. Move \
5272 those names onto the edge they place: `machines = [...]` inside the \
5273 `[[ingress]]` entry.",
5274 self.ingress_machines
5275 );
5276 }
5277 for edge in edges {
5278 if !edge.provider.is_declared() {
5279 bail!(
5280 "{}: `provider = \"none\"` fronts nothing. An edge exists to name a \
5281 front door — delete the entry instead.",
5282 edge.label()
5283 );
5284 }
5285 edge.validate_auth()?;
5286 edge.validate_via()?;
5287 edge.validate_tunnel_door()?;
5288 }
5289 Ok(edges.clone())
5290 }
5291 }
5292 }
5293
5294 /// Nodes this mirror's **passway** front doors are placed on, in
5295 /// declaration order and de-duplicated — or `None` when the mirror declares
5296 /// no passway edge at all.
5297 ///
5298 /// `Some(vec![])` is a real and different answer from `None`: a passway edge
5299 /// is declared but names no machine, so its placement falls back to the
5300 /// fronted slot's own. That fallback is placement *resolution* — it belongs
5301 /// to [`IngressRule::machines`](crate::reconciler::IngressRule::machines)
5302 /// and the plan it is built from, not to a mirror read in isolation — so it
5303 /// is reported as "declared, placement unknown from here" rather than
5304 /// half-derived. A caller that needs a node to dial has to say so.
5305 ///
5306 /// Passway-only because the caller is tenant DNS onboarding: only a passway
5307 /// node serves yubaba's `GET /domains/{domain}/onboarding`. A
5308 /// cloudflare-tunnel edge publishes through Cloudflare's own DNS and has no
5309 /// such record to hand a tenant, so folding its machines in would point the
5310 /// UI at a node that cannot answer.
5311 ///
5312 /// Read through [`ingress_edges`](Self::ingress_edges), so both spellings
5313 /// are covered by construction. A declaration that cannot mean anything
5314 /// (`ingress_machines` with no `ingress`, or both spellings at once) reads
5315 /// as `None` rather than propagating an error: those are reported by
5316 /// cross-reference validation, which can name the offending file.
5317 pub fn passway_machines(&self) -> Option<Vec<String>> {
5318 let edges = self.ingress_edges().ok()?;
5319 let mut declared = false;
5320 let mut machines: Vec<String> = Vec::new();
5321 for edge in edges
5322 .iter()
5323 .filter(|e| matches!(e.provider, IngressProvider::Passway))
5324 {
5325 declared = true;
5326 for m in &edge.machines {
5327 if !machines.iter().any(|seen| seen == m) {
5328 machines.push(m.clone());
5329 }
5330 }
5331 }
5332 declared.then_some(machines)
5333 }
5334
5335 /// Parse a single `mirrors/<env>.toml` file.
5336 pub fn load(path: &Path) -> Result<Self> {
5337 let src =
5338 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
5339 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
5340 }
5341
5342 /// Persist to `.yah/services/<service>/mirrors/<env>.toml`, creating the
5343 /// `mirrors/` directory if needed. Create-or-overwrite. The mirror file is
5344 /// named by `env` (its stem); `service` selects the owning service dir.
5345 pub fn save(&self, workspace_root: &Path, service: &str, env: &str) -> Result<()> {
5346 let dir = crate::paths::service_mirrors_dir(workspace_root, service);
5347 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
5348 let path = crate::paths::service_mirror_toml(workspace_root, service, env);
5349 let s = toml::to_string_pretty(self)
5350 .with_context(|| format!("serializing mirror {service}/{env}"))?;
5351 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
5352 }
5353
5354 /// Remove `.yah/services/<service>/mirrors/<env>.toml`. Returns `false`
5355 /// when the file was already absent. Leaves the service and its other
5356 /// mirrors untouched.
5357 ///
5358 /// Also checks legacy stems (e.g. `local-sim` when `env = "pond"`) so
5359 /// deleting a canonical tier name removes whichever file exists on disk.
5360 pub fn delete(workspace_root: &Path, service: &str, env: &str) -> Result<bool> {
5361 let path = crate::paths::service_mirror_toml(workspace_root, service, env);
5362 if path.exists() {
5363 std::fs::remove_file(&path).with_context(|| format!("removing {}", path.display()))?;
5364 return Ok(true);
5365 }
5366 // Try legacy file stems for canonical tier names.
5367 let legacy: &[&str] = match env {
5368 "dev" => &["local"],
5369 "pond" => &["local-sim", "sim"],
5370 "prod" => &["cloud"],
5371 _ => &[],
5372 };
5373 for stem in legacy {
5374 let alt = crate::paths::service_mirror_toml(workspace_root, service, stem);
5375 if alt.exists() {
5376 std::fs::remove_file(&alt)
5377 .with_context(|| format!("removing {}", alt.display()))?;
5378 return Ok(true);
5379 }
5380 }
5381 Ok(false)
5382 }
5383}
5384
5385/// A provider slot inside a [`MirrorConfig`]. Two shapes:
5386/// - **Reference** (`use = "<provider-id>"`) — point at an infra-declared
5387/// provider; extra fields are slot-specific (bucket, zone, dns, …).
5388/// - **Inline** (`kind = "local-*"`) — for providers that need no infra
5389/// declaration because they carry no credentials.
5390#[derive(Debug, Clone, Serialize, Deserialize)]
5391#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5392#[serde(untagged)]
5393pub enum MirrorProviderSlot {
5394 Reference {
5395 #[serde(rename = "use")]
5396 provider_id: String,
5397 #[serde(flatten)]
5398 #[cfg_attr(
5399 feature = "json-schema",
5400 schemars(with = "std::collections::BTreeMap<String, serde_json::Value>")
5401 )]
5402 fields: BTreeMap<String, toml::Value>,
5403 },
5404 Inline {
5405 kind: Provider,
5406 #[serde(flatten)]
5407 #[cfg_attr(
5408 feature = "json-schema",
5409 schemars(with = "std::collections::BTreeMap<String, serde_json::Value>")
5410 )]
5411 fields: BTreeMap<String, toml::Value>,
5412 },
5413}
5414
5415impl MirrorProviderSlot {
5416 /// Provider id this slot references, or `None` for inline slots.
5417 pub fn provider_id(&self) -> Option<&str> {
5418 match self {
5419 Self::Reference { provider_id, .. } => Some(provider_id),
5420 Self::Inline { .. } => None,
5421 }
5422 }
5423
5424 /// Provider kind for inline slots, or `None` for reference slots
5425 /// (resolve via the referenced [`ProviderConfig`]).
5426 pub fn inline_kind(&self) -> Option<Provider> {
5427 match self {
5428 Self::Reference { .. } => None,
5429 Self::Inline { kind, .. } => Some(*kind),
5430 }
5431 }
5432
5433 pub fn fields(&self) -> &BTreeMap<String, toml::Value> {
5434 match self {
5435 Self::Reference { fields, .. } | Self::Inline { fields, .. } => fields,
5436 }
5437 }
5438
5439 /// F16 placement: parse the optional `required = { … }` sub-table on this
5440 /// slot. Returns `None` when absent or unparseable (callers treat as no
5441 /// constraint). See [`RequiredSpec`] for the field grammar.
5442 pub fn required(&self) -> Option<RequiredSpec> {
5443 let v = self.fields().get("required")?.clone();
5444 v.try_into().ok()
5445 }
5446}
5447
5448/// F16 placement constraints declared on a [`MirrorProviderSlot`], lives under
5449/// `[providers.<role>] required = { regions = [...], mesh_tags = [...] }` in
5450/// `mirrors/<env>.toml`.
5451///
5452/// Hard (must-satisfy) axes, all AND-ed together:
5453/// - `regions` / `zones` / `providers` — *membership*: the machine's
5454/// `region` / `zone` / `provider` must be one of the listed values.
5455/// - `mesh_tags` — *superset*: the machine's `mesh_tags` must contain every
5456/// listed tag.
5457/// - `memory_mb` / `cpu_millis` — *capacity floor* (R572-F5): the machine's
5458/// `allocatable` budget must cover the demand. `0` = no constraint.
5459/// - *taint repulsion* — **unconditional** (R876-B7): the machine must not
5460/// carry any taint that [`taint_effect`] classifies as
5461/// [`TaintEffect::Repels`], unless that exact key is listed in
5462/// [`Self::tolerates`]. This axis is not declared; it applies to every spec.
5463/// - `requires_taint` — *taint affinity* (R572-F5): the machine must carry
5464/// this taint key (in `taints` or `mesh_tags`). `None` = no affinity.
5465///
5466/// These are the **only** readers of [`MachineConfig::taints`], which is
5467/// what makes [`taint_effect`]'s closed vocabulary well-founded.
5468///
5469/// [`MachineConfig::sovereign_group`] is deliberately **not** an axis here and
5470/// must not become one (W305/R742-F1). A sovereign group is a blast radius,
5471/// not a filter: which quorum a box votes in says nothing about whether a
5472/// workload may run on it, and a dev-group node exists precisely so dev-mode
5473/// services — stateful ones included — can be scheduled onto it. Filtering on
5474/// it would re-make the mistake W305 exists to undo, where one mechanism
5475/// silently carried three unrelated properties.
5476///
5477/// An empty / zero / None on every axis means "no constraint on that axis".
5478/// A fully-unconstrained `RequiredSpec` matches every machine (see
5479/// [`RequiredSpec::is_unconstrained`]).
5480#[derive(Debug, Clone, Default, Serialize, Deserialize)]
5481#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5482pub struct RequiredSpec {
5483 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5484 pub regions: Vec<String>,
5485 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5486 pub zones: Vec<String>,
5487 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5488 pub providers: Vec<String>,
5489 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5490 pub mesh_tags: Vec<String>,
5491
5492 /// R833-F8: **imperative** placement — the machine must be one of these by
5493 /// `name`. Empty (the default) = no constraint, which is every pre-R833-F8
5494 /// caller.
5495 ///
5496 /// This is the one axis that is not a *capability* the scheduler infers.
5497 /// The operator typed `--where=node:us-west-003`, so it composes with the
5498 /// other axes exactly like the rest — a named node that fails the capacity
5499 /// floor or carries a repelling taint still does not match, and the refusal
5500 /// names why rather than silently placing the work somewhere else.
5501 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5502 pub nodes: Vec<String>,
5503
5504 /// R572-F5: minimum memory (MiB) the target node must have in its
5505 /// declared `allocatable` budget. `0` = no constraint. Filled by
5506 /// [`CloudConfig::admit_workload`] from the workload's
5507 /// `memory_request_mb()` — its placement **request**, which is not the
5508 /// same number as the `resources.memory_mb` cgroup **ceiling**.
5509 #[serde(default, skip_serializing_if = "is_zero_u32")]
5510 pub memory_mb: u32,
5511 /// R572-F5: minimum CPU (millicores) the target node must have in its
5512 /// declared `allocatable` budget. `0` = no constraint. Filled by
5513 /// [`CloudConfig::admit_workload`] from the workload's `resources.cpu_millis`.
5514 #[serde(default, skip_serializing_if = "is_zero_u32")]
5515 pub cpu_millis: u32,
5516 /// R876-B7: repelling node taints this placement **opts back in to**.
5517 /// Each entry is a machine taint key spelled exactly as it appears in
5518 /// [`MachineConfig::taints`] — `"no-appliance"`, not `"appliance"` — so the
5519 /// node side and the workload side share one vocabulary and nothing has to
5520 /// translate between them.
5521 ///
5522 /// # Why this replaced `repel_archetypes`
5523 ///
5524 /// Repulsion used to be **opt-in-to-be-repelled**: the spec named the
5525 /// archetypes it was, and only a `no-<that archetype>` taint blocked it.
5526 /// That field was `#[serde(skip)]`, so a slot declared as
5527 /// `required = { ... }` in a mirror TOML always deserialized with it empty
5528 /// and [`Self::matches`] never read [`MachineConfig::taints`] at all. Node
5529 /// taints were therefore structurally inert for every mirror-declared
5530 /// placement, and inert *silently* — `no-server` is a legal key, so
5531 /// `yah cloud validate` passed and an operator draining a node before
5532 /// maintenance got a green run and a workload that never moved (R876-S2's
5533 /// drill measured exactly this against the real tree).
5534 ///
5535 /// The sense is now inverted, which is the only shape that can survive a
5536 /// field the wire does not carry: **repulsion is unconditional and
5537 /// toleration is declared.** A spec that says nothing is repelled by every
5538 /// repelling taint — the reading an operator writing `taints = ["no-server"]`
5539 /// on a machine already assumed they were getting.
5540 ///
5541 /// Toleration is per-key and absolute; there is no wildcard. Listing a key
5542 /// no machine declares is harmless and matches nothing.
5543 ///
5544 /// [`admission_spec`] fills this from the placement group's archetypes —
5545 /// every repelling key that is *not* the group's own class — which is what
5546 /// makes the `admit_workload` path behave identically across this change
5547 /// (R860-T4 / W338 §Placement consequences 2 still hold: the group's
5548 /// archetypes are the union over `local` requirement edges, so a `Server`
5549 /// bound to an `Appliance` tolerates neither `no-server` nor
5550 /// `no-appliance`).
5551 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5552 pub tolerates: Vec<String>,
5553 /// R572-F5: taint the workload requires the target node to carry
5554 /// (annotation `yah.placement.requires-taint`). The node must have the
5555 /// key in its `taints` list or `mesh_tags`. `None` = no affinity constraint.
5556 #[serde(skip)]
5557 pub requires_taint: Option<String>,
5558
5559 /// R844-F8: **how many** machines this constraint places onto. `None` — the
5560 /// only shape on disk before this field — means one, so every mirror in the
5561 /// tree resolves byte-identically across the change.
5562 ///
5563 /// This is not a match axis: it never appears in [`Self::matches`] and never
5564 /// changes whether a given machine qualifies. It is the *cardinality* of the
5565 /// answer, which is why it lives here rather than as another filter — the
5566 /// operator declares what is required and how many of it, and the scheduler
5567 /// picks which.
5568 ///
5569 /// **Declared, never inferred.** The count is emphatically not "how many
5570 /// machines happen to match": deriving it that way would make adding a box
5571 /// to the fleet silently scale a production front door. A constraint that
5572 /// matches four machines and asks for two places on two.
5573 ///
5574 /// **Fewer matches than asked is an error** ([`select_matching`]), not a
5575 /// partial placement. Placing one of two and reporting success is the
5576 /// subset-that-looks-like-it-worked failure R844 exists to close.
5577 ///
5578 /// Deliberately absent from [`Self::is_unconstrained`], which answers "does
5579 /// every machine match" — a question about the predicate, not the count. A
5580 /// `required = { replicas = 2 }` with no axis is therefore still
5581 /// unconstrained, and the deploy side still refuses it as an
5582 /// underspecified placement.
5583 #[serde(default, skip_serializing_if = "Option::is_none")]
5584 pub replicas: Option<u32>,
5585}
5586
5587impl RequiredSpec {
5588 /// How many machines this constraint places onto — [`Self::replicas`],
5589 /// resolving the absent case to the pre-R844-F8 answer of one.
5590 ///
5591 /// The single place that default is spelled, so the ingress planner and the
5592 /// deploy resolver cannot disagree about what "no replica count" means.
5593 pub fn replica_count(&self) -> usize {
5594 self.replicas.unwrap_or(1) as usize
5595 }
5596
5597 /// True when no *declared* axis carries a constraint — every untainted
5598 /// machine matches.
5599 ///
5600 /// R876-B7: taint repulsion is deliberately absent from this conjunction,
5601 /// unlike the `repel_archetypes` it replaced. Repulsion is no longer an axis
5602 /// a spec declares — it applies to every spec — so including it would make
5603 /// the answer a property of the fleet rather than of the constraint. Nor
5604 /// does [`Self::tolerates`] belong here: a toleration *widens* the candidate
5605 /// set, and the callers of this predicate ask "did the operator narrow
5606 /// anything" in order to refuse an underspecified placement. A slot that
5607 /// declares only a toleration has still narrowed nothing.
5608 pub fn is_unconstrained(&self) -> bool {
5609 self.regions.is_empty()
5610 && self.zones.is_empty()
5611 && self.providers.is_empty()
5612 && self.mesh_tags.is_empty()
5613 && self.nodes.is_empty()
5614 && self.memory_mb == 0
5615 && self.cpu_millis == 0
5616 && self.requires_taint.is_none()
5617 }
5618
5619 /// Whether `machine` satisfies every hard axis.
5620 ///
5621 /// - Membership axes (region/zone/provider): machine must carry the field
5622 /// and it must appear in the constraint list.
5623 /// - `mesh_tags`: machine tags must be a superset of the required set.
5624 /// - **R572-F5 capacity floor**: `machine.allocatable.{memory,cpu}` must
5625 /// cover `self.{memory,cpu}`. A machine with no `allocatable` block passes
5626 /// unconditionally (capacity unknown → no constraint enforced).
5627 /// - **Taint repulsion (R876-B7)**: machine must not carry *any* taint that
5628 /// [`taint_effect`] classifies as [`TaintEffect::Repels`], unless that key
5629 /// is listed in [`Self::tolerates`]. Applied unconditionally — this is the
5630 /// axis no spec has to declare, and the one that makes a node drainable.
5631 /// - **R572-F5 taint affinity**: if `requires_taint` is set, the machine
5632 /// must carry that key in its `taints` list or `mesh_tags`.
5633 ///
5634 /// A [`TaintEffect::Attracts`] key (today just `public-ip`) does **not**
5635 /// repel: it is the affinity vocabulary, so reading it as repulsion would
5636 /// evict every workload from the three nodes that carry it. Only the
5637 /// `no-<archetype>` class repels, and [`taint_effect`] is the single
5638 /// authority on which is which — which is why
5639 /// [`crate::validate::check_inert_taints`] refuses to let an unclassifiable
5640 /// key be declared: it would read as a constraint and be none.
5641 pub fn matches(&self, machine: &MachineConfig) -> bool {
5642 let member_ok = |constraint: &[String], value: Option<&str>| -> bool {
5643 constraint.is_empty() || value.map_or(false, |v| constraint.iter().any(|c| c == v))
5644 };
5645
5646 // R833-F8: imperative node pin, checked first because it is the axis a
5647 // human asserted rather than one the scheduler derived — a refusal
5648 // should read "us-west-003 does not match" and not lead with a tag set
5649 // the operator never typed.
5650 if !member_ok(&self.nodes, Some(machine.name.as_str())) {
5651 return false;
5652 }
5653
5654 // Membership + mesh-tags (pre-existing axes).
5655 if !member_ok(&self.regions, machine.region.as_deref())
5656 || !member_ok(&self.zones, machine.zone.as_deref())
5657 || !member_ok(&self.providers, Some(machine.provider.as_str()))
5658 || !self
5659 .mesh_tags
5660 .iter()
5661 .all(|t| machine.mesh_tags.iter().any(|mt| mt == t))
5662 {
5663 return false;
5664 }
5665
5666 // R572-F5: capacity floor. Skipped when machine has no allocatable
5667 // declaration (unknown capacity → passes, consistent with pre-F5 behaviour).
5668 if self.memory_mb > 0 || self.cpu_millis > 0 {
5669 if let Some(alloc) = &machine.allocatable {
5670 if self.memory_mb > alloc.memory_mb || self.cpu_millis > alloc.cpu_millis {
5671 return false;
5672 }
5673 }
5674 }
5675
5676 // R876-B7: taint repulsion, repel-by-default. Every repelling taint on
5677 // the machine blocks placement unless this spec names it in
5678 // `tolerates`. Driven off `machine.taints` rather than off a field of
5679 // `self`, which is the whole point: a spec that arrives by deserializing
5680 // a mirror's `required = {...}` carries no repulsion declaration and
5681 // never could, so making repulsion conditional on one made node taints
5682 // structurally unreadable on that path (R876-S2).
5683 for taint in &machine.taints {
5684 if !matches!(taint_effect(taint), TaintEffect::Repels(_)) {
5685 continue;
5686 }
5687 if !self.tolerates.iter().any(|t| t == taint) {
5688 return false;
5689 }
5690 }
5691
5692 // R572-F5: taint affinity. Machine must carry the required taint key
5693 // in either its `taints` list or `mesh_tags`.
5694 if let Some(req) = &self.requires_taint {
5695 let has_it = machine.taints.iter().any(|t| t == req)
5696 || machine.mesh_tags.iter().any(|t| t == req);
5697 if !has_it {
5698 return false;
5699 }
5700 }
5701
5702 true
5703 }
5704
5705 /// Human-readable summary of the constraints, for fail-loud error messages.
5706 /// Example: `required.regions=[us-west] + required.mesh_tags=[tag:cloud-runner]`.
5707 pub fn describe(&self) -> String {
5708 let mut parts = Vec::new();
5709 let mut push = |label: &str, vals: &[String]| {
5710 if !vals.is_empty() {
5711 parts.push(format!("required.{label}=[{}]", vals.join(",")));
5712 }
5713 };
5714 push("nodes", &self.nodes);
5715 push("regions", &self.regions);
5716 push("zones", &self.zones);
5717 push("providers", &self.providers);
5718 push("mesh_tags", &self.mesh_tags);
5719 // Kept with the other list axes, and NOT moved below: `push` borrows
5720 // `parts` mutably for as long as it is live, so interleaving it with the
5721 // direct `parts.push` calls under it does not compile.
5722 push("tolerates", &self.tolerates);
5723 if self.memory_mb > 0 {
5724 parts.push(format!("memory_mb>={}", self.memory_mb));
5725 }
5726 if self.cpu_millis > 0 {
5727 parts.push(format!("cpu_millis>={}", self.cpu_millis));
5728 }
5729 if let Some(req) = &self.requires_taint {
5730 parts.push(format!("requires_taint={req}"));
5731 }
5732 if parts.is_empty() {
5733 "no constraints".to_string()
5734 } else {
5735 parts.join(" + ")
5736 }
5737 }
5738}
5739
5740/// Which front door actually serves a domain's requests (R594-F12).
5741///
5742/// Every domain manifest must say this out loud. Before it existed the
5743/// difference between "R2 serves this hostname directly" and "a Worker
5744/// serves it" was expressed *only* by whether the file happened to carry
5745/// `[[routes]]` — so binding a route-carrying domain straight to R2 was
5746/// accepted silently and served 200s on its SSG half while losing clean
5747/// URLs, SPA shell fallback, deferred-route pointers and branded error
5748/// pages. All of those live in the Worker
5749/// (`oss/mesofact/packages/mesofact-edge/src/router.ts`) or in
5750/// mesofact-serve; an R2 custom domain has none of them.
5751///
5752/// The vocabulary mirrors `scripts/cf-apex-mode.sh` (worker | grey | orange)
5753/// — this moves the choice into the config where it can be checked instead
5754/// of living in one bash script.
5755///
5756/// A front door does **fan-in** only. The render cube (SSG / SPA / SSR /
5757/// deferred / 404) is mesofact's manifest, not this one — see W173 and
5758/// `.yah/docs/working/W267-sovereign-public-ingress.md`
5759/// §"Two front doors, one render contract".
5760#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
5761#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5762#[serde(rename_all = "kebab-case")]
5763pub enum FrontDoor {
5764 /// Cloudflare R2 custom domain. Requests hit R2 objects with edge
5765 /// caching and nothing else — no clean URLs, no SPA fallback, no
5766 /// branded errors. Correct for a pure asset tier (W175's verdict for
5767 /// `cdn.yah.dev`) and wrong for anything that renders pages.
5768 /// Implies zero `[[routes]]` and no `worker_bundle_path`.
5769 BucketDirect,
5770 /// Cloudflare Worker generated from this manifest's route table.
5771 Worker,
5772 /// Sovereign L7 ingress — the `passway` proxy on yah-owned metal
5773 /// (`oss/passway`, W267). Same route table as `worker`; different
5774 /// machine terminates TLS.
5775 Passway,
5776}
5777
5778impl FrontDoor {
5779 /// Whether this front door consumes the manifest's `[[routes]]` table.
5780 /// `bucket-direct` does not; the other two are nothing without it.
5781 pub fn is_route_driven(self) -> bool {
5782 matches!(self, FrontDoor::Worker | FrontDoor::Passway)
5783 }
5784
5785 /// The manifest spelling, for error messages.
5786 pub fn as_str(self) -> &'static str {
5787 match self {
5788 FrontDoor::BucketDirect => "bucket-direct",
5789 FrontDoor::Worker => "worker",
5790 FrontDoor::Passway => "passway",
5791 }
5792 }
5793}
5794
5795/// A routing manifest for one domain, from `.yah/domains/<name>.toml`.
5796///
5797/// The domain manifest is the *only* place that knows about path routing:
5798/// services declare static/backend components by opaque ID, and this
5799/// manifest binds those components to URL paths on a public-facing
5800/// domain. Generated Worker bundles consume this. See
5801/// `.yah/docs/working/W118-yah-domain-tiers.md` (R347).
5802#[derive(Debug, Clone, Serialize, Deserialize)]
5803#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5804pub struct DomainConfig {
5805 pub schema_version: u32,
5806 /// Stable identifier for this domain (file stem of the manifest).
5807 /// Example: `"yah-dev"` for the `yah.dev` zone.
5808 pub name: String,
5809 /// The fully-qualified domain this manifest routes for. Example:
5810 /// `"yah.dev"`, `"app.yah.dev"`.
5811 pub domain: String,
5812 /// Which front door serves this domain (R594-F12). **Required** — a
5813 /// default here would silently re-create the defect the field exists to
5814 /// close. Cross-checked against `routes` / `worker_bundle_path` by
5815 /// [`DomainConfig::validate_front_door`] at load time.
5816 pub front_door: FrontDoor,
5817 /// Public CDN bucket name. Static-mode route components publish into
5818 /// this bucket. Owned by the domain, *not* by any single service.
5819 pub cdn_bucket: String,
5820 /// Optional path (relative to workspace root) where the generated
5821 /// Worker bundle lands. `None` while the bundle generator (R347-F4)
5822 /// is still being wired up.
5823 #[serde(default, skip_serializing_if = "Option::is_none")]
5824 pub worker_bundle_path: Option<String>,
5825 #[serde(default, skip_serializing_if = "Vec::is_empty")]
5826 pub routes: Vec<DomainRoute>,
5827}
5828
5829/// One entry in a [`DomainConfig`]'s route table.
5830///
5831/// The `mode` discriminator picks the variant's body via serde's
5832/// internally-tagged enum representation. Path patterns follow the
5833/// Worker convention: a trailing `*` matches everything underneath.
5834#[derive(Debug, Clone, Serialize, Deserialize)]
5835#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5836pub struct DomainRoute {
5837 /// URL pattern this route matches. Examples: `"/"`, `"/dashboard/*"`,
5838 /// `"/camp/ws"`.
5839 pub path: String,
5840 /// Response headers the front door sets on every response served under
5841 /// this route (R746). Empty by default.
5842 ///
5843 /// This is the manifest's answer to "who decides a path's response
5844 /// headers". Before it existed the answer was *nobody*: a `_headers` file
5845 /// is a Cloudflare Pages / Netlify convention, and neither of this
5846 /// repo's front doors reads one — a Worker returns what it fetched from
5847 /// R2, and R2 serves only the object's own httpMetadata. So a site could
5848 /// carry a `_headers` file declaring COOP/COEP and ship without them,
5849 /// which is exactly how it was found: `SharedArrayBuffer` is simply
5850 /// absent in a document served cross-origin-isolation-free, with no
5851 /// error anywhere to say why.
5852 ///
5853 /// Deliberately a free-form `name -> value` map rather than named fields
5854 /// for the isolation headers: the domain manifest has no business
5855 /// knowing which headers a route's payload happens to need. Ordering
5856 /// follows the route table's own rule — first matching route wins, no
5857 /// merging across routes (see the Worker's `applyRouteHeaders`).
5858 #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
5859 pub headers: BTreeMap<String, String>,
5860 #[serde(flatten)]
5861 pub mode: RouteMode,
5862}
5863
5864/// Body of a [`DomainRoute`]. Three modes on the wire, four shapes here:
5865/// - **Static** (`mode = "static"`, `component = …`) — the door serves a
5866/// published component's bytes from its CDN prefix. Component ref points at
5867/// a `kind = "mesofact-static"` (or similar) service component.
5868/// - **StaticBucket** (`mode = "static"`, `bucket = …`, R560-F13) — the door
5869/// serves an R2 bucket's ROOT, keyed by the request path minus its leading
5870/// slash with nothing stripped or prepended. For a CDN whose route prefixes
5871/// ARE its buckets' top-level key prefixes (cdn.noisetable.com), where xlb
5872/// derives a blob's key from the very URL path it serves at, so path == key
5873/// is an invariant rather than a convention.
5874/// - **Backend** — Worker proxies to an HTTP origin owned by a backend
5875/// component (yubaba workload, gateway, etc.).
5876/// - **Redirect** — Worker emits a 30x to the target URL. Used to keep
5877/// old paths alive during domain refactors.
5878///
5879/// The two static shapes are separate variants rather than one variant with
5880/// two optional fields, so "both" and "neither" have no spelling in Rust. The
5881/// TOML keeps one `mode = "static"` and the choice between `component` and
5882/// `bucket` is refused at parse time unless exactly one is present — see
5883/// [`RouteModeWire`].
5884#[derive(Debug, Clone, Serialize, Deserialize)]
5885#[serde(try_from = "RouteModeWire", into = "RouteModeWire")]
5886pub enum RouteMode {
5887 Static {
5888 /// Component reference `"<service>/<component-id>"`. Validated
5889 /// at [`CloudConfig::load`] time.
5890 component: String,
5891 },
5892 StaticBucket {
5893 /// R2 bucket name. Validated against R2's naming rule at parse time,
5894 /// which is also what makes [`crate::route_table::r2_binding_name`]
5895 /// injective.
5896 bucket: String,
5897 },
5898 Backend {
5899 /// Component reference `"<service>/<component-id>"`. Validated
5900 /// at [`CloudConfig::load`] time.
5901 component: String,
5902 /// Origin URL the Worker `fetch()`es. Schema-permissive — could
5903 /// be `https://...`, `wss://...`, or a yah-internal mesh URL
5904 /// resolved by yubaba.
5905 origin: String,
5906 /// The path prefix this route's `path` becomes at the origin, when the
5907 /// two differ (R898-F3). `/api/issues*` with `origin_path = "/issues"`
5908 /// proxies `/api/issues/42` to `<origin>/issues/42`.
5909 ///
5910 /// Absent means an identity proxy — the public path reaches the origin
5911 /// unchanged. It is declared here rather than derived because it is a
5912 /// fact about the *upstream's* path layout, which this repo does not
5913 /// own: the issue tracker serves `/issues` and the almanac serves
5914 /// `/releases`, and both are live contracts that predate the domain
5915 /// manifest. Compiled into [`RouteRewrite`](crate::route_table::RouteRewrite)
5916 /// by [`DomainConfig::route_table`](crate::route_table).
5917 #[serde(default, skip_serializing_if = "Option::is_none")]
5918 origin_path: Option<String>,
5919 },
5920 Redirect {
5921 /// Absolute URL or path the Worker emits a 30x to.
5922 target: String,
5923 /// HTTP status code. Defaults to 308 (permanent + method-preserving)
5924 /// so deprecations don't silently turn POSTs into GETs.
5925 #[serde(default = "default_redirect_status")]
5926 status: u16,
5927 },
5928}
5929
5930fn default_redirect_status() -> u16 {
5931 308
5932}
5933
5934/// The TOML/JSON shape of a [`RouteMode`]: one `mode = "static"` whose body
5935/// names a `component` OR a `bucket`.
5936///
5937/// Private to parsing. The two optional fields exist only here, and
5938/// `TryFrom` turns them into exactly one [`RouteMode`] variant or refuses the
5939/// manifest naming both keys — so the contradictory route never reaches a
5940/// consumer. It is also the JSON schema's source, since the schema describes
5941/// what an author writes.
5942#[derive(Debug, Clone, Serialize, Deserialize)]
5943#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
5944#[serde(tag = "mode", rename_all = "kebab-case")]
5945enum RouteModeWire {
5946 Static {
5947 /// Component reference `"<service>/<component-id>"`: serve that
5948 /// component's published bytes. Exactly one of `component` / `bucket`.
5949 #[serde(default, skip_serializing_if = "Option::is_none")]
5950 component: Option<String>,
5951 /// R2 bucket name: serve the bucket ROOT, the key being the request
5952 /// path minus its leading slash with nothing stripped or prepended
5953 /// (R560-F13). Requires `front_door = "worker"`. Exactly one of
5954 /// `component` / `bucket`.
5955 #[serde(default, skip_serializing_if = "Option::is_none")]
5956 bucket: Option<String>,
5957 },
5958 // Field docs below are the schema's text; they restate [`RouteMode`]'s.
5959 Backend {
5960 /// Component reference `"<service>/<component-id>"`. Validated
5961 /// at [`CloudConfig::load`] time.
5962 component: String,
5963 /// Origin URL the Worker `fetch()`es. Schema-permissive — could
5964 /// be `https://...`, `wss://...`, or a yah-internal mesh URL
5965 /// resolved by yubaba.
5966 origin: String,
5967 /// The path prefix this route's `path` becomes at the origin, when the
5968 /// two differ (R898-F3). `/api/issues*` with `origin_path = "/issues"`
5969 /// proxies `/api/issues/42` to `<origin>/issues/42`.
5970 ///
5971 /// Absent means an identity proxy — the public path reaches the origin
5972 /// unchanged. It is declared here rather than derived because it is a
5973 /// fact about the *upstream's* path layout, which this repo does not
5974 /// own: the issue tracker serves `/issues` and the almanac serves
5975 /// `/releases`, and both are live contracts that predate the domain
5976 /// manifest. Compiled into [`RouteRewrite`](crate::route_table::RouteRewrite)
5977 /// by [`DomainConfig::route_table`](crate::route_table).
5978 #[serde(default, skip_serializing_if = "Option::is_none")]
5979 origin_path: Option<String>,
5980 },
5981 Redirect {
5982 /// Absolute URL or path the Worker emits a 30x to.
5983 target: String,
5984 /// HTTP status code. Defaults to 308 (permanent + method-preserving)
5985 /// so deprecations don't silently turn POSTs into GETs.
5986 #[serde(default = "default_redirect_status")]
5987 status: u16,
5988 },
5989}
5990
5991impl TryFrom<RouteModeWire> for RouteMode {
5992 type Error = String;
5993
5994 fn try_from(wire: RouteModeWire) -> std::result::Result<Self, String> {
5995 Ok(match wire {
5996 RouteModeWire::Static {
5997 component: Some(component),
5998 bucket: None,
5999 } => Self::Static { component },
6000 RouteModeWire::Static {
6001 component: None,
6002 bucket: Some(bucket),
6003 } => {
6004 validate_r2_bucket_name(&bucket)?;
6005 Self::StaticBucket { bucket }
6006 }
6007 RouteModeWire::Static {
6008 component: Some(component),
6009 bucket: Some(bucket),
6010 } => {
6011 return Err(format!(
6012 "a static route declares both component = \"{component}\" and bucket = \
6013 \"{bucket}\" — declare exactly one: `component` serves a published \
6014 component from its CDN prefix, `bucket` serves an R2 bucket root with \
6015 the request path as the key"
6016 ))
6017 }
6018 RouteModeWire::Static {
6019 component: None,
6020 bucket: None,
6021 } => {
6022 return Err("a static route declares neither `component` nor `bucket` — \
6023 declare exactly one"
6024 .to_string())
6025 }
6026 RouteModeWire::Backend {
6027 component,
6028 origin,
6029 origin_path,
6030 } => Self::Backend {
6031 component,
6032 origin,
6033 origin_path,
6034 },
6035 RouteModeWire::Redirect { target, status } => Self::Redirect { target, status },
6036 })
6037 }
6038}
6039
6040impl From<RouteMode> for RouteModeWire {
6041 fn from(mode: RouteMode) -> Self {
6042 match mode {
6043 RouteMode::Static { component } => Self::Static {
6044 component: Some(component),
6045 bucket: None,
6046 },
6047 RouteMode::StaticBucket { bucket } => Self::Static {
6048 component: None,
6049 bucket: Some(bucket),
6050 },
6051 RouteMode::Backend {
6052 component,
6053 origin,
6054 origin_path,
6055 } => Self::Backend {
6056 component,
6057 origin,
6058 origin_path,
6059 },
6060 RouteMode::Redirect { target, status } => Self::Redirect { target, status },
6061 }
6062 }
6063}
6064
6065// Hand-written only because schemars 0.8 does not follow `#[serde(try_from)]`:
6066// a derive would describe the Rust variants (`static-bucket`), which no
6067// manifest may spell. The schema is the wire shape an author writes.
6068#[cfg(feature = "json-schema")]
6069impl schemars::JsonSchema for RouteMode {
6070 fn schema_name() -> String {
6071 "RouteMode".to_string()
6072 }
6073
6074 fn json_schema(gen: &mut schemars::gen::SchemaGenerator) -> schemars::schema::Schema {
6075 <RouteModeWire as schemars::JsonSchema>::json_schema(gen)
6076 }
6077}
6078
6079/// R2's bucket naming rule: 3–63 characters of lowercase letters, digits and
6080/// hyphens, starting and ending with a letter or digit.
6081///
6082/// Checked at parse time rather than left to the deploy's 400, and load-bearing
6083/// beyond that: the Worker binding name derived from a bucket
6084/// ([`crate::route_table::r2_binding_name`]) only maps `-` to `_`, which is
6085/// collision-free exactly because a bucket name can carry no `_` of its own.
6086fn validate_r2_bucket_name(bucket: &str) -> std::result::Result<(), String> {
6087 let valid_chars = bucket
6088 .bytes()
6089 .all(|b| b.is_ascii_lowercase() || b.is_ascii_digit() || b == b'-');
6090 let valid_ends = bucket.starts_with(|c: char| c.is_ascii_alphanumeric())
6091 && bucket.ends_with(|c: char| c.is_ascii_alphanumeric());
6092 if (3..=63).contains(&bucket.len()) && valid_chars && valid_ends {
6093 Ok(())
6094 } else {
6095 Err(format!(
6096 "bucket = \"{bucket}\" is not an R2 bucket name — 3 to 63 characters of \
6097 lowercase letters, digits and hyphens, starting and ending with a letter or digit"
6098 ))
6099 }
6100}
6101
6102/// Normalize a component `mount` to a storage/URL key prefix: strip the
6103/// surrounding slashes. `"/app"`, `"app/"`, `"/app/"` → `"app"`; `"/"`, `""`
6104/// → `""` (the service root).
6105///
6106/// One producer on purpose — the publisher's key prefix, the route-path
6107/// cross-check and the front door's key lookup must all agree on what `/app`
6108/// means down to the byte, and three copies of `trim_matches('/')` is how they
6109/// stop agreeing.
6110pub fn normalize_mount(raw: &str) -> String {
6111 raw.trim_matches('/').to_string()
6112}
6113
6114/// The key prefix a domain route pattern serves under: `"/*"` → `""`,
6115/// `"/app/*"` and `"/app"` → `"app"`. The twin of [`normalize_mount`] on the
6116/// routing side.
6117pub fn route_path_prefix(path: &str) -> String {
6118 normalize_mount(path.strip_suffix('*').unwrap_or(path))
6119}
6120
6121/// The route-driven domain whose route table binds a component of `service`,
6122/// if any. Used by static publishers to pick up the per-route response
6123/// headers a service's paths were declared with.
6124///
6125/// Deterministic by `BTreeMap` key order when more than one domain routes the
6126/// same service (a legitimate shape: an apex and a staging host serving one
6127/// bundle). Returning the first is a real limitation, not a considered
6128/// choice — the day two such domains want *different* headers for one
6129/// component, this needs the domain identity threaded in rather than inferred.
6130pub fn domain_serving_service<'a>(
6131 domains: &'a BTreeMap<String, DomainConfig>,
6132 service: &str,
6133) -> Option<&'a DomainConfig> {
6134 domains
6135 .values()
6136 .find(|d| d.front_door.is_route_driven() && d.serves_service(service))
6137}
6138
6139/// The `ROUTE_HEADERS` Worker-binding value for `service`, read from the
6140/// workspace's domain manifests. `"[]"` when no route-driven domain routes the
6141/// service, or when the one that does declares no headers.
6142///
6143/// R898-F1 widened this into
6144/// [`route_table_for_service`](crate::route_table::route_table_for_service):
6145/// same lookup, same `.yah/domains/` read, the whole compiled table
6146/// (`{path, mode, resolved origin, headers, auth}`) instead of its header
6147/// column. This spelling survives because it is what the *bindings* carry until
6148/// R898-F2/F3 widen those consumers — and because it needs no placement, which
6149/// the reconcilers calling it do not have.
6150///
6151/// Reads `.yah/domains/` directly rather than taking a loaded [`CloudConfig`]:
6152/// the static reconcilers are handed a per-component [`ReconcileCtx`], not the
6153/// whole workspace config, and threading a config reference through all 22 of
6154/// its construction sites to reach one string would be a wide change for a
6155/// narrow read. Manifest parse errors propagate — a domain file that no longer
6156/// loads is a deploy-stopping fact, not a reason to ship a Worker with the
6157/// headers quietly missing.
6158pub fn route_headers_for_service(workspace_root: &Path, service: &str) -> Result<String> {
6159 Ok(domain_for_service(workspace_root, service)?
6160 .as_ref()
6161 .map(DomainConfig::route_headers_json)
6162 .unwrap_or_else(|| "[]".to_string()))
6163}
6164
6165/// The route-driven domain manifest serving `service`, loaded from
6166/// `.yah/domains/`.
6167///
6168/// The whole manifest rather than one projection of it, for the consumer that
6169/// needs more than the header column: R898-F3's Worker reconciler compiles the
6170/// full route table (`{path, mode, origin, rewrite, headers, auth}`) and cannot
6171/// re-derive modes and origins from `route_headers_json`'s output. Same lookup
6172/// [`route_headers_for_service`] makes — it is now a caller of this.
6173pub fn domain_for_service(workspace_root: &Path, service: &str) -> Result<Option<DomainConfig>> {
6174 let domains = load_domains(&crate::paths::domains_dir(workspace_root))?;
6175 Ok(domain_serving_service(&domains, service).cloned())
6176}
6177
6178impl DomainConfig {
6179 /// Parse a single `.yah/domains/<name>.toml`, rejecting a manifest whose
6180 /// declared front door contradicts its route table
6181 /// ([`Self::validate_front_door`]).
6182 pub fn load(path: &Path) -> Result<Self> {
6183 let src =
6184 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
6185 let dom: Self =
6186 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
6187 dom.validate_front_door()
6188 .with_context(|| format!("validating {}", path.display()))?;
6189 dom.validate_route_headers()
6190 .with_context(|| format!("validating {}", path.display()))?;
6191 Ok(dom)
6192 }
6193
6194 /// R594-F12 — the front door must agree with the rest of the manifest.
6195 ///
6196 /// - `bucket-direct` is an R2 custom domain: a Worker route table would
6197 /// never be consulted, so declaring one means the author expected
6198 /// Worker behaviour (clean URLs, SPA fallback, branded errors) from a
6199 /// surface that cannot provide it. Rejected rather than silently
6200 /// ignored. Same for `worker_bundle_path` — nothing would deploy it.
6201 /// - `worker` / `passway` with an empty route table is a silent 404
6202 /// machine: the front door exists, has nothing to serve, and every
6203 /// request falls through to the catch-all.
6204 ///
6205 /// Called from [`Self::load`], so both [`CloudConfig::load`] and
6206 /// [`CloudConfig::load_from_config_dir`] enforce it.
6207 pub fn validate_front_door(&self) -> Result<()> {
6208 match self.front_door {
6209 FrontDoor::BucketDirect => {
6210 if let Some(route) = self.routes.first() {
6211 anyhow::bail!(
6212 "front_door = \"bucket-direct\" but routes[0].path = \"{}\" — \
6213 an R2 custom domain never consults a route table, so this \
6214 route would silently do nothing (no clean URLs, no SPA \
6215 fallback, no branded errors). Set front_door = \"worker\" \
6216 (or \"passway\") to keep the routes, or drop the [[routes]] \
6217 to keep the bucket-direct binding.",
6218 route.path
6219 );
6220 }
6221 if let Some(path) = &self.worker_bundle_path {
6222 anyhow::bail!(
6223 "front_door = \"bucket-direct\" but worker_bundle_path = \
6224 \"{path}\" — nothing deploys a Worker bundle for a domain \
6225 bound straight to R2"
6226 );
6227 }
6228 }
6229 FrontDoor::Worker | FrontDoor::Passway => {
6230 // R560-F13: a bucket route is read through a Worker R2 binding.
6231 // Passway has no R2 read path, so accepting one there would
6232 // compile an entry the door cannot serve — refused naming the
6233 // route instead.
6234 if self.front_door == FrontDoor::Passway {
6235 if let Some(route) = self
6236 .routes
6237 .iter()
6238 .find(|r| matches!(r.mode, RouteMode::StaticBucket { .. }))
6239 {
6240 anyhow::bail!(
6241 "front_door = \"passway\" but routes path = \"{}\" declares \
6242 `bucket` — a bucket route is served through a Cloudflare Worker \
6243 R2 binding, and passway has no R2 read path. Set front_door = \
6244 \"worker\", or serve the path from a published `component`.",
6245 route.path
6246 );
6247 }
6248 }
6249 if self.routes.is_empty() {
6250 anyhow::bail!(
6251 "front_door = \"{}\" but [[routes]] is empty — a front door \
6252 with no route table is a silent 404 machine. Declare at \
6253 least one route, or set front_door = \"bucket-direct\" if \
6254 this domain really is served straight from R2.",
6255 self.front_door.as_str()
6256 );
6257 }
6258 }
6259 }
6260 Ok(())
6261 }
6262
6263 // `route_headers_json` — the header column of the compiled route table —
6264 // lives in `crate::route_table` alongside `route_table`, `RouteTable` and
6265 // the one serializer both projections share (R898-F1).
6266
6267 /// R749-T5 — everything [`Self::route_headers_json`] emits must be
6268 /// *applicable*, checked here where the table is PRODUCED.
6269 ///
6270 /// That method serializes a typed struct, so the table's JSON *shape* is
6271 /// sound by construction. Its contents are not: a route's `headers` map is
6272 /// a free-form `name -> value` read verbatim out of hand-written TOML, so
6273 /// `"Cross Origin Opener Policy"` (spaces instead of hyphens) or a value
6274 /// carrying a newline ships a structurally-valid table that neither front
6275 /// door can apply — and they fail *differently*, neither naming the
6276 /// manifest line responsible:
6277 ///
6278 /// - **passway** — `mesofact::route_headers::RouteHeaderTable::parse`
6279 /// refuses the start, so the origin is simply down.
6280 /// - **worker** — `validateRouteHeaderTable` accepts it (it checks shape,
6281 /// not header validity) and `applyRouteHeaders` then throws inside the
6282 /// exported `fetch`, which is a 500 on every request, not the
6283 /// serve-without-the-headers degradation that code intends.
6284 ///
6285 /// So the strictness lives at the producer: a table that cannot be applied
6286 /// fails `yah cloud apply` at manifest load, naming domain, route and
6287 /// header. This is deliberately *not* a second parser — the check is
6288 /// `HeaderName`/`HeaderValue`'s own, the very constructors the passway door
6289 /// runs on the far side, and route *matching* semantics stay defined once,
6290 /// at the doors. Only routes that contribute to the table are checked, so
6291 /// the invariant is exactly "`route_headers_json`'s output parses".
6292 ///
6293 /// Called from [`Self::load`], alongside [`Self::validate_front_door`].
6294 pub fn validate_route_headers(&self) -> Result<()> {
6295 use axum::http::{HeaderName, HeaderValue};
6296
6297 for route in self.routes.iter().filter(|r| !r.headers.is_empty()) {
6298 if route.path.is_empty() {
6299 anyhow::bail!(
6300 "domain \"{}\" declares response headers on a route whose `path` is \
6301 empty — a rule that matches nothing (or everything, depending on \
6302 which front door reads it) is not a policy",
6303 self.name
6304 );
6305 }
6306 for (name, value) in &route.headers {
6307 HeaderName::try_from(name.as_str()).with_context(|| {
6308 format!(
6309 "domain \"{}\" route \"{}\" declares {name:?}, which is not a valid \
6310 HTTP header name — names are token characters only, so it is \
6311 `Cross-Origin-Opener-Policy`, never `Cross Origin Opener Policy`",
6312 self.name, route.path
6313 )
6314 })?;
6315 HeaderValue::try_from(value.as_str()).with_context(|| {
6316 format!(
6317 "domain \"{}\" route \"{}\" declares {name} = {value:?}, which is not \
6318 a valid HTTP header value — no newlines and no control characters",
6319 self.name, route.path
6320 )
6321 })?;
6322 }
6323 }
6324 Ok(())
6325 }
6326
6327 /// Whether this domain's route table binds any component of `service`.
6328 pub fn serves_service(&self, service: &str) -> bool {
6329 self.routes.iter().any(|r| {
6330 r.mode
6331 .component()
6332 .and_then(split_component_ref)
6333 .is_some_and(|(svc, _)| svc == service)
6334 })
6335 }
6336
6337 /// Persist to `.yah/domains/<name>.toml`, creating the domains
6338 /// directory if needed. Create-or-overwrite.
6339 pub fn save(&self, workspace_root: &Path) -> Result<()> {
6340 let dir = crate::paths::domains_dir(workspace_root);
6341 std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?;
6342 let path = crate::paths::domain_toml(workspace_root, &self.name);
6343 let s = toml::to_string_pretty(self)
6344 .with_context(|| format!("serializing domain {}", self.name))?;
6345 std::fs::write(&path, s).with_context(|| format!("writing {}", path.display()))
6346 }
6347
6348 /// Remove `.yah/domains/<name>.toml`. Returns `false` when the file
6349 /// was already absent.
6350 pub fn delete(workspace_root: &Path, name: &str) -> Result<bool> {
6351 let path = crate::paths::domain_toml(workspace_root, name);
6352 if !path.exists() {
6353 return Ok(false);
6354 }
6355 std::fs::remove_file(&path).with_context(|| format!("removing {}", path.display()))?;
6356 Ok(true)
6357 }
6358}
6359
6360impl RouteMode {
6361 /// Component reference for static/backend modes; `None` for redirects and
6362 /// bucket-root static routes, which reference no component.
6363 pub fn component(&self) -> Option<&str> {
6364 match self {
6365 Self::Static { component } | Self::Backend { component, .. } => Some(component),
6366 Self::StaticBucket { .. } | Self::Redirect { .. } => None,
6367 }
6368 }
6369}
6370
6371// ─── Service-group vault (R706 / W294) ───────────────────────────────────────
6372
6373/// A camp's declaration of one cluster secret, from
6374/// `.yah/infra/secrets/<slug>.toml`.
6375///
6376/// This is the *authoring* side of the fleet's cluster-secret store: it names
6377/// where the value lives in the camp (a `fob` vault slot), what the fleet should
6378/// call it, and — the point of R706 — which workloads are allowed to mount it.
6379///
6380/// The declaration is not itself the enforcement point. `yah cloud secret put`
6381/// reads this file, seals the vault value under the cluster KEK, and ships the
6382/// ciphertext **with its access rule** into raft; yubaba's `ClusterResolver`
6383/// evaluates the rule on the node at mount time. Deleting this file does not
6384/// revoke anything — the record in raft is the live authority. That asymmetry is
6385/// deliberate: a rule that lived only in a git-tracked camp file would be
6386/// trivially bypassed by anyone who could reach the fleet without the camp.
6387///
6388/// ```toml
6389/// #:schema ../../schema/secret.toml.schema.json
6390/// schema_version = 2
6391/// name = "cheers/cloud-admin/verify-key"
6392/// vault_slot = "cheers-cloud-admin-verify-key"
6393/// groups = ["prod"]
6394/// description = "Ed25519 public key yah-cloud-admin verifies operator PASETOs with"
6395///
6396/// [access]
6397/// workloads = [{ workload = "yah-cloud-admin" }]
6398///
6399/// [target]
6400/// kind = "file"
6401/// path = "/run/secrets/cheers-verify.key"
6402/// mode = 0o400
6403/// ```
6404#[derive(Debug, Clone, Serialize, Deserialize)]
6405#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
6406pub struct SecretConfig {
6407 pub schema_version: u32,
6408
6409 /// Logical cluster-secret key, as `SecretRef::Cluster { name }` spells it —
6410 /// e.g. `"tls/yah.dev/cert"`, `"cheers/cloud-admin/verify-key"`. May contain
6411 /// `/`; the file stem is a filesystem-safe slug and carries no meaning.
6412 pub name: String,
6413
6414 /// The `fob` vault slot in this camp holding the plaintext value. Read by
6415 /// `yah cloud secret put` at ship time and never recorded anywhere else — in
6416 /// particular the value is not in this file, so the declaration is safe to
6417 /// commit.
6418 pub vault_slot: String,
6419
6420 /// The sovereign groups this secret belongs to (R911-F8). Required and
6421 /// non-empty; each entry must be a `sovereign_group` that some
6422 /// `.yah/infra/machines/*.toml` declares ([`SecretConfig::load`] checks).
6423 ///
6424 /// Cluster secrets live per group in the fleet object store
6425 /// (`secrets/<group>/<name>.sealed`), so a declaration with no group has
6426 /// nowhere to land and nothing to be compared against. `yah cloud secret
6427 /// put` refuses a node whose `/raft/status` group is not listed here, and
6428 /// `status` counts a declaration only against nodes in one of these groups.
6429 pub groups: Vec<String>,
6430
6431 /// Human note for `yah cloud secret ls`. What this secret is and who minted
6432 /// it — the thing nobody remembers 6 months later.
6433 #[serde(default, skip_serializing_if = "Option::is_none")]
6434 pub description: Option<String>,
6435
6436 /// How the vault slot's text decodes into the bytes the consumer expects.
6437 ///
6438 /// `fob` slots hold strings, but plenty of real secrets are **binary** — an
6439 /// Ed25519 key is exactly 32 raw bytes, and `yah-cloud-admin` rejects a key
6440 /// file of any other length. Without this field the only way to ship such a
6441 /// key would be to hope its bytes happened to be valid UTF-8, which for a
6442 /// random key they are not.
6443 ///
6444 /// Defaults to [`SecretEncoding::Utf8`] — the right answer for tokens,
6445 /// passwords, and PEM, which is most secrets.
6446 #[serde(default)]
6447 pub encoding: SecretEncoding,
6448
6449 /// Who may mount it. Stamped onto the raft record verbatim.
6450 ///
6451 /// Defaults to [`SecretAccess::default`] — the deny-all empty allow-list. A
6452 /// declaration that forgets this field produces a secret nobody can mount,
6453 /// which is the correct direction to fail in.
6454 ///
6455 /// Three forms:
6456 ///
6457 /// ```toml
6458 /// access = "allow_any" # explicit escape hatch
6459 ///
6460 /// [access] # named workloads
6461 /// workloads = [{ workload = "yah-cloud-admin" }]
6462 ///
6463 /// [access] # signed recipes (R555-F5)
6464 /// recipes = [{ recipe = "rusty-v8-musl", key = "3d40…" }]
6465 /// ```
6466 ///
6467 /// Use the `recipes` form for a credential a **dispatched build** needs (the
6468 /// R2 write key, the cosign signing key). A remote QED run's workload name
6469 /// is a fresh `forge-<uuid>` every time, so `workloads` cannot name it and
6470 /// `allow_any` over-answers — see W235 §Seam (c) secret scoping. `key` is
6471 /// the hex Ed25519 public key from the recipe's `[admission]` block.
6472 #[serde(default)]
6473 pub access: SecretAccess,
6474
6475 /// Advisory: the mount shape a consuming workload should declare. Not
6476 /// enforced — yubaba honours whatever the `WorkloadSpec` asks for — but it
6477 /// lets `yah cloud secret put` print the exact `SecretMount` to paste, so
6478 /// the consumer and the declaration can't drift on path or mode.
6479 #[serde(default, skip_serializing_if = "Option::is_none")]
6480 pub target: Option<SecretTargetDecl>,
6481}
6482
6483/// How a [`SecretConfig`]'s vault text becomes the bytes delivered to the
6484/// container (R706 / W294).
6485#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
6486#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
6487#[serde(rename_all = "kebab-case")]
6488pub enum SecretEncoding {
6489 /// Ship the vault string's UTF-8 bytes verbatim. Tokens, passwords, PEM.
6490 #[default]
6491 Utf8,
6492 /// The vault string is hex; ship the decoded bytes. Use for binary key
6493 /// material — e.g. a raw Ed25519 key, which must land as exactly 32 bytes.
6494 Hex,
6495}
6496
6497/// Advisory mount shape on a [`SecretConfig`]. Mirrors
6498/// `workload_spec::SecretTarget` in a TOML-friendly, externally-tagged-free
6499/// shape (a `kind` discriminator reads better in a hand-written manifest than
6500/// serde's default enum encoding).
6501#[derive(Debug, Clone, Serialize, Deserialize)]
6502#[cfg_attr(feature = "json-schema", derive(schemars::JsonSchema))]
6503#[serde(tag = "kind", rename_all = "kebab-case")]
6504pub enum SecretTargetDecl {
6505 /// Mounted as a tmpfs-backed file inside the container.
6506 File {
6507 /// Absolute path inside the container.
6508 path: String,
6509 /// Unix permission bits. Defaults to `0o400` (owner-read-only).
6510 #[serde(default = "default_secret_mode")]
6511 mode: u32,
6512 },
6513 /// Injected as an environment variable. Prefer `file` — env vars leak
6514 /// through subprocess environments and log dumps.
6515 EnvVar { name: String },
6516}
6517
6518fn default_secret_mode() -> u32 {
6519 0o400
6520}
6521
6522impl SecretTargetDecl {
6523 /// The `workload_spec` target this declaration describes.
6524 pub fn to_target(&self) -> workload_spec::SecretTarget {
6525 match self {
6526 Self::File { path, mode } => workload_spec::SecretTarget::File {
6527 path: path.into(),
6528 mode: *mode,
6529 },
6530 Self::EnvVar { name } => workload_spec::SecretTarget::EnvVar { name: name.clone() },
6531 }
6532 }
6533}
6534
6535/// The `schema_version` every [`SecretConfig`] must carry. Version 2 added the
6536/// required `groups` (R911-F8).
6537pub const SECRET_CONFIG_SCHEMA_VERSION: u32 = 2;
6538
6539impl SecretConfig {
6540 /// Parse a single `.yah/infra/secrets/<slug>.toml`.
6541 ///
6542 /// `sovereign_groups` is the camp's group vocabulary
6543 /// ([`CloudConfig::declared_sovereign_groups`]); every entry in `groups`
6544 /// must be one of them.
6545 pub fn load(path: &Path, sovereign_groups: &[&str]) -> Result<Self> {
6546 let src =
6547 std::fs::read_to_string(path).with_context(|| format!("reading {}", path.display()))?;
6548 let cfg: Self =
6549 toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
6550 cfg.validate()
6551 .and_then(|()| cfg.validate_groups(sovereign_groups))
6552 .with_context(|| format!("validating {}", path.display()))?;
6553 Ok(cfg)
6554 }
6555
6556 /// Load every declaration in `dir`, keyed by logical secret name. A missing
6557 /// directory is an empty map (a camp with no cluster secrets is normal).
6558 ///
6559 /// Two files declaring the same `name` is a hard error, not a last-writer-
6560 /// wins merge: they would race to define the access rule for one record, and
6561 /// whichever lost would look correct in git while being inert on the fleet.
6562 pub fn load_dir(dir: &Path, sovereign_groups: &[&str]) -> Result<BTreeMap<String, Self>> {
6563 let mut out: BTreeMap<String, Self> = BTreeMap::new();
6564 if !dir.exists() {
6565 return Ok(out);
6566 }
6567 for entry in std::fs::read_dir(dir).with_context(|| format!("reading {}", dir.display()))? {
6568 let path = entry?.path();
6569 if path.extension().is_none_or(|e| e != "toml") {
6570 continue;
6571 }
6572 let cfg = Self::load(&path, sovereign_groups)?;
6573 if let Some(prev) = out.insert(cfg.name.clone(), cfg) {
6574 anyhow::bail!(
6575 "two secret declarations both claim name {:?} (one of them is {}); \
6576 a cluster secret must have exactly one declaration so its access \
6577 rule has one author",
6578 prev.name,
6579 path.display()
6580 );
6581 }
6582 }
6583 Ok(out)
6584 }
6585
6586 /// Reject declarations that would produce an unusable or dangerous record.
6587 ///
6588 /// Structural only: whether each of `groups` names a real sovereign group
6589 /// needs the camp's machines, so that half is [`Self::validate_groups`].
6590 pub fn validate(&self) -> Result<()> {
6591 if self.schema_version != SECRET_CONFIG_SCHEMA_VERSION {
6592 anyhow::bail!(
6593 "`schema_version` is {}, expected {SECRET_CONFIG_SCHEMA_VERSION}: version 2 \
6594 added the required `groups = [\"<sovereign group>\", ...]` (R911-F8)",
6595 self.schema_version
6596 );
6597 }
6598 if self.name.trim().is_empty() {
6599 anyhow::bail!("`name` must not be empty");
6600 }
6601 if self.groups.is_empty() {
6602 anyhow::bail!(
6603 "`groups` must name at least one sovereign group: cluster secrets live per \
6604 group (secrets/<group>/), so a declaration in no group has nowhere to land"
6605 );
6606 }
6607 if let Some(bad) = self.groups.iter().find(|g| g.trim().is_empty()) {
6608 anyhow::bail!("`groups` has an empty entry: {bad:?}");
6609 }
6610 let mut seen = std::collections::BTreeSet::new();
6611 if let Some(dup) = self.groups.iter().find(|g| !seen.insert(g.as_str())) {
6612 anyhow::bail!("`groups` lists {dup:?} twice");
6613 }
6614 if self.vault_slot.trim().is_empty() {
6615 anyhow::bail!(
6616 "`vault_slot` must not be empty — it names the fob slot holding the value"
6617 );
6618 }
6619 // A deny-all rule is a *valid* record (it is the fail-closed default the
6620 // resolver relies on) but it is never a useful thing to deliberately
6621 // ship, so catching it here saves an operator the round-trip of
6622 // deploying a workload that mysteriously can't see its own secret.
6623 if let SecretAccess::Workloads(entries) = &self.access {
6624 if entries.is_empty() {
6625 anyhow::bail!(
6626 "`[access]` admits nobody: list the workloads allowed to mount {:?} \
6627 (e.g. `workloads = [{{ workload = \"my-service\" }}]`), or set \
6628 `access = \"allow_any\"` to store it unrestricted",
6629 self.name
6630 );
6631 }
6632 if let Some(bad) = entries.iter().find(|e| e.workload.trim().is_empty()) {
6633 anyhow::bail!("`[access]` entry has an empty `workload` name: {bad:?}");
6634 }
6635 }
6636 Ok(())
6637 }
6638
6639 /// Every entry in `groups` must be a sovereign group the camp's machines
6640 /// declare. A typo would otherwise produce a declaration no node ever
6641 /// matches, which `put` would refuse everywhere and `status` would never
6642 /// count, without either one saying why.
6643 pub fn validate_groups(&self, sovereign_groups: &[&str]) -> Result<()> {
6644 if let Some(unknown) = self
6645 .groups
6646 .iter()
6647 .find(|g| !sovereign_groups.contains(&g.as_str()))
6648 {
6649 anyhow::bail!(
6650 "`groups` names {unknown:?}, which no .yah/infra/machines/*.toml declares as a \
6651 `sovereign_group` (declared: {})",
6652 if sovereign_groups.is_empty() {
6653 "(none)".to_string()
6654 } else {
6655 sovereign_groups.join(", ")
6656 }
6657 );
6658 }
6659 Ok(())
6660 }
6661}
6662
6663#[cfg(test)]
6664mod secret_config_tests {
6665 use super::*;
6666
6667 fn parse(body: &str) -> Result<SecretConfig> {
6668 let cfg: SecretConfig = toml::from_str(body)?;
6669 cfg.validate()?;
6670 Ok(cfg)
6671 }
6672
6673 #[test]
6674 fn minimal_declaration_parses_with_narrow_defaults() {
6675 let cfg = parse(
6676 r#"
6677schema_version = 2
6678name = "svc/token"
6679vault_slot = "svc-token"
6680groups = ["prod"]
6681[access]
6682workloads = [{ workload = "svc" }]
6683"#,
6684 )
6685 .unwrap();
6686
6687 assert_eq!(cfg.groups, vec!["prod".to_string()]);
6688 assert_eq!(cfg.encoding, SecretEncoding::Utf8, "text is the default");
6689 assert!(cfg.target.is_none());
6690 // The omitted tenant/namespace must narrow to the singletons, not widen
6691 // to a wildcard.
6692 assert!(cfg
6693 .access
6694 .admits(&workload_spec::secrets::SecretConsumer::workload("svc")));
6695 assert!(!cfg
6696 .access
6697 .admits(&workload_spec::secrets::SecretConsumer::workload("other")));
6698 }
6699
6700 #[test]
6701 fn allow_any_is_spelled_as_a_bare_string() {
6702 // The operator-facing spelling, pinned: `access = "allow_any"`.
6703 let cfg = parse(
6704 r#"
6705schema_version = 2
6706name = "public/thing"
6707vault_slot = "slot"
6708groups = ["prod"]
6709access = "allow_any"
6710"#,
6711 )
6712 .unwrap();
6713 assert_eq!(cfg.access, SecretAccess::AllowAny);
6714 }
6715
6716 #[test]
6717 fn a_declaration_with_no_access_block_is_rejected() {
6718 // Omitting `[access]` defaults to deny-all, which is the correct
6719 // *runtime* default but never a correct authoring intent — so it must
6720 // not silently produce a secret nobody can mount.
6721 let err = parse(
6722 r#"
6723schema_version = 2
6724name = "svc/token"
6725vault_slot = "svc-token"
6726groups = ["prod"]
6727"#,
6728 )
6729 .unwrap_err()
6730 .to_string();
6731 assert!(err.contains("admits nobody"), "got {err}");
6732 }
6733
6734 #[test]
6735 fn empty_name_or_slot_is_rejected() {
6736 assert!(parse(
6737 r#"
6738schema_version = 2
6739name = ""
6740vault_slot = "slot"
6741groups = ["prod"]
6742access = "allow_any"
6743"#
6744 )
6745 .is_err());
6746 assert!(parse(
6747 r#"
6748schema_version = 2
6749name = "x"
6750vault_slot = " "
6751groups = ["prod"]
6752access = "allow_any"
6753"#
6754 )
6755 .is_err());
6756 }
6757
6758 #[test]
6759 fn target_declaration_maps_onto_the_workload_spec_type() {
6760 let cfg = parse(
6761 r#"
6762schema_version = 2
6763name = "svc/token"
6764vault_slot = "slot"
6765groups = ["prod"]
6766access = "allow_any"
6767[target]
6768kind = "file"
6769path = "/run/secrets/t"
6770"#,
6771 )
6772 .unwrap();
6773 match cfg.target.unwrap().to_target() {
6774 workload_spec::SecretTarget::File { path, mode } => {
6775 assert_eq!(path, std::path::PathBuf::from("/run/secrets/t"));
6776 assert_eq!(mode, 0o400, "owner-read-only by default");
6777 }
6778 other => panic!("expected File, got {other:?}"),
6779 }
6780 }
6781
6782 #[test]
6783 fn load_dir_is_empty_for_a_camp_with_no_secrets() {
6784 let tmp = tempfile::TempDir::new().unwrap();
6785 assert!(SecretConfig::load_dir(&tmp.path().join("nope"), &["prod"])
6786 .unwrap()
6787 .is_empty());
6788 }
6789
6790 /// Write `body` to a temp file and run the real loader over it, so the
6791 /// assertions see the error chain an operator would.
6792 fn load_body(body: &str, sovereign_groups: &[&str]) -> Result<SecretConfig> {
6793 let tmp = tempfile::TempDir::new().unwrap();
6794 let path = tmp.path().join("decl.toml");
6795 std::fs::write(&path, body).unwrap();
6796 SecretConfig::load(&path, sovereign_groups)
6797 }
6798
6799 #[test]
6800 fn a_declaration_without_groups_fails_to_load_naming_the_field() {
6801 let err = format!(
6802 "{:#}",
6803 load_body(
6804 "schema_version = 2\nname = \"svc/token\"\nvault_slot = \"slot\"\n\
6805 access = \"allow_any\"\n",
6806 &["prod"],
6807 )
6808 .unwrap_err()
6809 );
6810 assert!(err.contains("groups"), "must name the field: {err}");
6811 assert!(err.contains("decl.toml"), "must name the file: {err}");
6812 }
6813
6814 #[test]
6815 fn an_empty_or_duplicated_group_list_is_rejected() {
6816 let err = format!(
6817 "{:#}",
6818 load_body(
6819 "schema_version = 2\nname = \"s\"\nvault_slot = \"slot\"\ngroups = []\n\
6820 access = \"allow_any\"\n",
6821 &["prod"],
6822 )
6823 .unwrap_err()
6824 );
6825 assert!(err.contains("at least one sovereign group"), "got {err}");
6826
6827 let err = format!(
6828 "{:#}",
6829 load_body(
6830 "schema_version = 2\nname = \"s\"\nvault_slot = \"slot\"\n\
6831 groups = [\"prod\", \"prod\"]\naccess = \"allow_any\"\n",
6832 &["prod"],
6833 )
6834 .unwrap_err()
6835 );
6836 assert!(err.contains("twice"), "got {err}");
6837 }
6838
6839 #[test]
6840 fn a_group_no_machine_declares_is_rejected_naming_the_vocabulary() {
6841 let err = format!(
6842 "{:#}",
6843 load_body(
6844 "schema_version = 2\nname = \"s\"\nvault_slot = \"slot\"\n\
6845 groups = [\"stagin\"]\naccess = \"allow_any\"\n",
6846 &["dev", "prod"],
6847 )
6848 .unwrap_err()
6849 );
6850 assert!(err.contains("\"stagin\""), "names the bad entry: {err}");
6851 assert!(err.contains("dev, prod"), "names what is declared: {err}");
6852 }
6853
6854 #[test]
6855 fn a_version_1_declaration_is_refused_naming_the_migration() {
6856 let err = format!(
6857 "{:#}",
6858 load_body(
6859 "schema_version = 1\nname = \"s\"\nvault_slot = \"slot\"\n\
6860 groups = [\"prod\"]\naccess = \"allow_any\"\n",
6861 &["prod"],
6862 )
6863 .unwrap_err()
6864 );
6865 assert!(err.contains("schema_version") && err.contains("groups"), "got {err}");
6866 }
6867}
6868
6869/// Split a `"<service>/<component-id>"` ref. Returns `None` if the ref
6870/// isn't shaped like `service/component`.
6871pub(crate) fn split_component_ref(s: &str) -> Option<(&str, &str)> {
6872 let (svc, comp) = s.split_once('/')?;
6873 if svc.is_empty() || comp.is_empty() || comp.contains('/') {
6874 return None;
6875 }
6876 Some((svc, comp))
6877}
6878
6879#[cfg(test)]
6880mod tests {
6881 use super::*;
6882 use std::path::PathBuf;
6883
6884 fn make_machine(name: &str, mesh_tags: Vec<&str>) -> MachineConfig {
6885 MachineConfig {
6886 name: name.into(),
6887 provider: "hetzner".into(),
6888 location: Some("hil".into()),
6889 server_type: Some("ccx13".into()),
6890 hosts_mirrors: vec![],
6891 mesh_tags: mesh_tags.into_iter().map(String::from).collect(),
6892 region: None,
6893 zone: None,
6894 arch: None,
6895 bucket: None,
6896 vendor: None,
6897 nickname: None,
6898 legacy_hostkey_fingerprint: None,
6899 registration: Default::default(),
6900 ssh_keys: vec![],
6901 cloudflared: None,
6902 hosts_operator_bridge: false,
6903 connect: None,
6904 allocatable: None,
6905 taints: vec![],
6906 sovereign_group: None,
6907 sovereign_role: None,
6908 ingress_floating_ip: None,
6909 }
6910 }
6911
6912 /// Like [`make_machine`] but with explicit topology axes for F16 tests.
6913 fn make_machine_topo(
6914 name: &str,
6915 provider: &str,
6916 region: &str,
6917 mesh_tags: Vec<&str>,
6918 ) -> MachineConfig {
6919 MachineConfig {
6920 provider: provider.into(),
6921 region: Some(region.into()),
6922 zone: Some(region.into()),
6923 ..make_machine(name, mesh_tags)
6924 }
6925 }
6926
6927 fn make_empty_cfg(machines: Vec<MachineConfig>) -> CloudConfig {
6928 CloudConfig {
6929 workspace_root: PathBuf::new(),
6930 machines,
6931 providers: vec![],
6932 machine_origins: BTreeMap::new(),
6933 provider_origins: BTreeMap::new(),
6934 recovery_measurements: BTreeMap::new(),
6935 services: BTreeMap::new(),
6936 domains: BTreeMap::new(),
6937 legacy_mirrors: vec![],
6938 workloads: vec![],
6939 topology: TopologyConfig::default(),
6940 }
6941 }
6942
6943 #[test]
6944 fn required_spec_parses_from_provider_fields() {
6945 let toml_src = r#"
6946use = "hetzner-primary"
6947[required]
6948mesh_tags = ["tag:cloud-runner"]
6949"#;
6950 let slot: MirrorProviderSlot = toml::from_str(toml_src).unwrap();
6951 let req = slot.required().expect("required block present");
6952 assert_eq!(req.mesh_tags, vec!["tag:cloud-runner"]);
6953 }
6954
6955 #[test]
6956 fn required_spec_absent_when_field_missing() {
6957 let slot: MirrorProviderSlot = toml::from_str(r#"use = "hetzner-primary""#).unwrap();
6958 assert!(slot.required().is_none());
6959 }
6960
6961 #[test]
6962 fn db_catalog_parses_all_env_blocks() {
6963 // W241 / R571-F8: a service.toml [db] table with dev/pond/cloud.
6964 let toml_src = r#"
6965schema_version = 1
6966name = "scrabcake"
6967[address]
6968kind = "front-door"
6969domain = "scrabcake.net.yah.dev"
6970
6971[[db.dev]]
6972name = "main"
6973path = "data/dev.sqlite"
6974
6975[[db.pond]]
6976name = "main"
6977port = 5433
6978
6979[[db.pond]]
6980name = "pg"
6981port = 5432
6982kind = "postgres"
6983
6984[[db.cloud]]
6985name = "main"
6986url = "libsql://scrabcake.turso.io"
6987auth_token_env = "SCRABCAKE_TURSO_TOKEN"
6988"#;
6989 let svc: ServiceConfig = toml::from_str(toml_src).unwrap();
6990 assert_eq!(svc.db.dev.len(), 1);
6991 assert_eq!(svc.db.dev[0].path, "data/dev.sqlite");
6992 assert_eq!(svc.db.pond.len(), 2);
6993 assert_eq!(svc.db.pond[0].port, Some(5433));
6994 assert_eq!(svc.db.pond[0].kind, PondDbKind::Turso); // default
6995 assert_eq!(svc.db.pond[1].kind, PondDbKind::Postgres);
6996 assert_eq!(
6997 svc.db.cloud[0].auth_token_env.as_deref(),
6998 Some("SCRABCAKE_TURSO_TOKEN")
6999 );
7000 }
7001
7002 #[test]
7003 fn service_without_db_table_has_empty_catalog() {
7004 let svc: ServiceConfig =
7005 toml::from_str("schema_version = 1\nname = \"s\"\n[address]\nkind = \"front-door\"\ndomain = \"s.dev\"\n").unwrap();
7006 assert!(svc.db.is_empty());
7007 // And an empty [db] must not appear when re-serialized.
7008 let out = toml::to_string(&svc).unwrap();
7009 assert!(
7010 !out.contains("[db"),
7011 "empty db table should be skipped: {out}"
7012 );
7013 }
7014
7015 #[test]
7016 fn camp_shared_cloud_toml_parses() {
7017 let src = r#"
7018[[cloud]]
7019name = "analytics"
7020url = "postgres://shared/analytics"
7021"#;
7022 let shared: CampCloudDbs = toml::from_str(src).unwrap();
7023 assert_eq!(shared.cloud.len(), 1);
7024 assert_eq!(shared.cloud[0].name, "analytics");
7025 }
7026
7027 #[test]
7028 fn resolve_machine_by_mesh_tags_superset_match() {
7029 let cfg = make_empty_cfg(vec![
7030 make_machine("yah-bnt-1", vec!["tag:primary-yah", "tag:tier-scratch"]),
7031 make_machine("us-west-001", vec!["tag:primary-yah", "tag:cloud-runner"]),
7032 ]);
7033 let picked = cfg
7034 .resolve_machine_by_mesh_tags(&["tag:cloud-runner".into()])
7035 .map(|m| m.name.as_str());
7036 assert_eq!(picked, Some("us-west-001"));
7037 }
7038
7039 #[test]
7040 fn resolve_machine_by_mesh_tags_returns_none_when_no_match() {
7041 let cfg = make_empty_cfg(vec![make_machine("yah-bnt-1", vec!["tag:primary-yah"])]);
7042 assert!(cfg
7043 .resolve_machine_by_mesh_tags(&["tag:cloud-runner".into()])
7044 .is_none());
7045 }
7046
7047 // ─── R590-F1 mesh-tag node-selector admission ───────────────────────────
7048
7049 /// Build a forge WorkloadSpec carrying the R594 node-selector annotation.
7050 /// `selector` is the comma-joined mesh-tag set; `None` omits the annotation
7051 /// entirely (pre-R594 "no constraint").
7052 fn ws_with_selector(selector: Option<&str>) -> WorkloadSpec {
7053 use workload_spec::{ImageRef, TierTag};
7054 let mut ws = WorkloadSpec::for_forge(
7055 "R590-F1-test",
7056 ImageRef {
7057 registry: "docker.io".into(),
7058 repository: "library/busybox".into(),
7059 tag: "latest".into(),
7060 digest: workload_spec::testing::test_digest(),
7061 },
7062 TierTag("infra".into()),
7063 vec![],
7064 );
7065 if let Some(sel) = selector {
7066 ws.annotations.insert(
7067 velveteen_exec::remote::NODE_SELECTOR_MESH_TAGS_ANNOTATION.into(),
7068 sel.into(),
7069 );
7070 }
7071 ws
7072 }
7073
7074 /// The build-worker fleet shape: one x86 node (us-west-002) and one arm
7075 /// node (a Pi5), both carrying `tag:build-worker`.
7076 fn build_worker_fleet() -> CloudConfig {
7077 make_empty_cfg(vec![
7078 make_machine("us-west-002", vec!["tag:build-worker", "arch:x86"]),
7079 make_machine("pi5-001", vec!["tag:build-worker", "arch:arm"]),
7080 ])
7081 }
7082
7083 #[test]
7084 fn admit_workload_routes_amd64_to_x86_worker() {
7085 let cfg = build_worker_fleet();
7086 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
7087 let picked = cfg.admit_workload(&ws).unwrap();
7088 assert_eq!(picked.name, "us-west-002");
7089 }
7090
7091 #[test]
7092 fn admit_workload_routes_arm64_to_pi5_worker() {
7093 let cfg = build_worker_fleet();
7094 let ws = ws_with_selector(Some("tag:build-worker,arch:arm"));
7095 let picked = cfg.admit_workload(&ws).unwrap();
7096 assert_eq!(picked.name, "pi5-001");
7097 }
7098
7099 /// A forge run must be admissible on a build-worker smaller than its own
7100 /// cgroup ceiling.
7101 ///
7102 /// The fleet's arm build-workers are 8 GiB Pi-5s and `for_forge` sets a
7103 /// 32 GiB ceiling, so while admission read `resources.memory_mb` as the
7104 /// capacity floor this returned "no candidates" and *every* offloaded qed
7105 /// step to those nodes failed at dispatch — measured on desktop-release run
7106 /// b04cef47, where the aarch64-linux row died in 1.6s. The other
7107 /// build-workers (16 GiB us-west-003, and the arm Pi-5s) were excluded the
7108 /// same way, leaving one 47 GiB node as the fleet's only legal target for
7109 /// remote CI.
7110 #[test]
7111 fn admit_workload_places_a_forge_run_on_a_worker_smaller_than_its_ceiling() {
7112 let mut pi = make_machine("pi5-001", vec!["tag:build-worker", "arch:arm"]);
7113 pi.allocatable = Some(NodeAllocatable {
7114 memory_mb: 8192,
7115 cpu_millis: 4000,
7116 });
7117 let cfg = make_empty_cfg(vec![pi]);
7118
7119 let ws = ws_with_selector(Some("tag:build-worker,arch:arm"));
7120 assert!(
7121 ws.resources.memory_mb > 8192,
7122 "precondition: the ceiling must exceed the node, or this proves nothing"
7123 );
7124
7125 let picked = cfg
7126 .admit_workload(&ws)
7127 .expect("an 8 GiB build-worker must admit a forge run");
7128 assert_eq!(picked.name, "pi5-001");
7129 }
7130
7131 /// The floor is still enforced — the fix separates two numbers, it does not
7132 /// disable the R572-F5 capacity check.
7133 #[test]
7134 fn admit_workload_still_rejects_a_node_below_the_declared_request() {
7135 let mut tiny = make_machine("tiny-001", vec!["tag:build-worker", "arch:arm"]);
7136 tiny.allocatable = Some(NodeAllocatable {
7137 memory_mb: 512,
7138 cpu_millis: 4000,
7139 });
7140 let cfg = make_empty_cfg(vec![tiny]);
7141
7142 let ws = ws_with_selector(Some("tag:build-worker,arch:arm"));
7143 assert!(
7144 cfg.admit_workload(&ws).is_err(),
7145 "a 512 MiB node cannot satisfy a 2 GiB forge request"
7146 );
7147 }
7148
7149 // ─── R833-F8 imperative node-selector admission ─────────────────────────
7150
7151 /// Build a forge WorkloadSpec carrying the R833-F8 imperative node
7152 /// selector — the operator's `--where=node:<machine>`.
7153 fn ws_pinned_to(node: &str) -> WorkloadSpec {
7154 let mut ws = ws_with_selector(None);
7155 ws.annotations.insert(
7156 velveteen_exec::remote::NODE_SELECTOR_NODE_ANNOTATION.into(),
7157 node.into(),
7158 );
7159 ws
7160 }
7161
7162 /// The ticket's acceptance shape: a named node wins over the
7163 /// declaration-order tie-break that would otherwise decide placement.
7164 /// `us-west-002` is declared first and carries every tag, so an inferred
7165 /// placement lands there; the pin must reach `pi5-001` regardless.
7166 #[test]
7167 fn admit_workload_honours_an_explicitly_named_node() {
7168 let cfg = build_worker_fleet();
7169 assert_eq!(
7170 cfg.admit_workload(&ws_with_selector(Some("tag:build-worker")))
7171 .unwrap()
7172 .name,
7173 "us-west-002",
7174 "precondition: inference elects the first-declared node",
7175 );
7176 assert_eq!(
7177 cfg.admit_workload(&ws_pinned_to("pi5-001")).unwrap().name,
7178 "pi5-001",
7179 );
7180 }
7181
7182 /// A pin at a machine that is not declared fails loud, naming the
7183 /// constraint and the pool — the operator mistyped a node, and silently
7184 /// running the build somewhere else is the one outcome that must not
7185 /// happen.
7186 #[test]
7187 fn admit_workload_refuses_a_node_that_is_not_declared() {
7188 let cfg = build_worker_fleet();
7189 let err = cfg
7190 .admit_workload(&ws_pinned_to("us-west-404"))
7191 .unwrap_err()
7192 .to_string();
7193 assert!(err.contains("required.nodes=[us-west-404]"), "{err}");
7194 assert!(err.contains("us-west-002"), "the pool must be named: {err}");
7195 }
7196
7197 /// The pin narrows the candidate set; it does not suspend the other axes.
7198 /// A named node that cannot fit the workload still refuses, rather than
7199 /// being handed work it has no room for.
7200 #[test]
7201 fn a_pinned_node_is_still_checked_against_capacity() {
7202 let mut tiny = make_machine("tiny-001", vec!["tag:build-worker", "arch:arm"]);
7203 tiny.allocatable = Some(NodeAllocatable {
7204 memory_mb: 512,
7205 cpu_millis: 4000,
7206 });
7207 let cfg = make_empty_cfg(vec![tiny]);
7208 assert!(cfg.admit_workload(&ws_pinned_to("tiny-001")).is_err());
7209 }
7210
7211 /// Inference is untouched: with no node annotation the `nodes` axis is
7212 /// empty, which is "no constraint" — every pre-R833-F8 workload is admitted
7213 /// exactly as before.
7214 #[test]
7215 fn an_unpinned_workload_carries_no_node_constraint() {
7216 assert!(node_selector_node(&ws_with_selector(Some("arch:x86"))).is_none());
7217 assert_eq!(
7218 node_selector_node(&ws_pinned_to("us-west-003")).as_deref(),
7219 Some("us-west-003")
7220 );
7221 assert!(RequiredSpec::default().is_unconstrained());
7222 assert!(!RequiredSpec {
7223 nodes: vec!["us-west-003".into()],
7224 ..Default::default()
7225 }
7226 .is_unconstrained());
7227 }
7228
7229 #[test]
7230 fn admit_workload_rejects_node_missing_required_tag() {
7231 // Only an arm worker exists; an x86 build must NOT land on it.
7232 let cfg = make_empty_cfg(vec![make_machine(
7233 "pi5-001",
7234 vec!["tag:build-worker", "arch:arm"],
7235 )]);
7236 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
7237 assert!(cfg.admit_workload(&ws).is_err());
7238 }
7239
7240 /// R555-S1 regression: with TWO nodes carrying the same tag set, which one
7241 /// admits must be decided by *declaration order* (file name), which is the
7242 /// contract `admit_workload` documents — not by `read_dir` order, which is
7243 /// filesystem-dependent and can change when an unrelated file appears in
7244 /// the directory. Written creation-order-reversed so a filesystem that
7245 /// yields creation order (rather than sorted order) trips it without the
7246 /// sort in `load_dir`.
7247 ///
7248 /// Live consequence this guards: `.yah/infra/machines/` carries both
7249 /// us-west-002 and us-west-003 on `[tag:build-worker, arch:x86, os:linux]`,
7250 /// so an x86 QED offload has two equal candidates. Unstable selection means
7251 /// a retried build cannot be relied on to land back on the node whose
7252 /// working state it left behind.
7253 #[test]
7254 fn equally_matching_machines_admit_in_file_name_order() {
7255 let tmp = tempfile::TempDir::new().unwrap();
7256 let machines = tmp.path().join(".yah").join("infra").join("machines");
7257 std::fs::create_dir_all(&machines).unwrap();
7258 let toml_for = |name: &str| {
7259 format!(
7260 r#"name = "{name}"
7261provider = "static"
7262mesh_tags = ["tag:build-worker", "arch:x86"]
7263"#
7264 )
7265 };
7266 // Reverse-of-sorted creation order on purpose.
7267 std::fs::write(machines.join("b-second.toml"), toml_for("b-second")).unwrap();
7268 std::fs::write(machines.join("a-first.toml"), toml_for("a-first")).unwrap();
7269
7270 let cfg = CloudConfig::load(tmp.path()).unwrap();
7271 assert_eq!(
7272 cfg.machines.iter().map(|m| m.name.as_str()).collect::<Vec<_>>(),
7273 vec!["a-first", "b-second"],
7274 "machines must load in file-name order, not read_dir order"
7275 );
7276
7277 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
7278 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "a-first");
7279
7280 // R605-T14: the same two nodes, seen as the pool they are. The head is
7281 // what `admit_workload` returns, and the tail is what the dispatcher
7282 // fails over to when the head does not answer — so these two views must
7283 // come from one predicate, not two.
7284 assert_eq!(
7285 cfg.admit_workload_candidates(&ws)
7286 .unwrap()
7287 .iter()
7288 .map(|m| m.name.as_str())
7289 .collect::<Vec<_>>(),
7290 vec!["a-first", "b-second"],
7291 "the pool must be every admissible node, in the same declaration order"
7292 );
7293 }
7294
7295 /// A pool of one is still a pool, and a pool of none is an `Err` that reads
7296 /// exactly like `admit_workload`'s — "nothing admits this" is one failure
7297 /// with one wording, not two.
7298 #[test]
7299 fn admit_workload_candidates_matches_admit_workload_on_the_edges() {
7300 let cfg = make_empty_cfg(vec![
7301 make_machine("x86-box", vec!["tag:build-worker", "arch:x86"]),
7302 make_machine("arm-box", vec!["tag:build-worker", "arch:arm"]),
7303 ]);
7304
7305 let one = ws_with_selector(Some("arch:arm"));
7306 assert_eq!(
7307 cfg.admit_workload_candidates(&one)
7308 .unwrap()
7309 .iter()
7310 .map(|m| m.name.as_str())
7311 .collect::<Vec<_>>(),
7312 vec!["arm-box"],
7313 "only one node carries arch:arm, so the pool is that one node"
7314 );
7315
7316 let none = ws_with_selector(Some("arch:riscv"));
7317 let pool_err = cfg.admit_workload_candidates(&none).unwrap_err().to_string();
7318 let single_err = cfg.admit_workload(&none).unwrap_err().to_string();
7319 assert_eq!(
7320 pool_err, single_err,
7321 "an empty pool must be refused in the same words as an unadmitted workload"
7322 );
7323 }
7324
7325 /// R844-B7 — the wrong-root half of the distinction. A directory with no
7326 /// `.yah/` at all used to load as a valid config with zero machines, so a
7327 /// caller pointed at the wrong directory got a green result that measured
7328 /// nothing. Asserting `load` merely *succeeds* is what let that through;
7329 /// the shape that catches it is a non-zero machine count, or — here — an
7330 /// `Err` naming the path that was looked for.
7331 #[test]
7332 fn loading_a_directory_that_is_not_a_yah_workspace_is_an_error() {
7333 let tmp = tempfile::TempDir::new().unwrap();
7334 // A plausible-looking package root: real files, real subdirectories,
7335 // no `.yah/`. This is exactly what `load_cloud(".")` reads when a test
7336 // runs under `cargo test` from a member crate.
7337 std::fs::create_dir_all(tmp.path().join("src")).unwrap();
7338 std::fs::write(tmp.path().join("Cargo.toml"), "[package]\nname = \"x\"\n").unwrap();
7339
7340 let err = CloudConfig::load(tmp.path()).expect_err(
7341 "a directory with no .yah/ is the WRONG DIRECTORY, not a fleet with no machines",
7342 );
7343 let msg = format!("{err:#}");
7344 assert!(
7345 msg.contains("not a yah workspace"),
7346 "error must say the root is not a workspace, got: {msg}"
7347 );
7348 assert!(
7349 msg.contains(&tmp.path().join(".yah").display().to_string()),
7350 "error must name the path it looked for so an operator sees the \
7351 wrong-root immediately, got: {msg}"
7352 );
7353 }
7354
7355 /// R844-B7 — the other half, and the reason the check is drawn at `.yah/`
7356 /// rather than at the machine list: a camp that declares no machines is a
7357 /// real workspace and must keep loading. Blanket-erroring on an empty
7358 /// fleet would conflate `unknown` with `answered with none`, which is the
7359 /// exact confusion the check exists to remove.
7360 #[test]
7361 fn a_workspace_with_no_machines_declared_still_loads() {
7362 let tmp = tempfile::TempDir::new().unwrap();
7363 std::fs::create_dir_all(tmp.path().join(".yah")).unwrap();
7364
7365 let cfg = CloudConfig::load(tmp.path())
7366 .expect("a `.yah/` with no infra/machines/ is an empty fleet, not a wrong root");
7367 assert!(cfg.machines.is_empty(), "nothing was declared");
7368 assert!(cfg.services.is_empty());
7369 assert!(cfg.providers.is_empty());
7370
7371 // And an existing-but-empty machines dir is the same answer, not a
7372 // second special case.
7373 std::fs::create_dir_all(crate::paths::machines_dir(tmp.path())).unwrap();
7374 let cfg = CloudConfig::load(tmp.path()).expect("an empty machines/ dir still loads");
7375 assert!(cfg.machines.is_empty());
7376 }
7377
7378 #[test]
7379 fn admit_workload_empty_selector_is_unconstrained() {
7380 // Absent annotation ⇒ no mesh-tag constraint ⇒ first declared machine
7381 // (pre-R594 behavior preserved).
7382 let cfg = build_worker_fleet();
7383 let ws = ws_with_selector(None);
7384 let picked = cfg.admit_workload(&ws).unwrap();
7385 assert_eq!(picked.name, "us-west-002");
7386 }
7387
7388 #[test]
7389 fn node_selector_mesh_tags_trims_and_drops_empties() {
7390 let ws = ws_with_selector(Some(" tag:build-worker , arch:x86 ,"));
7391 assert_eq!(
7392 node_selector_mesh_tags(&ws),
7393 vec!["tag:build-worker".to_string(), "arch:x86".to_string()]
7394 );
7395 assert!(node_selector_mesh_tags(&ws_with_selector(None)).is_empty());
7396 }
7397
7398 // ─── F16 topology-aware resolver ────────────────────────────────────────
7399
7400 fn two_region_fleet() -> CloudConfig {
7401 make_empty_cfg(vec![
7402 make_machine_topo(
7403 "us-west-001",
7404 "hetzner",
7405 "us-west",
7406 vec!["tag:cloud-runner"],
7407 ),
7408 make_machine_topo(
7409 "eu-west-001",
7410 "hetzner",
7411 "eu-west",
7412 vec!["tag:cloud-runner"],
7413 ),
7414 ])
7415 }
7416
7417 #[test]
7418 fn resolve_machine_matches_on_region_plus_mesh_tags() {
7419 let cfg = two_region_fleet();
7420 let req = RequiredSpec {
7421 regions: vec!["us-west".into()],
7422 mesh_tags: vec!["tag:cloud-runner".into()],
7423 ..Default::default()
7424 };
7425 let picked = cfg.resolve_machine(&req).unwrap();
7426 assert_eq!(picked.name, "us-west-001");
7427 }
7428
7429 #[test]
7430 fn resolve_machine_region_disambiguates_same_tag() {
7431 // Both boxes carry tag:cloud-runner; the region axis selects eu-west.
7432 let cfg = two_region_fleet();
7433 let req = RequiredSpec {
7434 regions: vec!["eu-west".into()],
7435 mesh_tags: vec!["tag:cloud-runner".into()],
7436 ..Default::default()
7437 };
7438 assert_eq!(cfg.resolve_machine(&req).unwrap().name, "eu-west-001");
7439 }
7440
7441 #[test]
7442 fn resolve_machine_fails_loud_with_constraint_summary() {
7443 let cfg = two_region_fleet();
7444 let req = RequiredSpec {
7445 regions: vec!["us-central".into()],
7446 mesh_tags: vec!["tag:cloud-runner".into()],
7447 ..Default::default()
7448 };
7449 let err = cfg.resolve_machine(&req).unwrap_err().to_string();
7450 assert!(err.contains("required.regions=[us-central]"), "got: {err}");
7451 assert!(
7452 err.contains("required.mesh_tags=[tag:cloud-runner]"),
7453 "got: {err}"
7454 );
7455 // Names the candidates it rejected.
7456 assert!(err.contains("us-west-001"), "got: {err}");
7457 }
7458
7459 #[test]
7460 fn resolve_machine_provider_axis_filters() {
7461 let cfg = make_empty_cfg(vec![
7462 make_machine_topo("aws-west-1", "aws", "us-west", vec!["tag:cloud-runner"]),
7463 make_machine_topo("hz-west-1", "hetzner", "us-west", vec!["tag:cloud-runner"]),
7464 ]);
7465 let req = RequiredSpec {
7466 regions: vec!["us-west".into()],
7467 providers: vec!["hetzner".into()],
7468 ..Default::default()
7469 };
7470 assert_eq!(cfg.resolve_machine(&req).unwrap().name, "hz-west-1");
7471 }
7472
7473 #[test]
7474 fn unconstrained_required_spec_matches_first_machine() {
7475 let cfg = two_region_fleet();
7476 assert!(RequiredSpec::default().is_unconstrained());
7477 assert_eq!(
7478 cfg.resolve_machine(&RequiredSpec::default()).unwrap().name,
7479 "us-west-001"
7480 );
7481 }
7482
7483 #[test]
7484 fn required_spec_parses_topology_axes_from_toml() {
7485 let toml_src = r#"
7486use = "hetzner-primary"
7487[required]
7488regions = ["us-west"]
7489mesh_tags = ["tag:cloud-runner"]
7490"#;
7491 let slot: MirrorProviderSlot = toml::from_str(toml_src).unwrap();
7492 let req = slot.required().expect("required block present");
7493 assert_eq!(req.regions, vec!["us-west"]);
7494 assert_eq!(req.mesh_tags, vec!["tag:cloud-runner"]);
7495 assert!(req.zones.is_empty());
7496 }
7497
7498 // ─── R844-F8 replica count ──────────────────────────────────────────────
7499
7500 fn three_runner_fleet() -> CloudConfig {
7501 make_empty_cfg(vec![
7502 make_machine_topo("us-east-001", "hetzner", "us-east", vec!["tag:cloud-runner"]),
7503 make_machine_topo(
7504 "us-south-001",
7505 "hetzner",
7506 "us-south",
7507 vec!["tag:cloud-runner"],
7508 ),
7509 make_machine_topo(
7510 "us-west-001",
7511 "hetzner",
7512 "us-west",
7513 vec!["tag:cloud-runner"],
7514 ),
7515 ])
7516 }
7517
7518 #[test]
7519 fn an_absent_replica_count_still_places_exactly_one_machine() {
7520 // The migration is additive: every mirror on disk omits `replicas`, and
7521 // must resolve byte-identically to the pre-R844-F8 answer.
7522 let cfg = three_runner_fleet();
7523 let req = RequiredSpec {
7524 mesh_tags: vec!["tag:cloud-runner".into()],
7525 ..Default::default()
7526 };
7527 assert_eq!(req.replica_count(), 1);
7528 let names: Vec<&str> = cfg
7529 .resolve_machines(&req)
7530 .unwrap()
7531 .iter()
7532 .map(|m| m.name.as_str())
7533 .collect();
7534 assert_eq!(names, vec!["us-east-001"]);
7535 assert_eq!(cfg.resolve_machine(&req).unwrap().name, "us-east-001");
7536 }
7537
7538 #[test]
7539 fn a_replica_count_places_that_many_machines_not_every_match() {
7540 // Three machines match; two are asked for; two are placed. Inferring the
7541 // count from the match count would make adding a box to the fleet
7542 // silently scale a production front door.
7543 let cfg = three_runner_fleet();
7544 let req = RequiredSpec {
7545 mesh_tags: vec!["tag:cloud-runner".into()],
7546 replicas: Some(2),
7547 ..Default::default()
7548 };
7549 let names: Vec<&str> = cfg
7550 .resolve_machines(&req)
7551 .unwrap()
7552 .iter()
7553 .map(|m| m.name.as_str())
7554 .collect();
7555 assert_eq!(names, vec!["us-east-001", "us-south-001"]);
7556 }
7557
7558 #[test]
7559 fn fewer_matches_than_replicas_is_an_error_naming_both_numbers() {
7560 // Never a partial placement: one of two reported as success is the
7561 // subset-that-looks-like-it-worked failure in its purest form.
7562 let cfg = three_runner_fleet();
7563 let req = RequiredSpec {
7564 regions: vec!["us-east".into()],
7565 replicas: Some(2),
7566 ..Default::default()
7567 };
7568 let err = cfg.resolve_machines(&req).unwrap_err().to_string();
7569 assert!(err.contains("only 1 of 2"), "got: {err}");
7570 assert!(err.contains("required.regions=[us-east]"), "got: {err}");
7571 // …and names the pool it searched, like every other placement refusal.
7572 assert!(err.contains("declared machines"), "got: {err}");
7573 assert!(err.contains("us-south-001"), "got: {err}");
7574 }
7575
7576 #[test]
7577 fn zero_replicas_is_refused_rather_than_placing_nothing() {
7578 let cfg = three_runner_fleet();
7579 let req = RequiredSpec {
7580 mesh_tags: vec!["tag:cloud-runner".into()],
7581 replicas: Some(0),
7582 ..Default::default()
7583 };
7584 let err = cfg.resolve_machines(&req).unwrap_err().to_string();
7585 assert!(err.contains("replicas = 0"), "got: {err}");
7586 }
7587
7588 #[test]
7589 fn replicas_parses_from_the_inline_required_form() {
7590 // The INLINE form specifically: `[providers.bundle.required]` as a table
7591 // HEADER ends the slot's table and reparents every key below it.
7592 let slot: MirrorProviderSlot = toml::from_str(
7593 r#"
7594use = "hetzner-primary"
7595port = 8080
7596required = { regions = ["us-east"], mesh_tags = ["tag:cloud-runner"], replicas = 2 }
7597"#,
7598 )
7599 .unwrap();
7600 assert_eq!(
7601 slot.fields().get("port").and_then(|v| v.as_integer()),
7602 Some(8080),
7603 "the inline form leaves the slot's other keys where they were"
7604 );
7605 let req = slot.required().expect("required block present");
7606 assert_eq!(req.replicas, Some(2));
7607 assert_eq!(req.replica_count(), 2);
7608 // A count is not a match axis — it says how many, not which.
7609 assert!(!req.is_unconstrained());
7610 assert!(RequiredSpec {
7611 replicas: Some(2),
7612 ..Default::default()
7613 }
7614 .is_unconstrained());
7615 }
7616
7617 #[test]
7618 fn round_trip_machine() {
7619 let cfg = MachineConfig {
7620 name: "test-pdx-1".into(),
7621 provider: "hetzner".into(),
7622 location: Some("pdx".into()),
7623 server_type: Some("cpx22".into()),
7624 hosts_mirrors: vec!["noisetable".into()],
7625 mesh_tags: vec!["region:pdx".into()],
7626 region: Some("us-west".into()),
7627 zone: Some("pdx".into()),
7628 arch: None,
7629 bucket: Some(BucketSpec {
7630 name: "test-assets-pdx-1".into(),
7631 public_read: false,
7632 }),
7633 vendor: None,
7634 nickname: None,
7635 legacy_hostkey_fingerprint: None,
7636 registration: Default::default(),
7637 ssh_keys: vec![],
7638 cloudflared: None,
7639 hosts_operator_bridge: false,
7640 connect: None,
7641 allocatable: None,
7642 taints: vec![],
7643 sovereign_group: None,
7644 sovereign_role: None,
7645 ingress_floating_ip: None,
7646 };
7647 let s = toml::to_string(&cfg).unwrap();
7648 let back: MachineConfig = toml::from_str(&s).unwrap();
7649 assert_eq!(back.name, cfg.name);
7650 assert_eq!(back.location, cfg.location);
7651 assert_eq!(back.region.as_deref(), Some("us-west"));
7652 assert_eq!(back.zone.as_deref(), Some("pdx"));
7653 }
7654
7655 #[test]
7656 fn round_trip_mirror() {
7657 let cfg = LegacyMirrorConfig {
7658 camp: "noisetable".into(),
7659 regions: vec!["pdx".into(), "iad".into()],
7660 workloads: vec!["asset-registry".into()],
7661 cloud_domain: None,
7662 };
7663 let s = toml::to_string(&cfg).unwrap();
7664 let back: LegacyMirrorConfig = toml::from_str(&s).unwrap();
7665 assert_eq!(back.camp, cfg.camp);
7666 assert_eq!(back.regions, cfg.regions);
7667 assert_eq!(back.workloads, cfg.workloads);
7668 }
7669
7670 #[test]
7671 fn mirror_serialises_as_camp_key() {
7672 // Serialised form should use `camp`, not `rig`.
7673 let cfg = LegacyMirrorConfig {
7674 camp: "noisetable".into(),
7675 regions: vec!["pdx".into()],
7676 workloads: vec![],
7677 cloud_domain: None,
7678 };
7679 let s = toml::to_string(&cfg).unwrap();
7680 assert!(
7681 s.contains("camp = "),
7682 "serialised key should be 'camp': {s}"
7683 );
7684 assert!(!s.contains("rig = "), "old key should not appear: {s}");
7685 }
7686
7687 #[test]
7688 fn mirror_rig_alias_still_loads() {
7689 // Old mirrors/*.toml files use `rig = "..."` before the R137 rename;
7690 // the alias keeps them loading until the one-time `sed` migration runs.
7691 let toml_str =
7692 "rig = \"noisetable\"\nregions = [\"pdx\"]\nworkloads = [\"asset-registry\"]\n";
7693 let cfg: LegacyMirrorConfig = toml::from_str(toml_str).unwrap();
7694 assert_eq!(cfg.camp, "noisetable");
7695 }
7696
7697 #[test]
7698 fn mirror_services_alias_still_loads() {
7699 // Old mirrors/*.toml files use `services = [...]`; the alias keeps them
7700 // loading without a migration step.
7701 let toml_str =
7702 "camp = \"noisetable\"\nregions = [\"pdx\"]\nservices = [\"asset-registry\"]\n";
7703 let cfg: LegacyMirrorConfig = toml::from_str(toml_str).unwrap();
7704 assert_eq!(cfg.workloads, vec!["asset-registry"]);
7705 }
7706
7707 #[test]
7708 fn load_dir_missing_is_empty() {
7709 let dir = std::path::PathBuf::from("/nonexistent/path");
7710 let result: Vec<MachineConfig> = load_dir(dir).unwrap();
7711 assert!(result.is_empty());
7712 }
7713
7714 #[test]
7715 fn topology_round_trip() {
7716 let topo = TopologyConfig {
7717 assignments: vec![
7718 MirrorAssignment {
7719 mirror: "noisetable-pdx".into(),
7720 machine: "noisetable-pdx-1".into(),
7721 },
7722 MirrorAssignment {
7723 mirror: "noisetable-iad".into(),
7724 machine: "noisetable-iad-1".into(),
7725 },
7726 ],
7727 buckets: vec![],
7728 };
7729 let s = toml::to_string(&topo).unwrap();
7730 let back: TopologyConfig = toml::from_str(&s).unwrap();
7731 assert_eq!(back.assignments.len(), 2);
7732 assert_eq!(back.assignments[0].mirror, "noisetable-pdx");
7733 assert_eq!(back.assignments[1].machine, "noisetable-iad-1");
7734 }
7735
7736 #[test]
7737 fn topology_absent_returns_default() {
7738 let tmp = tempfile::TempDir::new().unwrap();
7739 let path = tmp.path().join("topology.toml");
7740 // file doesn't exist
7741 let topo = load_topology(path).unwrap();
7742 assert!(topo.assignments.is_empty());
7743 }
7744
7745 /// Helper: lay out a `<workspace_root>/.yah/cloud/` legacy tree for the
7746 /// pre-R215 cargo tests below; returns the legacy cloud_dir for writes.
7747 fn make_legacy_cloud_dir(root: &std::path::Path) -> std::path::PathBuf {
7748 let cloud_dir = root.join(".yah").join("cloud");
7749 std::fs::create_dir_all(&cloud_dir).unwrap();
7750 cloud_dir
7751 }
7752
7753 #[test]
7754 fn cloud_config_load_and_lookup() {
7755 let tmp = tempfile::TempDir::new().unwrap();
7756 let root = tmp.path();
7757 let cloud_dir = make_legacy_cloud_dir(root);
7758
7759 let machine = MachineConfig {
7760 name: "noisetable-pdx-1".into(),
7761 provider: "hetzner".into(),
7762 location: Some("pdx".into()),
7763 server_type: Some("cpx22".into()),
7764 hosts_mirrors: vec!["noisetable".into(), "yah".into()],
7765 mesh_tags: vec!["region:pdx".into(), "tier:t2".into()],
7766 region: None,
7767 zone: None,
7768 arch: None,
7769 bucket: Some(BucketSpec {
7770 name: "noisetable-assets-pdx-1".into(),
7771 public_read: false,
7772 }),
7773 vendor: None,
7774 nickname: None,
7775 legacy_hostkey_fingerprint: None,
7776 registration: Default::default(),
7777 ssh_keys: vec![],
7778 cloudflared: None,
7779 hosts_operator_bridge: false,
7780 connect: None,
7781 allocatable: None,
7782 taints: vec![],
7783 sovereign_group: None,
7784 sovereign_role: None,
7785 ingress_floating_ip: None,
7786 };
7787 // Land in the legacy tree so the legacy machine loader picks it up.
7788 machine.save(&cloud_dir).unwrap();
7789
7790 let mirror_toml = "camp = \"noisetable\"\nregions = [\"pdx\", \"iad\", \"fsn\"]\nworkloads = [\"asset-registry\"]\n";
7791 std::fs::create_dir_all(cloud_dir.join("mirrors")).unwrap();
7792 std::fs::write(cloud_dir.join("mirrors/noisetable.toml"), mirror_toml).unwrap();
7793
7794 let cfg = CloudConfig::load(root).unwrap();
7795
7796 assert_eq!(cfg.machines.len(), 1);
7797 assert_eq!(cfg.legacy_mirrors.len(), 1);
7798 assert_eq!(cfg.workloads.len(), 0); // no workloads/ dir yet
7799 assert!(cfg.services.is_empty(), "no R215+ services/ tree");
7800 assert!(cfg.providers.is_empty(), "no R215+ providers/ tree");
7801
7802 let m = cfg.machine("noisetable-pdx-1").unwrap();
7803 assert_eq!(m.location(), "pdx");
7804 assert_eq!(m.bucket.as_ref().unwrap().name, "noisetable-assets-pdx-1");
7805
7806 let mir = cfg.legacy_mirror("noisetable").unwrap();
7807 assert_eq!(mir.regions, vec!["pdx", "iad", "fsn"]);
7808 assert_eq!(mir.workloads, vec!["asset-registry"]);
7809 }
7810
7811 #[test]
7812 fn mirror_folder_layout_loads() {
7813 // Folder layout: mirrors/<id>/mirror.toml — new preferred form.
7814 let tmp = tempfile::TempDir::new().unwrap();
7815 let root = tmp.path();
7816 let cloud_dir = make_legacy_cloud_dir(root);
7817 let mirror_dir = cloud_dir.join("mirrors").join("yah-com");
7818 std::fs::create_dir_all(&mirror_dir).unwrap();
7819 std::fs::write(
7820 mirror_dir.join("mirror.toml"),
7821 "camp = \"yah\"\nregions = [\"pdx\"]\nworkloads = [\"yah-web\"]\n",
7822 )
7823 .unwrap();
7824
7825 let cfg = CloudConfig::load(root).unwrap();
7826 assert_eq!(cfg.legacy_mirrors.len(), 1);
7827 let mir = cfg.legacy_mirror("yah").unwrap();
7828 assert_eq!(mir.camp, "yah");
7829 assert_eq!(mir.workloads, vec!["yah-web"]);
7830 }
7831
7832 #[test]
7833 fn mirror_folder_and_flat_coexist() {
7834 // Both layouts may coexist in the same mirrors/ directory.
7835 let tmp = tempfile::TempDir::new().unwrap();
7836 let root = tmp.path();
7837 let cloud_dir = make_legacy_cloud_dir(root);
7838 let mirrors_root = cloud_dir.join("mirrors");
7839 std::fs::create_dir_all(&mirrors_root).unwrap();
7840
7841 // Flat legacy mirror
7842 std::fs::write(
7843 mirrors_root.join("noisetable.toml"),
7844 "camp = \"noisetable\"\nregions = [\"pdx\"]\nworkloads = []\n",
7845 )
7846 .unwrap();
7847
7848 // Folder-form mirror
7849 let yah_com_dir = mirrors_root.join("yah-com");
7850 std::fs::create_dir_all(&yah_com_dir).unwrap();
7851 std::fs::write(
7852 yah_com_dir.join("mirror.toml"),
7853 "camp = \"yah\"\nregions = [\"pdx\"]\nworkloads = []\n",
7854 )
7855 .unwrap();
7856
7857 let cfg = CloudConfig::load(root).unwrap();
7858 assert_eq!(cfg.legacy_mirrors.len(), 2);
7859 assert!(cfg.legacy_mirror("noisetable").is_some());
7860 assert!(cfg.legacy_mirror("yah").is_some());
7861 }
7862
7863 #[test]
7864 fn mirror_malformed_fails_with_field_path() {
7865 // A malformed mirror.toml should fail at load with a clear error
7866 // that includes the file path.
7867 let tmp = tempfile::TempDir::new().unwrap();
7868 let root = tmp.path();
7869 let cloud_dir = make_legacy_cloud_dir(root);
7870 let mirror_dir = cloud_dir.join("mirrors").join("bad");
7871 std::fs::create_dir_all(&mirror_dir).unwrap();
7872 // Missing required `camp` field
7873 std::fs::write(
7874 mirror_dir.join("mirror.toml"),
7875 "regions = [\"pdx\"]\nworkloads = []\n",
7876 )
7877 .unwrap();
7878
7879 let err = CloudConfig::load(root).unwrap_err();
7880 let msg = err.to_string();
7881 assert!(
7882 msg.contains("mirror.toml"),
7883 "error should reference the file path, got: {msg}"
7884 );
7885 }
7886
7887 #[test]
7888 fn workload_config_load_and_validate() {
7889 use workload_spec::{
7890 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
7891 RestartPolicy, StopPolicy, TenantId, TierTag, WorkloadSpec,
7892 };
7893
7894 let tmp = tempfile::TempDir::new().unwrap();
7895 let root = tmp.path();
7896 let cloud_dir = make_legacy_cloud_dir(root);
7897 std::fs::create_dir_all(cloud_dir.join("workloads")).unwrap();
7898
7899 let spec = WorkloadSpec {
7900 name: "asset-registry".into(),
7901 image: ImageRef {
7902 registry: "ghcr.io".into(),
7903 repository: "noisetable/asset-registry".into(),
7904 tag: "v1.0.0".into(),
7905 digest: workload_spec::testing::test_digest(),
7906 },
7907 tier: TierTag("tenant".into()),
7908 replicas: 1,
7909 command: None,
7910 entrypoint: None,
7911 workdir: None,
7912 user: None,
7913 env: vec![],
7914 secrets: vec![],
7915 volumes: vec![],
7916 resources: ResourceLimits {
7917 memory_mb: 256,
7918 cpu_millis: 512,
7919 memory_request_mb: None,
7920 cpu_limit_millis: None,
7921 pids_max: None,
7922 scratch_floor_mb: None,
7923 },
7924 depends_on: vec![],
7925 requires: vec![],
7926 healthcheck: None,
7927 restart_policy: RestartPolicy::Always,
7928 archetype: None,
7929 stop_policy: StopPolicy {
7930 signal: 15,
7931 grace_period: workload_spec::Millis::from_secs(10),
7932 },
7933 expose: ExposeSpec {
7934 mesh: MeshExpose {
7935 identity: MeshIdent("asset-registry.pdx".into()),
7936 ports: MeshExpose::anonymous_ports([8080]),
7937 allow_from: vec![],
7938 },
7939 public: None,
7940 operator: None,
7941 },
7942 tenant: TenantId::singleton(),
7943 namespace: NamespaceId::singleton(),
7944 labels: Default::default(),
7945 durability: None,
7946 annotations: Default::default(),
7947 files: Vec::new(),
7948 };
7949
7950 let toml_str = toml::to_string_pretty(&spec).unwrap();
7951 std::fs::write(cloud_dir.join("workloads/asset-registry.toml"), &toml_str).unwrap();
7952
7953 let cfg = CloudConfig::load(root).unwrap();
7954 assert_eq!(cfg.workloads.len(), 1);
7955 assert_eq!(cfg.workloads[0].spec.name, "asset-registry");
7956 assert_eq!(cfg.workload("asset-registry").unwrap().spec.replicas, 1);
7957 }
7958
7959 /// Minimal valid spec for the R215+ loader tests below. Kept as a helper so
7960 /// the two tests differ only in *where* the file lands, which is the whole
7961 /// thing under test.
7962 #[cfg(test)]
7963 fn minimal_spec(name: &str, replicas: u32) -> workload_spec::WorkloadSpec {
7964 use workload_spec::{
7965 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
7966 RestartPolicy, StopPolicy, TenantId, TierTag, WorkloadSpec,
7967 };
7968 WorkloadSpec {
7969 name: name.into(),
7970 image: ImageRef {
7971 registry: "cr.yah.dev".into(),
7972 repository: name.into(),
7973 tag: "v1".into(),
7974 digest: workload_spec::testing::test_digest(),
7975 },
7976 tier: TierTag("infra".into()),
7977 replicas,
7978 command: None,
7979 entrypoint: None,
7980 workdir: None,
7981 user: None,
7982 env: vec![],
7983 secrets: vec![],
7984 volumes: vec![],
7985 resources: ResourceLimits {
7986 memory_mb: 256,
7987 cpu_millis: 250,
7988 memory_request_mb: None,
7989 cpu_limit_millis: None,
7990 pids_max: None,
7991 scratch_floor_mb: None,
7992 },
7993 depends_on: vec![],
7994 requires: vec![],
7995 healthcheck: None,
7996 restart_policy: RestartPolicy::Always,
7997 archetype: None,
7998 stop_policy: StopPolicy {
7999 signal: 15,
8000 grace_period: workload_spec::Millis::from_secs(10),
8001 },
8002 expose: ExposeSpec {
8003 mesh: MeshExpose {
8004 identity: MeshIdent(name.into()),
8005 ports: MeshExpose::anonymous_ports([4325]),
8006 allow_from: vec![],
8007 },
8008 public: None,
8009 operator: None,
8010 },
8011 tenant: TenantId::singleton(),
8012 namespace: NamespaceId::singleton(),
8013 labels: Default::default(),
8014 durability: None,
8015 annotations: Default::default(),
8016 files: Vec::new(),
8017 }
8018 }
8019
8020 /// R568-T7. Workloads must load from the R215+ tree.
8021 ///
8022 /// Before the fix this function tested, `CloudConfig::load` read workloads
8023 /// ONLY from the pre-R215 `.yah/cloud/workloads/` — which R222-B1 emptied —
8024 /// so in any modern camp `cfg.workload(name)` returned `None` for every
8025 /// name and the entire `yah cloud workload …` surface was unreachable. The
8026 /// CLI's own error text has said `.yah/infra/workloads/` throughout, so the
8027 /// bug read as "you must have typoed the filename".
8028 ///
8029 /// Note the fixture writes NO legacy `.yah/cloud/` dir at all: that is the
8030 /// shape of a real post-R215 camp, and it is exactly the shape the old code
8031 /// could not serve.
8032 #[test]
8033 fn workloads_load_from_the_infra_tree() {
8034 let tmp = tempfile::TempDir::new().unwrap();
8035 let root = tmp.path();
8036 let dir = crate::paths::workloads_dir(root);
8037 std::fs::create_dir_all(&dir).unwrap();
8038 std::fs::write(
8039 dir.join("yah-cloud-admin.toml"),
8040 toml::to_string_pretty(&minimal_spec("yah-cloud-admin", 1)).unwrap(),
8041 )
8042 .unwrap();
8043
8044 let cfg = CloudConfig::load(root).unwrap();
8045 assert_eq!(cfg.workloads.len(), 1);
8046 assert_eq!(
8047 cfg.workload("yah-cloud-admin").unwrap().spec.replicas,
8048 1,
8049 "a workload declared under .yah/infra/workloads/ must be resolvable by name"
8050 );
8051 }
8052
8053 /// A camp mid-migration can have both trees. R215+ wins on a name
8054 /// collision — same precedence the machine loader applies — so moving a
8055 /// declaration into `.yah/infra/workloads/` takes effect immediately
8056 /// instead of being silently shadowed by the copy left behind.
8057 #[test]
8058 fn infra_workload_shadows_the_legacy_copy_of_the_same_name() {
8059 let tmp = tempfile::TempDir::new().unwrap();
8060 let root = tmp.path();
8061
8062 let legacy = make_legacy_cloud_dir(root);
8063 std::fs::create_dir_all(legacy.join("workloads")).unwrap();
8064 std::fs::write(
8065 legacy.join("workloads/shared.toml"),
8066 toml::to_string_pretty(&minimal_spec("shared", 9)).unwrap(),
8067 )
8068 .unwrap();
8069 // Legacy-only name, to prove the old tree is still read rather than
8070 // replaced wholesale.
8071 std::fs::write(
8072 legacy.join("workloads/legacy-only.toml"),
8073 toml::to_string_pretty(&minimal_spec("legacy-only", 3)).unwrap(),
8074 )
8075 .unwrap();
8076
8077 let infra = crate::paths::workloads_dir(root);
8078 std::fs::create_dir_all(&infra).unwrap();
8079 std::fs::write(
8080 infra.join("shared.toml"),
8081 toml::to_string_pretty(&minimal_spec("shared", 1)).unwrap(),
8082 )
8083 .unwrap();
8084
8085 let cfg = CloudConfig::load(root).unwrap();
8086 assert_eq!(cfg.workloads.len(), 2, "one `shared`, plus `legacy-only`");
8087 assert_eq!(
8088 cfg.workload("shared").unwrap().spec.replicas,
8089 1,
8090 "the .yah/infra/ copy must win over the legacy one"
8091 );
8092 assert_eq!(cfg.workload("legacy-only").unwrap().spec.replicas, 3);
8093 }
8094
8095 #[test]
8096 fn workload_loader_rejects_bad_spec() {
8097 use workload_spec::{
8098 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
8099 RestartPolicy, StopPolicy, TenantId, TierTag, WorkloadSpec,
8100 };
8101
8102 let tmp = tempfile::TempDir::new().unwrap();
8103 let root = tmp.path();
8104 let cloud_dir = make_legacy_cloud_dir(root);
8105 std::fs::create_dir_all(cloud_dir.join("workloads")).unwrap();
8106
8107 // Construct a spec that round-trips through TOML but fails shape
8108 // validation: replicas = 200 is above the max of 100.
8109 let mut spec = WorkloadSpec {
8110 name: "asset-registry".into(),
8111 image: ImageRef {
8112 registry: "ghcr.io".into(),
8113 repository: "test/app".into(),
8114 tag: "v1".into(),
8115 digest: workload_spec::testing::test_digest(),
8116 },
8117 tier: TierTag("tenant".into()),
8118 replicas: 200, // ← invalid: exceeds max 100
8119 command: None,
8120 entrypoint: None,
8121 workdir: None,
8122 user: None,
8123 env: vec![],
8124 secrets: vec![],
8125 volumes: vec![],
8126 resources: ResourceLimits {
8127 memory_mb: 256,
8128 cpu_millis: 512,
8129 memory_request_mb: None,
8130 cpu_limit_millis: None,
8131 pids_max: None,
8132 scratch_floor_mb: None,
8133 },
8134 depends_on: vec![],
8135 requires: vec![],
8136 healthcheck: None,
8137 restart_policy: RestartPolicy::Always,
8138 archetype: None,
8139 stop_policy: StopPolicy {
8140 signal: 15,
8141 grace_period: workload_spec::Millis::from_secs(10),
8142 },
8143 expose: ExposeSpec {
8144 mesh: MeshExpose {
8145 identity: MeshIdent("asset-registry.pdx".into()),
8146 ports: MeshExpose::anonymous_ports([8080]),
8147 allow_from: vec![],
8148 },
8149 public: None,
8150 operator: None,
8151 },
8152 tenant: TenantId::singleton(),
8153 namespace: NamespaceId::singleton(),
8154 labels: Default::default(),
8155 durability: None,
8156 annotations: Default::default(),
8157 files: Vec::new(),
8158 };
8159
8160 let toml_str = toml::to_string_pretty(&spec).unwrap();
8161 std::fs::write(cloud_dir.join("workloads/bad.toml"), &toml_str).unwrap();
8162
8163 let result = CloudConfig::load(root);
8164 assert!(
8165 result.is_err(),
8166 "loading a WorkloadSpec with replicas=200 should return Err"
8167 );
8168 let msg = result.unwrap_err().to_string();
8169 assert!(
8170 msg.contains("shape validation")
8171 || msg.contains("Replicas")
8172 || msg.contains("replicas"),
8173 "error should mention shape validation or replicas field, got: {msg}"
8174 );
8175
8176 // The `spec` binding is only used for the write — suppress warning.
8177 let _ = &mut spec;
8178 }
8179
8180 #[test]
8181 fn workload_config_save_round_trip() {
8182 use workload_spec::{
8183 ExposeSpec, ImageRef, MeshExpose, MeshIdent, NamespaceId, ResourceLimits,
8184 RestartPolicy, StopPolicy, TenantId, TierTag, WorkloadSpec,
8185 };
8186
8187 let tmp = tempfile::TempDir::new().unwrap();
8188 let root = tmp.path();
8189
8190 let spec = WorkloadSpec {
8191 name: "signing-service".into(),
8192 image: ImageRef {
8193 registry: "ghcr.io".into(),
8194 repository: "noisetable/signing".into(),
8195 tag: "v2.0.0".into(),
8196 digest: workload_spec::testing::test_digest(),
8197 },
8198 tier: TierTag("private".into()),
8199 replicas: 2,
8200 command: None,
8201 entrypoint: None,
8202 workdir: None,
8203 user: None,
8204 env: vec![],
8205 secrets: vec![],
8206 volumes: vec![],
8207 resources: ResourceLimits {
8208 memory_mb: 128,
8209 cpu_millis: 256,
8210 memory_request_mb: None,
8211 cpu_limit_millis: None,
8212 pids_max: None,
8213 scratch_floor_mb: None,
8214 },
8215 depends_on: vec![],
8216 requires: vec![],
8217 healthcheck: None,
8218 restart_policy: RestartPolicy::Always,
8219 archetype: None,
8220 stop_policy: StopPolicy {
8221 signal: 15,
8222 grace_period: workload_spec::Millis::from_secs(5),
8223 },
8224 expose: ExposeSpec {
8225 mesh: MeshExpose {
8226 identity: MeshIdent("signing.pdx".into()),
8227 ports: MeshExpose::anonymous_ports([9090]),
8228 allow_from: vec![],
8229 },
8230 public: None,
8231 operator: None,
8232 },
8233 tenant: TenantId::singleton(),
8234 namespace: NamespaceId::singleton(),
8235 labels: Default::default(),
8236 durability: None,
8237 annotations: Default::default(),
8238 files: Vec::new(),
8239 };
8240
8241 let wc = WorkloadConfig { spec };
8242 let cloud_dir = make_legacy_cloud_dir(root);
8243 wc.save(&cloud_dir).unwrap();
8244
8245 let loaded = CloudConfig::load(root).unwrap();
8246 assert_eq!(loaded.workloads.len(), 1);
8247 assert_eq!(loaded.workloads[0].spec.name, "signing-service");
8248 assert_eq!(loaded.workloads[0].spec.replicas, 2);
8249 }
8250
8251 #[test]
8252 fn machine_save_write_back_fingerprint() {
8253 let tmp = tempfile::TempDir::new().unwrap();
8254 let root = tmp.path();
8255
8256 let mut machine = MachineConfig {
8257 name: "test-pdx-1".into(),
8258 provider: "hetzner".into(),
8259 location: Some("pdx".into()),
8260 server_type: Some("cpx22".into()),
8261 hosts_mirrors: vec![],
8262 mesh_tags: vec![],
8263 region: None,
8264 zone: None,
8265 arch: None,
8266 bucket: None,
8267 vendor: None,
8268 nickname: None,
8269 legacy_hostkey_fingerprint: None,
8270 registration: Default::default(),
8271 ssh_keys: vec![],
8272 cloudflared: None,
8273 hosts_operator_bridge: false,
8274 connect: None,
8275 allocatable: None,
8276 taints: vec![],
8277 sovereign_group: None,
8278 sovereign_role: None,
8279 ingress_floating_ip: None,
8280 };
8281 machine.save(root).unwrap();
8282
8283 // Simulate A4: write back the hostkey fingerprint after provision.
8284 // R707-T1: registration is the write target; the accessor is the read.
8285 machine.registration.hostkey_fingerprint = Some("SHA256:abc123".into());
8286 machine.save(root).unwrap();
8287
8288 let reloaded: Vec<MachineConfig> = load_dir(root.join("machines")).unwrap();
8289 assert_eq!(reloaded.len(), 1);
8290 assert_eq!(reloaded[0].hostkey_fingerprint(), Some("SHA256:abc123"));
8291 }
8292
8293 // ─── New-shape (R222 B2) parse tests ────────────────────────────────────
8294 //
8295 // These mirror the Phase-A manifests committed under `.yah/services/` and
8296 // `.yah/infra/providers/`. Keeping the test strings inline (rather than
8297 // reading the on-disk files) so the loader stays runnable in any workdir
8298 // and so accidental edits to the on-disk files don't silently change
8299 // schema expectations.
8300
8301 #[test]
8302 fn provider_cloudflare_round_trips() {
8303 let src = r#"
8304schema_version = 1
8305id = "cloudflare"
8306kind = "cloudflare"
8307credentials = "keystore://cloudflare/yah"
8308default_zone = "yah.dev"
8309"#;
8310 let cfg: ProviderConfig = toml::from_str(src).unwrap();
8311 assert_eq!(cfg.id, "cloudflare");
8312 assert_eq!(cfg.kind, Provider::Cloudflare);
8313 assert_eq!(
8314 cfg.credentials.as_deref(),
8315 Some("keystore://cloudflare/yah")
8316 );
8317 assert_eq!(
8318 cfg.fields.get("default_zone").and_then(|v| v.as_str()),
8319 Some("yah.dev"),
8320 );
8321 let back = toml::to_string(&cfg).unwrap();
8322 let again: ProviderConfig = toml::from_str(&back).unwrap();
8323 assert_eq!(again.id, cfg.id);
8324 assert_eq!(again.kind, cfg.kind);
8325 }
8326
8327 #[test]
8328 fn provider_hetzner_round_trips() {
8329 let src = r#"
8330schema_version = 1
8331id = "hetzner"
8332kind = "hetzner"
8333credentials = "keystore://hetzner/yah"
8334default_location = "pdx"
8335default_server_type = "cpx11"
8336ssh_keys = []
8337"#;
8338 let cfg: ProviderConfig = toml::from_str(src).unwrap();
8339 assert_eq!(cfg.kind, Provider::Hetzner);
8340 assert_eq!(
8341 cfg.fields.get("default_location").and_then(|v| v.as_str()),
8342 Some("pdx"),
8343 );
8344 assert!(
8345 cfg.fields
8346 .get("ssh_keys")
8347 .map(|v| v.as_array().unwrap().is_empty())
8348 .unwrap_or(false),
8349 "ssh_keys must round-trip as empty array, got {:?}",
8350 cfg.fields.get("ssh_keys"),
8351 );
8352 }
8353
8354 #[test]
8355 fn provider_orbstack_local_container_round_trips() {
8356 let src = r#"
8357schema_version = 1
8358id = "orbstack"
8359kind = "local-container"
8360runtime = "auto"
8361
8362[discovery]
8363orbstack = "~/.orbstack/run/docker.sock"
8364colima = "~/.colima/default/docker.sock"
8365docker = "/var/run/docker.sock"
8366"#;
8367 let cfg: ProviderConfig = toml::from_str(src).unwrap();
8368 assert_eq!(cfg.kind, Provider::LocalContainer);
8369 assert_eq!(
8370 cfg.fields.get("runtime").and_then(|v| v.as_str()),
8371 Some("auto"),
8372 );
8373 let discovery = cfg
8374 .fields
8375 .get("discovery")
8376 .and_then(|v| v.as_table())
8377 .expect("discovery table");
8378 assert!(discovery.contains_key("orbstack"));
8379 assert!(discovery.contains_key("colima"));
8380 assert!(discovery.contains_key("docker"));
8381 }
8382
8383 #[test]
8384 fn provider_unknown_kind_fails() {
8385 let src = r#"
8386schema_version = 1
8387id = "made-up"
8388kind = "fly-io"
8389"#;
8390 let err = toml::from_str::<ProviderConfig>(src).unwrap_err();
8391 let msg = err.to_string();
8392 assert!(
8393 msg.contains("kind") || msg.contains("variant"),
8394 "unknown provider kind should surface as a serde error, got: {msg}"
8395 );
8396 }
8397
8398 #[test]
8399 fn service_dev_yah_round_trips() {
8400 let src = r#"
8401schema_version = 1
8402name = "dev-yah"
8403[address]
8404kind = "front-door"
8405domain = "yah.dev"
8406
8407[[components]]
8408id = "site"
8409kind = "mesofact-static"
8410path = "app/yah/web"
8411role = "static"
8412"#;
8413 let cfg: ServiceConfig = toml::from_str(src).unwrap();
8414 assert_eq!(cfg.name, "dev-yah");
8415 assert_eq!(cfg.domain(), Some("yah.dev"));
8416 assert_eq!(cfg.components.len(), 1);
8417 let c = &cfg.components[0];
8418 assert_eq!(c.id, "site");
8419 assert_eq!(c.kind, "mesofact-static");
8420 assert_eq!(c.path, "app/yah/web");
8421 assert_eq!(c.role, "static");
8422 assert!(c.publishes.is_none());
8423
8424 let back = toml::to_string(&cfg).unwrap();
8425 let again: ServiceConfig = toml::from_str(&back).unwrap();
8426 assert_eq!(again.name, cfg.name);
8427 assert_eq!(again.components[0].kind, c.kind);
8428 }
8429
8430 #[test]
8431 fn mirror_prod_cloudflare_reference_parses() {
8432 let src = r#"
8433schema_version = 1
8434shape = "single-machine"
8435
8436[providers.static]
8437use = "cloudflare"
8438bucket = "yah-dev"
8439zone = "yah.dev"
8440dns = { record = "@", type = "CNAME" }
8441"#;
8442 let cfg: MirrorConfig = toml::from_str(src).unwrap();
8443 assert_eq!(cfg.shape, MirrorShape::SingleMachine);
8444 let slot = cfg.providers.get("static").expect("static slot");
8445 assert_eq!(slot.provider_id(), Some("cloudflare"));
8446 assert!(slot.inline_kind().is_none());
8447 if let MirrorProviderSlot::Reference { fields, .. } = slot {
8448 assert_eq!(
8449 fields.get("bucket").and_then(|v| v.as_str()),
8450 Some("yah-dev")
8451 );
8452 assert_eq!(fields.get("zone").and_then(|v| v.as_str()), Some("yah.dev"));
8453 let dns = fields
8454 .get("dns")
8455 .and_then(|v| v.as_table())
8456 .expect("dns table");
8457 assert_eq!(dns.get("record").and_then(|v| v.as_str()), Some("@"));
8458 assert_eq!(dns.get("type").and_then(|v| v.as_str()), Some("CNAME"));
8459 } else {
8460 panic!("expected Reference slot");
8461 }
8462 }
8463
8464 #[test]
8465 fn mirror_local_inline_static_and_orbstack_compute_parse() {
8466 let src = r#"
8467schema_version = 1
8468shape = "local"
8469
8470[providers.static]
8471kind = "miniflare-native"
8472port = 4321
8473artifact_dir = ".yah/infra/state/local/static"
8474
8475[providers.compute]
8476use = "orbstack"
8477"#;
8478 let cfg: MirrorConfig = toml::from_str(src).unwrap();
8479 assert_eq!(cfg.shape, MirrorShape::Local);
8480
8481 let static_slot = cfg.providers.get("static").expect("static slot");
8482 assert_eq!(static_slot.inline_kind(), Some(Provider::MiniflareNative));
8483 assert!(static_slot.provider_id().is_none());
8484 if let MirrorProviderSlot::Inline { fields, .. } = static_slot {
8485 assert_eq!(fields.get("port").and_then(|v| v.as_integer()), Some(4321));
8486 assert_eq!(
8487 fields.get("artifact_dir").and_then(|v| v.as_str()),
8488 Some(".yah/infra/state/local/static"),
8489 );
8490 } else {
8491 panic!("expected Inline slot for static");
8492 }
8493
8494 let compute_slot = cfg.providers.get("compute").expect("compute slot");
8495 assert_eq!(compute_slot.provider_id(), Some("orbstack"));
8496 }
8497
8498 #[test]
8499 fn mirror_pond_miniflare_minio_parse() {
8500 // pond-tier mirror: miniflare-container + minio, both inline.
8501 // T1 just needs these inline kinds to parse — the reconciler dispatch
8502 // arrives in R256-T3.
8503 let src = r#"
8504schema_version = 1
8505shape = "local"
8506
8507[providers.static]
8508kind = "miniflare-container"
8509port = 4322
8510bucket = "yah-dev"
8511
8512[providers.object_store]
8513kind = "minio-container"
8514api_port = 9000
8515console_port = 9001
8516bucket = "yah-dev"
8517"#;
8518 let cfg: MirrorConfig = toml::from_str(src).unwrap();
8519 assert_eq!(cfg.shape, MirrorShape::Local);
8520
8521 let static_slot = cfg.providers.get("static").expect("static slot");
8522 assert_eq!(
8523 static_slot.inline_kind(),
8524 Some(Provider::MiniflareContainer)
8525 );
8526 if let MirrorProviderSlot::Inline { fields, .. } = static_slot {
8527 assert_eq!(fields.get("port").and_then(|v| v.as_integer()), Some(4322));
8528 assert_eq!(
8529 fields.get("bucket").and_then(|v| v.as_str()),
8530 Some("yah-dev")
8531 );
8532 } else {
8533 panic!("expected Inline slot for miniflare-container static");
8534 }
8535
8536 let object_store_slot = cfg
8537 .providers
8538 .get("object_store")
8539 .expect("object_store slot");
8540 assert_eq!(
8541 object_store_slot.inline_kind(),
8542 Some(Provider::MinioContainer)
8543 );
8544 if let MirrorProviderSlot::Inline { fields, .. } = object_store_slot {
8545 assert_eq!(
8546 fields.get("api_port").and_then(|v| v.as_integer()),
8547 Some(9000)
8548 );
8549 assert_eq!(
8550 fields.get("console_port").and_then(|v| v.as_integer()),
8551 Some(9001)
8552 );
8553 assert_eq!(
8554 fields.get("bucket").and_then(|v| v.as_str()),
8555 Some("yah-dev")
8556 );
8557 } else {
8558 panic!("expected Inline slot for minio-container object_store");
8559 }
8560 }
8561
8562 #[test]
8563 fn provider_miniflare_container_kind_round_trips() {
8564 // Inline-only kind; never declared as a standalone provider file but
8565 // the enum round-trip is still exercised through ProviderConfig because
8566 // schemars/serde share the variant table.
8567 let cfg = MirrorProviderSlot::Inline {
8568 kind: Provider::MiniflareContainer,
8569 fields: BTreeMap::new(),
8570 };
8571 let s = toml::to_string(&cfg).unwrap();
8572 assert!(
8573 s.contains("kind = \"miniflare-container\""),
8574 "kebab-case wire form expected, got: {s}"
8575 );
8576 let back: MirrorProviderSlot = toml::from_str(&s).unwrap();
8577 assert_eq!(back.inline_kind(), Some(Provider::MiniflareContainer));
8578 }
8579
8580 #[test]
8581 fn provider_minio_container_kind_round_trips() {
8582 let cfg = MirrorProviderSlot::Inline {
8583 kind: Provider::MinioContainer,
8584 fields: BTreeMap::new(),
8585 };
8586 let s = toml::to_string(&cfg).unwrap();
8587 assert!(
8588 s.contains("kind = \"minio-container\""),
8589 "kebab-case wire form expected, got: {s}"
8590 );
8591 let back: MirrorProviderSlot = toml::from_str(&s).unwrap();
8592 assert_eq!(back.inline_kind(), Some(Provider::MinioContainer));
8593 }
8594
8595 #[test]
8596 fn mirror_compute_slot_with_machine_reference_parses() {
8597 // The on-disk prod.toml has a commented-out compute slot; this test
8598 // covers the form Phase B will need once yubaba is provisioned.
8599 let src = r#"
8600schema_version = 1
8601shape = "single-machine"
8602
8603[providers.compute]
8604use = "hetzner"
8605machine = "yah-cloud-1"
8606"#;
8607 let cfg: MirrorConfig = toml::from_str(src).unwrap();
8608 let slot = cfg.providers.get("compute").expect("compute slot");
8609 assert_eq!(slot.provider_id(), Some("hetzner"));
8610 if let MirrorProviderSlot::Reference { fields, .. } = slot {
8611 assert_eq!(
8612 fields.get("machine").and_then(|v| v.as_str()),
8613 Some("yah-cloud-1"),
8614 );
8615 }
8616 }
8617
8618 #[test]
8619 fn machine_yah_cloud_1_round_trips_with_existing_shape() {
8620 // The current machine TOML predates B2 — MachineConfig hasn't been
8621 // reshaped yet. This locks the expected shape so we notice if B3
8622 // accidentally regresses it.
8623 let src = r#"
8624name = "yah-cloud-1"
8625provider = "hetzner"
8626location = "pdx"
8627server_type = "cpx11"
8628hosts_mirrors = []
8629mesh_tags = ["tag:tier-scratch", "tag:primary-yah"]
8630ssh_keys = [111513970, 111525493]
8631"#;
8632 let cfg: MachineConfig = toml::from_str(src).unwrap();
8633 assert_eq!(cfg.name, "yah-cloud-1");
8634 assert_eq!(cfg.provider, "hetzner");
8635 assert_eq!(cfg.ssh_keys.len(), 2);
8636 }
8637
8638 #[test]
8639 fn static_node_omits_location_server_type_and_carries_connect() {
8640 // BYO Phase-0: a `static` node we brought up over SSH has no provider
8641 // DC code or SKU; it declares reach in `[connect]` instead. Must load.
8642 let src = r#"
8643name = "us-south-001"
8644provider = "static"
8645region = "us-south"
8646mesh_tags = ["tag:cloud-runner", "tag:voter-candidate"]
8647
8648[connect]
8649address = "45.32.194.254"
8650ssh = "root@45.32.194.254"
8651identity_file = "~/.ssh/yah"
8652yubaba = "http://127.0.0.1:7443"
8653arch = "x86_64"
8654"#;
8655 let cfg: MachineConfig = toml::from_str(src).unwrap();
8656 assert_eq!(cfg.provider, "static");
8657 assert!(cfg.location.is_none());
8658 assert!(cfg.server_type.is_none());
8659 assert_eq!(cfg.location(), ""); // accessor defaults empty
8660 let c = cfg.connect.as_ref().expect("connect block");
8661 assert_eq!(c.ssh, "root@45.32.194.254");
8662 // Loopback is a *declared* reach placeholder, so it stays in [connect]
8663 // verbatim and composes straight through (R707-T1).
8664 assert_eq!(c.yubaba.as_deref(), Some("http://127.0.0.1:7443"));
8665 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://127.0.0.1:7443"));
8666 assert_eq!(cfg.mesh_ipv4(), None);
8667 // Static providers have no driver, so validate() is a no-op pass.
8668 assert!(!provider_has_machine_driver(&cfg.provider));
8669 cfg.validate().unwrap();
8670 }
8671
8672 // ─── R707-T1: declaration / registration split ──────────────────────────
8673
8674 /// The pre-split shape — top-level `hostkey_fingerprint`, mesh IP baked
8675 /// into `[connect].yubaba` — must keep parsing, and must read back through
8676 /// the accessors identically. Every machine TOML in the fleet was written
8677 /// this way, and other camps' inventories still are.
8678 #[test]
8679 fn legacy_shape_still_parses_and_reads_through_accessors() {
8680 let src = r#"
8681name = "us-west-001"
8682provider = "static"
8683region = "us-west"
8684arch = "x86_64"
8685mesh_tags = ["tag:cloud-runner"]
8686hostkey_fingerprint = "SHA256:dmpq"
8687
8688[connect]
8689address = "15.204.89.240"
8690ssh = "debian@15.204.89.240"
8691identity_file = "~/.ssh/yah"
8692yubaba = "http://100.64.0.1:7443"
8693"#;
8694 let cfg: MachineConfig = toml::from_str(src).unwrap();
8695 assert_eq!(cfg.hostkey_fingerprint(), Some("SHA256:dmpq"));
8696 assert_eq!(cfg.mesh_ipv4(), Some("100.64.0.1"));
8697 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://100.64.0.1:7443"));
8698 }
8699
8700 /// The post-split shape reads identically to the legacy one above — same
8701 /// three accessor answers from a file that separates the two halves. This
8702 /// is the "unchanged in meaning" guarantee the fleet migration rests on.
8703 #[test]
8704 fn split_shape_is_equivalent_to_legacy_shape() {
8705 let legacy = r#"
8706name = "m"
8707provider = "static"
8708mesh_tags = []
8709hostkey_fingerprint = "SHA256:dmpq"
8710
8711[connect]
8712address = "15.204.89.240"
8713ssh = "debian@15.204.89.240"
8714identity_file = "~/.ssh/yah"
8715yubaba = "http://100.64.0.1:7443"
8716"#;
8717 let split = r#"
8718name = "m"
8719provider = "static"
8720mesh_tags = []
8721
8722[connect]
8723address = "15.204.89.240"
8724ssh = "debian@15.204.89.240"
8725identity_file = "~/.ssh/yah"
8726
8727[registration]
8728hostkey_fingerprint = "SHA256:dmpq"
8729mesh_ipv4 = "100.64.0.1"
8730"#;
8731 let old: MachineConfig = toml::from_str(legacy).unwrap();
8732 let new: MachineConfig = toml::from_str(split).unwrap();
8733 assert_eq!(old.hostkey_fingerprint(), new.hostkey_fingerprint());
8734 assert_eq!(old.mesh_ipv4(), new.mesh_ipv4());
8735 assert_eq!(old.yubaba_url(), new.yubaba_url());
8736 }
8737
8738 /// A non-default `[connect].yubaba_port` is declared reach and composes
8739 /// with the observed mesh address rather than being pinned into a URL.
8740 #[test]
8741 fn declared_port_composes_with_observed_mesh_address() {
8742 let src = r#"
8743name = "m"
8744provider = "static"
8745mesh_tags = []
8746
8747[connect]
8748address = "10.0.0.1"
8749ssh = "yah@10.0.0.1"
8750identity_file = "~/.ssh/yah"
8751yubaba_port = 9443
8752
8753[registration]
8754mesh_ipv4 = "100.64.0.9"
8755"#;
8756 let cfg: MachineConfig = toml::from_str(src).unwrap();
8757 assert_eq!(cfg.connect.as_ref().unwrap().yubaba_port(), 9443);
8758 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://100.64.0.9:9443"));
8759 }
8760
8761 /// R605-T10 inverts R707-T6 for the private-literal case, and this is the
8762 /// node it was inverted for: us-west-014's shape, mesh-joined AND declaring
8763 /// a LAN `[connect].yubaba`. R707-T6 made the literal win outright so
8764 /// `rollout::yubaba::membership_to_nodes` could match the dev group's
8765 /// LAN-addressed raft membership — which fused identity into reach and made
8766 /// every automated dial go to an address only bldg-2506 can route.
8767 /// `lan_endpoint()` now serves that match, so the mesh address wins the
8768 /// dial and the literal is inert.
8769 #[test]
8770 fn a_private_literal_loses_to_the_registered_mesh_address() {
8771 let src = r#"
8772name = "us-west-014"
8773provider = "static"
8774mesh_tags = []
8775
8776[connect]
8777address = "192.168.10.14"
8778ssh = "yah@192.168.10.14"
8779identity_file = "~/.ssh/yah"
8780yubaba = "http://192.168.10.14:7443"
8781
8782[registration]
8783mesh_ipv4 = "100.64.0.6"
8784"#;
8785 let cfg: MachineConfig = toml::from_str(src).unwrap();
8786 assert_eq!(cfg.mesh_ipv4(), Some("100.64.0.6"), "still mesh-joined");
8787 assert_eq!(
8788 cfg.yubaba_url().as_deref(),
8789 Some("http://100.64.0.6:7443"),
8790 "automation dials the mesh, never the LAN literal"
8791 );
8792 assert_eq!(
8793 cfg.lan_endpoint().as_deref(),
8794 Some("192.168.10.14:7443"),
8795 "the LAN address is still recorded — as identity, not as reach"
8796 );
8797 }
8798
8799 /// The refusal R605-T10 asks for: a node whose ONLY declared reach is a LAN
8800 /// literal is unresolvable, and says so by name rather than returning a URL
8801 /// that will time out. us-west-011's shape before this ticket.
8802 #[test]
8803 fn a_lan_only_node_refuses_with_a_named_reason() {
8804 let src = r#"
8805name = "us-west-011"
8806provider = "static"
8807mesh_tags = []
8808
8809[connect]
8810address = "192.168.10.11"
8811ssh = "yah@192.168.10.11"
8812identity_file = "~/.ssh/yah"
8813yubaba = "http://192.168.10.11:7443"
8814"#;
8815 let cfg: MachineConfig = toml::from_str(src).unwrap();
8816 assert_eq!(cfg.yubaba_url(), None);
8817 let err = cfg.reach().unwrap_err();
8818 assert!(err.contains("us-west-011"), "{err}");
8819 assert!(err.contains("192.168.10.11"), "{err}");
8820 assert!(err.contains("mesh_ipv4"), "{err}");
8821 }
8822
8823 /// The loopback placeholder is a genuine declaration ("reach me through the
8824 /// SSH tunnel"), not a LAN literal — 127/8 is not RFC1918. It must keep
8825 /// resolving verbatim; `hub::coordinator::is_loopback_url` is what judges it
8826 /// downstream.
8827 #[test]
8828 fn a_loopback_placeholder_still_resolves_verbatim() {
8829 let src = r#"
8830name = "m"
8831provider = "static"
8832mesh_tags = []
8833
8834[connect]
8835address = "192.168.10.99"
8836ssh = "yah@192.168.10.99"
8837identity_file = "~/.ssh/yah"
8838yubaba = "http://127.0.0.1:7443"
8839"#;
8840 let cfg: MachineConfig = toml::from_str(src).unwrap();
8841 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://127.0.0.1:7443"));
8842 }
8843
8844 #[test]
8845 fn private_ranges_are_exactly_rfc1918() {
8846 for lan in [
8847 "http://192.168.10.11:7443",
8848 "http://10.0.0.5:7443",
8849 "http://172.16.4.1:7443",
8850 ] {
8851 assert!(private_ipv4_from_url(lan).is_some(), "{lan}");
8852 }
8853 for not_lan in [
8854 "http://100.64.0.6:7443", // mesh
8855 "http://127.0.0.1:7443", // loopback
8856 "http://172.32.0.1:7443", // just past 172.16/12
8857 "http://45.32.194.254:80", // public
8858 "http://us-west-001:7443", // name, not a literal
8859 ] {
8860 assert!(private_ipv4_from_url(not_lan).is_none(), "{not_lan}");
8861 }
8862 }
8863
8864 /// `normalize` migrates in place: the legacy fingerprint moves into
8865 /// `[registration]`, the mesh IP is lifted out of the URL, and the derived
8866 /// `[connect].yubaba` is cleared so the two halves cannot drift.
8867 #[test]
8868 fn normalize_migrates_legacy_fields_and_is_idempotent() {
8869 let src = r#"
8870name = "m"
8871provider = "static"
8872mesh_tags = []
8873hostkey_fingerprint = "SHA256:dmpq"
8874
8875[connect]
8876address = "15.204.89.240"
8877ssh = "debian@15.204.89.240"
8878identity_file = "~/.ssh/yah"
8879yubaba = "http://100.64.0.1:7443"
8880"#;
8881 let mut cfg: MachineConfig = toml::from_str(src).unwrap();
8882 cfg.normalize();
8883 assert!(cfg.legacy_hostkey_fingerprint.is_none());
8884 assert_eq!(
8885 cfg.registration.hostkey_fingerprint.as_deref(),
8886 Some("SHA256:dmpq")
8887 );
8888 assert_eq!(cfg.registration.mesh_ipv4.as_deref(), Some("100.64.0.1"));
8889 assert!(cfg.connect.as_ref().unwrap().yubaba.is_none());
8890 // Accessors still answer the same, and re-running changes nothing.
8891 assert_eq!(cfg.yubaba_url().as_deref(), Some("http://100.64.0.1:7443"));
8892 let once = format!("{cfg:?}");
8893 cfg.normalize();
8894 assert_eq!(once, format!("{cfg:?}"));
8895 }
8896
8897 /// A loopback `[connect].yubaba` is a declaration ("no mesh address yet —
8898 /// reach me through the SSH tunnel"), not a stale observation, so
8899 /// `normalize` must leave it alone. us-west-003/011/013 depend on this.
8900 #[test]
8901 fn normalize_leaves_pre_mesh_loopback_declaration_intact() {
8902 let src = r#"
8903name = "m"
8904provider = "static"
8905mesh_tags = []
8906
8907[connect]
8908address = "192.168.10.11"
8909ssh = "yah@192.168.10.11"
8910identity_file = "~/.ssh/yah"
8911yubaba = "http://127.0.0.1:7443"
8912"#;
8913 let mut cfg: MachineConfig = toml::from_str(src).unwrap();
8914 cfg.normalize();
8915 assert_eq!(
8916 cfg.connect.as_ref().unwrap().yubaba.as_deref(),
8917 Some("http://127.0.0.1:7443")
8918 );
8919 assert!(cfg.registration.is_empty());
8920 assert_eq!(cfg.mesh_ipv4(), None);
8921 }
8922
8923 /// `save` normalizes, so a legacy file that round-trips through the writer
8924 /// comes back on the split shape with nothing lost — the property that
8925 /// keeps `yah cloud machine attach` from re-emitting the old layout.
8926 #[test]
8927 fn save_writes_the_split_shape_from_a_legacy_config() {
8928 let tmp = tempfile::TempDir::new().unwrap();
8929 let root = tmp.path();
8930 let src = r#"
8931name = "m"
8932provider = "static"
8933mesh_tags = []
8934hostkey_fingerprint = "SHA256:dmpq"
8935
8936[connect]
8937address = "15.204.89.240"
8938ssh = "debian@15.204.89.240"
8939identity_file = "~/.ssh/yah"
8940yubaba = "http://100.64.0.1:7443"
8941"#;
8942 let cfg: MachineConfig = toml::from_str(src).unwrap();
8943 cfg.save(root).unwrap();
8944
8945 let written = std::fs::read_to_string(root.join("machines/m.toml")).unwrap();
8946 let reg_at = written
8947 .find("[registration]")
8948 .unwrap_or_else(|| panic!("no [registration] table: {written}"));
8949 let fp_at = written
8950 .find("hostkey_fingerprint")
8951 .unwrap_or_else(|| panic!("fingerprint dropped: {written}"));
8952 assert!(
8953 fp_at > reg_at,
8954 "legacy top-level field must not be re-emitted: {written}"
8955 );
8956 assert!(
8957 !written.contains("yubaba ="),
8958 "derived URL must not be re-emitted alongside mesh_ipv4: {written}"
8959 );
8960
8961 let reloaded: MachineConfig = toml::from_str(&written).unwrap();
8962 assert_eq!(reloaded.hostkey_fingerprint(), Some("SHA256:dmpq"));
8963 assert_eq!(
8964 reloaded.yubaba_url().as_deref(),
8965 Some("http://100.64.0.1:7443")
8966 );
8967 }
8968
8969 /// `[registration]` is omitted entirely for a machine nothing has been
8970 /// observed about — a scaffolded declaration stays clean.
8971 #[test]
8972 fn empty_registration_is_omitted_on_serialize() {
8973 let src = r#"
8974name = "m"
8975provider = "static"
8976mesh_tags = []
8977"#;
8978 let cfg: MachineConfig = toml::from_str(src).unwrap();
8979 assert!(cfg.registration.is_empty());
8980 let out = toml::to_string_pretty(&cfg).unwrap();
8981 assert!(!out.contains("[registration]"), "{out}");
8982 }
8983
8984 #[test]
8985 fn driver_provider_without_location_fails_validate() {
8986 // A driver-backed provider (hetzner/vultr) still MUST carry location +
8987 // server_type — the driver can't create a server without them. The
8988 // contract moved from load-time (required field) to provision-time
8989 // (validate), so the TOML loads but validate() rejects it.
8990 let src = r#"
8991name = "us-west-001"
8992provider = "hetzner"
8993mesh_tags = []
8994"#;
8995 let cfg: MachineConfig = toml::from_str(src).unwrap();
8996 assert!(provider_has_machine_driver(&cfg.provider));
8997 let err = cfg.validate().unwrap_err().to_string();
8998 assert!(
8999 err.contains("location"),
9000 "expected location complaint: {err}"
9001 );
9002 }
9003
9004 /// Helper for the new-tree integration tests below: lay out
9005 /// `<workspace>/.yah/{infra,services}/` with `dev-yah` + its mirrors and
9006 /// the three Phase-A providers (cloudflare, hetzner, orbstack).
9007 fn make_new_tree_with_dev_yah(root: &std::path::Path) {
9008 let infra = root.join(".yah").join("infra");
9009 let providers = infra.join("providers");
9010 std::fs::create_dir_all(&providers).unwrap();
9011 std::fs::write(
9012 providers.join("cloudflare.toml"),
9013 r#"schema_version = 1
9014id = "cloudflare"
9015kind = "cloudflare"
9016credentials = "keystore://cloudflare/yah"
9017default_zone = "yah.dev"
9018"#,
9019 )
9020 .unwrap();
9021 std::fs::write(
9022 providers.join("hetzner.toml"),
9023 r#"schema_version = 1
9024id = "hetzner"
9025kind = "hetzner"
9026credentials = "keystore://hetzner/yah"
9027default_location = "pdx"
9028default_server_type = "cpx11"
9029ssh_keys = []
9030"#,
9031 )
9032 .unwrap();
9033 std::fs::write(
9034 providers.join("orbstack.toml"),
9035 r#"schema_version = 1
9036id = "orbstack"
9037kind = "local-container"
9038runtime = "auto"
9039
9040[discovery]
9041orbstack = "~/.orbstack/run/docker.sock"
9042"#,
9043 )
9044 .unwrap();
9045
9046 let svc = root.join(".yah").join("services").join("dev-yah");
9047 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
9048 std::fs::write(
9049 svc.join("service.toml"),
9050 r#"schema_version = 1
9051name = "dev-yah"
9052[address]
9053kind = "front-door"
9054domain = "yah.dev"
9055
9056[[components]]
9057id = "site"
9058kind = "mesofact-static"
9059path = "app/yah/web"
9060role = "static"
9061"#,
9062 )
9063 .unwrap();
9064 std::fs::write(
9065 svc.join("mirrors/cloud.toml"),
9066 r#"schema_version = 1
9067shape = "single-machine"
9068
9069[providers.static]
9070use = "cloudflare"
9071bucket = "yah-dev"
9072zone = "yah.dev"
9073"#,
9074 )
9075 .unwrap();
9076 std::fs::write(
9077 svc.join("mirrors/local.toml"),
9078 r#"schema_version = 1
9079shape = "local"
9080
9081[providers.static]
9082kind = "miniflare-native"
9083port = 4321
9084
9085[providers.compute]
9086use = "orbstack"
9087"#,
9088 )
9089 .unwrap();
9090 }
9091
9092 #[test]
9093 fn cloud_config_load_new_tree_populates_providers_and_services() {
9094 let tmp = tempfile::TempDir::new().unwrap();
9095 let root = tmp.path();
9096 make_new_tree_with_dev_yah(root);
9097
9098 let cfg = CloudConfig::load(root).unwrap();
9099 assert_eq!(cfg.providers.len(), 3, "three providers loaded");
9100 assert!(cfg.provider("cloudflare").is_some());
9101 assert!(cfg.provider("hetzner").is_some());
9102 assert!(cfg.provider("orbstack").is_some());
9103
9104 let dev = cfg.service("dev-yah").expect("dev-yah service");
9105 assert_eq!(dev.service.domain(), Some("yah.dev"));
9106 assert_eq!(dev.service.components.len(), 1);
9107 assert_eq!(dev.mirrors.len(), 2);
9108 // Legacy file stems "cloud" and "local" are normalised to canonical tier names.
9109 assert!(dev.mirrors.contains_key("prod"), "cloud.toml → prod tier");
9110 assert!(dev.mirrors.contains_key("dev"), "local.toml → dev tier");
9111 assert_eq!(dev.mirrors["prod"].shape, MirrorShape::SingleMachine);
9112 assert_eq!(dev.mirrors["dev"].shape, MirrorShape::Local);
9113
9114 // Legacy fields stay empty when no .yah/cloud/ exists.
9115 assert!(cfg.legacy_mirrors.is_empty());
9116 assert!(cfg.workloads.is_empty());
9117 }
9118
9119 #[test]
9120 fn cloud_config_cross_ref_fails_on_missing_provider() {
9121 // Mirror references a provider id that doesn't exist.
9122 let tmp = tempfile::TempDir::new().unwrap();
9123 let root = tmp.path();
9124 let svc = root.join(".yah").join("services").join("dev-yah");
9125 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
9126 std::fs::write(
9127 svc.join("service.toml"),
9128 "schema_version = 1\nname = \"dev-yah\"\n[address]\nkind = \"front-door\"\ndomain = \"yah.dev\"\n",
9129 )
9130 .unwrap();
9131 std::fs::write(
9132 svc.join("mirrors/prod.toml"),
9133 "schema_version = 1\nshape = \"single-machine\"\n\n[providers.static]\nuse = \"fly-io\"\n",
9134 ).unwrap();
9135
9136 let err = CloudConfig::load(root).unwrap_err();
9137 let msg = err.to_string();
9138 assert!(
9139 msg.contains("fly-io"),
9140 "error should name the missing provider id, got: {msg}"
9141 );
9142 assert!(
9143 msg.contains("providers/fly-io.toml") || msg.contains("no such provider"),
9144 "error should hint at remedy, got: {msg}"
9145 );
9146 }
9147
9148 /// R905. `[build.<id>]` is the per-environment build override, and it is
9149 /// keyed by component id — so a key naming no declared component is a
9150 /// silent no-op: the environment goes on building with the command the
9151 /// operator believed they had replaced.
9152 #[test]
9153 fn cloud_config_cross_ref_fails_on_a_build_override_for_an_unknown_component() {
9154 let tmp = tempfile::TempDir::new().unwrap();
9155 let root = tmp.path();
9156 let svc = root.join(".yah").join("services").join("dev-yah");
9157 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
9158 std::fs::write(
9159 svc.join("service.toml"),
9160 "schema_version = 1\nname = \"dev-yah\"\n[address]\nkind = \"front-door\"\ndomain = \"yah.dev\"\n\n\
9161 [[components]]\nid = \"site\"\nkind = \"mesofact-spa\"\n\
9162 path = \"web/landing\"\nrole = \"static\"\n",
9163 )
9164 .unwrap();
9165 std::fs::write(
9166 svc.join("mirrors/staging.toml"),
9167 "schema_version = 1\nshape = \"single-machine\"\n\n\
9168 [build.sight]\ncommand = \"bun run build:staging\"\n",
9169 )
9170 .unwrap();
9171
9172 let msg = CloudConfig::load(root).unwrap_err().to_string();
9173 assert!(
9174 msg.contains("build.sight") && msg.contains("site"),
9175 "error should name the bad key and the declared ids, got: {msg}"
9176 );
9177 }
9178
9179 /// The same mirror, spelled correctly, loads and resolves — including the
9180 /// `env` half, which is the knob a project uses when it does not want a
9181 /// sibling `build:<env>` script per environment (R905).
9182 #[test]
9183 fn a_mirror_build_override_resolves_by_component_id() {
9184 let tmp = tempfile::TempDir::new().unwrap();
9185 let root = tmp.path();
9186 let svc = root.join(".yah").join("services").join("dev-yah");
9187 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
9188 std::fs::write(
9189 svc.join("service.toml"),
9190 "schema_version = 1\nname = \"dev-yah\"\n[address]\nkind = \"front-door\"\ndomain = \"yah.dev\"\n\n\
9191 [[components]]\nid = \"site\"\nkind = \"mesofact-spa\"\n\
9192 path = \"web/landing\"\nrole = \"static\"\n\n\
9193 [[components]]\nid = \"app\"\nkind = \"mesofact-static\"\n\
9194 path = \"app/browser\"\nrole = \"static\"\nmount = \"/app\"\n",
9195 )
9196 .unwrap();
9197 std::fs::write(
9198 svc.join("mirrors/staging.toml"),
9199 "schema_version = 1\nshape = \"single-machine\"\n\n\
9200 [build.site]\ncommand = \"bun run build:staging\"\n\n\
9201 [build.site.env]\nAPI_ORIGIN = \"https://api-staging.example.com\"\n",
9202 )
9203 .unwrap();
9204
9205 let cfg = CloudConfig::load(root).unwrap();
9206 let mirror = &cfg.service("dev-yah").unwrap().mirrors["staging"];
9207
9208 let site = mirror.build_override("site").expect("site override");
9209 assert_eq!(site.command.as_deref(), Some("bun run build:staging"));
9210 assert_eq!(
9211 site.env_pairs(),
9212 vec![(
9213 "API_ORIGIN".to_string(),
9214 "https://api-staging.example.com".to_string()
9215 )]
9216 );
9217 // A sibling component under the same mirror is untouched — the
9218 // override is per component, not per mirror.
9219 assert!(mirror.build_override("app").is_none());
9220 }
9221
9222 /// An override that names nothing reads as no override at all, so callers
9223 /// can treat `Some(_)` as "something differs here" (R905).
9224 #[test]
9225 fn an_empty_build_override_reads_as_absent() {
9226 let empty = MirrorBuildOverride::default();
9227 assert!(empty.is_empty());
9228 let mut m = mirror("");
9229 m.build.insert("site".into(), empty);
9230 assert!(m.build_override("site").is_none());
9231 }
9232
9233 #[test]
9234 fn cloud_config_cross_ref_fails_on_missing_provider_named_by_an_ingress_edge() {
9235 // R845: the edge's own `use` is a provider reference like any other, so
9236 // a typo has to fail here rather than at the Cloudflare arm of apply.
9237 let tmp = tempfile::TempDir::new().unwrap();
9238 let root = tmp.path();
9239 let svc = root.join(".yah").join("services").join("dev-yah");
9240 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
9241 std::fs::write(
9242 svc.join("service.toml"),
9243 "schema_version = 1\nname = \"dev-yah\"\n[address]\nkind = \"front-door\"\ndomain = \"yah.dev\"\n",
9244 )
9245 .unwrap();
9246 std::fs::write(
9247 svc.join("mirrors/prod.toml"),
9248 "schema_version = 1\nshape = \"single-machine\"\n\n\
9249 [providers.compute]\nkind = \"static\"\nmachine = \"borrowed-01\"\n\
9250 zone = \"a.yah.dev\"\nport = 8080\n\n\
9251 [[ingress]]\nprovider = \"cloudflare-tunnel\"\nuse = \"cloudflar\"\n",
9252 )
9253 .unwrap();
9254
9255 let msg = CloudConfig::load(root).unwrap_err().to_string();
9256 assert!(
9257 msg.contains("ingress[0].use") && msg.contains("cloudflar"),
9258 "error should name the edge and the typo'd id, got: {msg}"
9259 );
9260 }
9261
9262 #[test]
9263 fn cloud_config_cross_ref_passes_on_inline_only_mirror() {
9264 // Inline `kind = "miniflare-native"` doesn't require an infra provider.
9265 let tmp = tempfile::TempDir::new().unwrap();
9266 let root = tmp.path();
9267 let svc = root.join(".yah").join("services").join("local-only");
9268 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
9269 std::fs::write(
9270 svc.join("service.toml"),
9271 "schema_version = 1\nname = \"local-only\"\n[address]\nkind = \"front-door\"\ndomain = \"local.test\"\n",
9272 )
9273 .unwrap();
9274 std::fs::write(
9275 svc.join("mirrors/local.toml"),
9276 "schema_version = 1\nshape = \"local\"\n\n[providers.static]\nkind = \"miniflare-native\"\nport = 8080\n",
9277 ).unwrap();
9278
9279 // Should load fine: no `use=` references, no providers required.
9280 let cfg = CloudConfig::load(root).unwrap();
9281 assert!(cfg.service("local-only").is_some());
9282 }
9283
9284 fn mirror(src: &str) -> MirrorConfig {
9285 toml::from_str(&format!("schema_version = 1\nshape = \"single-machine\"\n{src}"))
9286 .expect("parse mirror")
9287 }
9288
9289 #[test]
9290 fn passway_machines_reads_both_ingress_spellings_the_same_way() {
9291 // The whole reason this is derived in Rust rather than read off a field
9292 // by the UI: these two mirrors say the identical thing, and a consumer
9293 // that reaches for `ingress_machines` sees the second one as empty.
9294 let scalar = mirror("ingress = \"passway\"\ningress_machines = [\"us-east-001\"]\n");
9295 let edges = mirror(
9296 "[[ingress]]\nprovider = \"passway\"\nmachines = [\"us-east-001\"]\n",
9297 );
9298 assert_eq!(scalar.passway_machines(), Some(vec!["us-east-001".into()]));
9299 assert_eq!(scalar.passway_machines(), edges.passway_machines());
9300 }
9301
9302 #[test]
9303 fn passway_machines_skips_a_cloudflare_tunnel_edge() {
9304 // A cloudflared node publishes through Cloudflare's DNS and does not
9305 // serve `GET /domains/{d}/onboarding`, so naming it here would point
9306 // the custom-domain UI at a node that cannot answer.
9307 let cf_only =
9308 mirror("[[ingress]]\nprovider = \"cloudflare-tunnel\"\nmachines = [\"cf-01\"]\n");
9309 assert_eq!(cf_only.passway_machines(), None);
9310
9311 let mixed = mirror(
9312 "[[ingress]]\nprovider = \"cloudflare-tunnel\"\nmachines = [\"cf-01\"]\n\
9313 slots = [\"static\"]\n\n\
9314 [[ingress]]\nprovider = \"passway\"\nmachines = [\"us-east-001\"]\n\
9315 slots = [\"bundle\"]\n",
9316 );
9317 assert_eq!(mixed.passway_machines(), Some(vec!["us-east-001".into()]));
9318 }
9319
9320 #[test]
9321 fn passway_machines_separates_declared_but_unplaced_from_undeclared() {
9322 // Some(vec![]) means "a passway front door exists, but its placement
9323 // falls back to the fronted slot's and is not knowable from the mirror".
9324 // None means there is no passway front door at all. Collapsing the two
9325 // would make a co-located edge indistinguishable from no edge.
9326 assert_eq!(mirror("ingress = \"passway\"\n").passway_machines(), Some(vec![]));
9327 assert_eq!(mirror("").passway_machines(), None);
9328 assert_eq!(mirror("ingress = \"none\"\n").passway_machines(), None);
9329 }
9330
9331 #[test]
9332 fn passway_machines_is_none_for_a_declaration_that_cannot_mean_anything() {
9333 // `ingress_machines` with no `ingress` is an error `ingress_edges` names
9334 // properly; swallowing it to None here is deliberate, because this is
9335 // read while loading every service in the workspace and hard-failing
9336 // would report an unrelated mirror's shape error from the wrong place.
9337 let orphaned = mirror("ingress_machines = [\"us-east-001\"]\n");
9338 assert!(orphaned.ingress_edges().is_err());
9339 assert_eq!(orphaned.passway_machines(), None);
9340 }
9341
9342 /// Same as [`mirror`] but surfacing the parse error instead of panicking —
9343 /// R870-F26's half-written-auth cases are refused BY serde, so the message
9344 /// only exists on this side of the `expect`.
9345 fn try_mirror(src: &str) -> std::result::Result<MirrorConfig, toml::de::Error> {
9346 toml::from_str(&format!("schema_version = 1\nshape = \"single-machine\"\n{src}"))
9347 }
9348
9349 const FULL_AUTH: &str = "[ingress.auth]\n\
9350 key_secret = \"cheers/yah-camp/verify\"\n\
9351 kid = \"YOHV4Riq-g8fX4uYl8rTjQ\"\n\
9352 iss = \"yah-camp\"\n\
9353 aud = \"analytics.yah.dev\"\n\
9354 require_prefixes = [\"/\"]\n";
9355
9356 /// R870-F26 — the vocabulary itself: a complete `[ingress.auth]` table
9357 /// lands on the edge as the renderer's own `PasswayAuth`, which is what
9358 /// `apply` hands to `PasswayIngressSpec` so the push carries the five
9359 /// variables rather than stripping them.
9360 #[test]
9361 fn an_ingress_edge_carries_a_declared_auth_table_through_to_the_renderers_type() {
9362 let m = mirror(&format!(
9363 "[[ingress]]\nprovider = \"passway\"\nmachines = [\"us-east-001\"]\n{FULL_AUTH}"
9364 ));
9365 let edges = m.ingress_edges().expect("a complete auth table is accepted");
9366 let auth = edges[0].auth.as_ref().expect("the table reached the edge");
9367 assert_eq!(auth.key_secret, "cheers/yah-camp/verify");
9368 assert_eq!(auth.kid, "YOHV4Riq-g8fX4uYl8rTjQ");
9369 assert_eq!(auth.iss, "yah-camp");
9370 assert_eq!(auth.aud, "analytics.yah.dev");
9371 assert_eq!(auth.require_prefixes, vec!["/".to_string()]);
9372 }
9373
9374 /// An edge with no auth is byte-identically what it was before the field
9375 /// existed. The default path is the one this must not move.
9376 #[test]
9377 fn an_edge_that_declares_no_auth_is_unchanged() {
9378 let m = mirror("[[ingress]]\nprovider = \"passway\"\nmachines = [\"us-east-001\"]\n");
9379 assert_eq!(m.ingress_edges().unwrap()[0].auth, None);
9380 // And the scalar spelling, which cannot express auth at all.
9381 assert_eq!(
9382 mirror("ingress = \"passway\"\n").ingress_edges().unwrap()[0].auth,
9383 None
9384 );
9385 }
9386
9387 /// A HALF-WRITTEN table is refused at load, naming the field that is
9388 /// missing — never deployed as a half-configured door.
9389 ///
9390 /// This is serde's own doing, and deliberately so: all five fields are
9391 /// required on `PasswayAuth`, so there is no partial value to construct.
9392 /// The dangerous half is the one that fails QUIETLY — `kid`/`iss`/`aud`
9393 /// without `key_secret` makes passway skip the whole feature and come up
9394 /// anonymous, with nothing anywhere complaining.
9395 #[test]
9396 fn a_half_written_auth_table_is_refused_naming_the_missing_field() {
9397 let cases = [
9398 ("key_secret", "kid = \"k\"\niss = \"i\"\naud = \"a\"\nrequire_prefixes = [\"/\"]\n"),
9399 ("kid", "key_secret = \"s\"\niss = \"i\"\naud = \"a\"\nrequire_prefixes = [\"/\"]\n"),
9400 ("iss", "key_secret = \"s\"\nkid = \"k\"\naud = \"a\"\nrequire_prefixes = [\"/\"]\n"),
9401 ("aud", "key_secret = \"s\"\nkid = \"k\"\niss = \"i\"\nrequire_prefixes = [\"/\"]\n"),
9402 ("require_prefixes", "key_secret = \"s\"\nkid = \"k\"\niss = \"i\"\naud = \"a\"\n"),
9403 ];
9404 for (missing, body) in cases {
9405 let err = try_mirror(&format!(
9406 "[[ingress]]\nprovider = \"passway\"\n[ingress.auth]\n{body}"
9407 ))
9408 .expect_err("a partial auth table must not load");
9409 assert!(
9410 err.to_string().contains(missing),
9411 "the refusal must name {missing}, got: {err}"
9412 );
9413 }
9414 }
9415
9416 /// The half serde cannot catch: a field that is PRESENT and empty. Refused
9417 /// by `PasswayAuth::validate`, the same implementation
9418 /// `yah cloud ingress deploy` runs against its flags — so the two authoring
9419 /// routes cannot disagree about what counts as configured.
9420 #[test]
9421 fn a_present_but_empty_auth_field_is_refused_naming_the_toml_key() {
9422 let err = mirror(
9423 "[[ingress]]\nprovider = \"passway\"\n[ingress.auth]\n\
9424 key_secret = \"s\"\nkid = \"\"\niss = \"i\"\naud = \"a\"\nrequire_prefixes = [\"/\"]\n",
9425 )
9426 .ingress_edges()
9427 .expect_err("an empty kid is a boot panic on a remote node")
9428 .to_string();
9429 assert!(err.contains("[ingress.auth].kid"), "{err}");
9430
9431 // The quiet one: a verify key protecting nothing is authenticated and
9432 // anonymous at once. The message names the TOML key, not the flag —
9433 // sending a mirror author to look for `--require-auth` costs them the
9434 // search this validation exists to save.
9435 let err = mirror(&format!(
9436 "[[ingress]]\nprovider = \"passway\"\n{}",
9437 FULL_AUTH.replace("require_prefixes = [\"/\"]", "require_prefixes = []")
9438 ))
9439 .ingress_edges()
9440 .expect_err("a door protecting no prefix must be refused")
9441 .to_string();
9442 assert!(err.contains("[ingress.auth].require_prefixes"), "{err}");
9443 assert!(!err.contains("--require-auth"), "wrong vocabulary: {err}");
9444 }
9445
9446 /// Auth on a non-passway edge is refused rather than ignored. Only passway
9447 /// renders `PASSWAY_AUTH_*`; silently dropping it hands the operator a door
9448 /// they believe is protected and is not — the exact outcome this whole
9449 /// vocabulary exists to prevent.
9450 #[test]
9451 fn auth_on_a_cloudflare_tunnel_edge_is_refused_rather_than_ignored() {
9452 let err = mirror(&format!(
9453 "[[ingress]]\nprovider = \"cloudflare-tunnel\"\nmachines = [\"cf-01\"]\n{FULL_AUTH}"
9454 ))
9455 .ingress_edges()
9456 .expect_err("only a passway edge can render bearer auth")
9457 .to_string();
9458 assert!(err.contains("passway"), "{err}");
9459 assert!(err.contains("[ingress.auth]"), "{err}");
9460 }
9461
9462 #[test]
9463 fn cloud_config_load_derives_passway_machines_only_for_passway_envs() {
9464 let tmp = tempfile::TempDir::new().unwrap();
9465 let root = tmp.path();
9466 let svc = root.join(".yah").join("services").join("dev-yah");
9467 std::fs::create_dir_all(svc.join("mirrors")).unwrap();
9468 std::fs::write(
9469 svc.join("service.toml"),
9470 "schema_version = 1\nname = \"dev-yah\"\n[address]\nkind = \"front-door\"\ndomain = \"yah.dev\"\n",
9471 )
9472 .unwrap();
9473 std::fs::write(
9474 svc.join("mirrors/prod.toml"),
9475 "schema_version = 1\nshape = \"single-machine\"\n\
9476 ingress = \"passway\"\ningress_machines = [\"us-east-001\", \"us-west-001\"]\n\n\
9477 [providers.static]\nkind = \"miniflare-native\"\nport = 8080\n",
9478 )
9479 .unwrap();
9480 std::fs::write(
9481 svc.join("mirrors/local.toml"),
9482 "schema_version = 1\nshape = \"local\"\n\n\
9483 [providers.static]\nkind = \"miniflare-native\"\nport = 8080\n",
9484 )
9485 .unwrap();
9486
9487 let cfg = CloudConfig::load(root).unwrap();
9488 let svc = cfg.service("dev-yah").unwrap();
9489 assert_eq!(
9490 svc.passway_machines.get("prod"),
9491 Some(&vec!["us-east-001".to_string(), "us-west-001".to_string()])
9492 );
9493 assert!(
9494 !svc.passway_machines.contains_key("local"),
9495 "an env with no front door must be absent, not empty: {:?}",
9496 svc.passway_machines
9497 );
9498 }
9499
9500 #[test]
9501 fn cloud_config_load_coexists_legacy_and_new_trees() {
9502 // Both trees present — both fields populated independently.
9503 let tmp = tempfile::TempDir::new().unwrap();
9504 let root = tmp.path();
9505 make_new_tree_with_dev_yah(root);
9506
9507 let cloud_dir = make_legacy_cloud_dir(root);
9508 std::fs::create_dir_all(cloud_dir.join("mirrors")).unwrap();
9509 std::fs::write(
9510 cloud_dir.join("mirrors/noisetable.toml"),
9511 "camp = \"noisetable\"\nregions = [\"pdx\"]\nworkloads = []\n",
9512 )
9513 .unwrap();
9514
9515 let cfg = CloudConfig::load(root).unwrap();
9516 assert_eq!(cfg.providers.len(), 3);
9517 assert!(cfg.service("dev-yah").is_some());
9518 assert_eq!(cfg.legacy_mirrors.len(), 1);
9519 assert!(cfg.legacy_mirror("noisetable").is_some());
9520 }
9521
9522 #[test]
9523 fn web_workload_round_trips() {
9524 // app/yah/web/workload.toml is parsed as a WorkloadSpec via the
9525 // workload-spec crate. The minimum-viable manifest here exercises
9526 // schema_version + kind + build fields.
9527 //
9528 // The on-disk file uses the abbreviated v1 form (kind + build); the
9529 // full WorkloadSpec is verbose, so this test asserts the new
9530 // mesofact-static abbreviated form parses as raw TOML (B3 will plumb
9531 // it through WorkloadSpec proper).
9532 // `routes` above [build] — it is a top-level field, and TOML would
9533 // scope it into that table if written below the header (R658-B1).
9534 let src = r#"
9535schema_version = 1
9536kind = "mesofact-static"
9537
9538routes = "./routes.ts"
9539
9540[build]
9541command = "bun run build"
9542out_dir = "dist"
9543"#;
9544 let v: toml::Value = toml::from_str(src).unwrap();
9545 assert_eq!(
9546 v.get("schema_version").and_then(|x| x.as_integer()),
9547 Some(1)
9548 );
9549 assert_eq!(
9550 v.get("kind").and_then(|x| x.as_str()),
9551 Some("mesofact-static")
9552 );
9553 let build = v
9554 .get("build")
9555 .and_then(|x| x.as_table())
9556 .expect("build table");
9557 assert_eq!(
9558 build.get("command").and_then(|x| x.as_str()),
9559 Some("bun run build")
9560 );
9561 assert_eq!(build.get("out_dir").and_then(|x| x.as_str()), Some("dist"));
9562 }
9563
9564 // ─── Canonical CRUD: ServiceConfig/MirrorConfig save + delete (R323-F1) ──
9565
9566 #[test]
9567 fn service_config_save_creates_canonical_toml_and_round_trips() {
9568 let tmp = tempfile::TempDir::new().unwrap();
9569 let root = tmp.path();
9570
9571 let svc = ServiceConfig {
9572 schema_version: 1,
9573 name: "dev-yah".into(),
9574 address: ServiceAddress::front_door("yah.dev"),
9575 description: None,
9576 db: DbCatalog::default(),
9577 components: vec![ServiceComponent {
9578 mount: None,
9579 id: "site".into(),
9580 kind: "mesofact-static".into(),
9581 path: "app/yah/web".into(),
9582 role: "static".into(),
9583 publishes: Some("static".into()),
9584 wave: 0,
9585 git: None,
9586 deploy: Default::default(),
9587 }],
9588 };
9589 svc.save(root).unwrap();
9590
9591 // Landed at the canonical path.
9592 let path = crate::paths::service_toml(root, "dev-yah");
9593 assert!(
9594 path.exists(),
9595 "service.toml should exist at {}",
9596 path.display()
9597 );
9598
9599 // Reloads through the full CloudConfig loader (no mirrors yet).
9600 let cfg = CloudConfig::load(root).unwrap();
9601 let loaded = cfg.service("dev-yah").expect("dev-yah service");
9602 assert_eq!(loaded.service.domain(), Some("yah.dev"));
9603 assert_eq!(loaded.service.components.len(), 1);
9604 assert_eq!(
9605 loaded.service.components[0].publishes.as_deref(),
9606 Some("static")
9607 );
9608 assert!(loaded.mirrors.is_empty());
9609 }
9610
9611 #[test]
9612 fn a_declared_health_path_survives_the_loader() {
9613 // The field is only worth having if it reaches the consumer — the
9614 // desktop front-door probe reads it off the loaded service, so a
9615 // round-trip that drops it would leave the probe on `/` with the
9616 // config still reading correctly.
9617 let tmp = tempfile::TempDir::new().unwrap();
9618 let root = tmp.path();
9619
9620 ServiceConfig {
9621 schema_version: 1,
9622 name: "api".into(),
9623 address: ServiceAddress::front_door_at("api.noisetable.com", "/api/v1/status"),
9624 description: None,
9625 components: vec![],
9626 db: DbCatalog::default(),
9627 }
9628 .save(root)
9629 .unwrap();
9630
9631 let cfg = CloudConfig::load(root).unwrap();
9632 assert_eq!(
9633 cfg.service("api").unwrap().service.health_path(),
9634 Some("/api/v1/status")
9635 );
9636 }
9637
9638 /// R926. Same reasoning as the health_path round-trip above, and the
9639 /// same failure mode: `description` is `skip_serializing_if =
9640 /// "Option::is_none"`, so a serde attribute that silently dropped it
9641 /// would leave every service listed without one while the file on
9642 /// disk still read correctly.
9643 #[test]
9644 fn a_declared_description_survives_the_loader() {
9645 let tmp = tempfile::TempDir::new().unwrap();
9646 let root = tmp.path();
9647
9648 ServiceConfig {
9649 schema_version: 1,
9650 name: "api".into(),
9651 address: ServiceAddress::front_door("api.noisetable.com"),
9652 description: Some("Account and RPC origin for the noisetable app.".into()),
9653 components: vec![],
9654 db: DbCatalog::default(),
9655 }
9656 .save(root)
9657 .unwrap();
9658
9659 let cfg = CloudConfig::load(root).unwrap();
9660 assert_eq!(
9661 cfg.service("api").unwrap().service.description.as_deref(),
9662 Some("Account and RPC origin for the noisetable app.")
9663 );
9664 }
9665
9666 /// The nine services that predate R926 carry no `description`, so the
9667 /// field has to be genuinely optional rather than optional-with-a-
9668 /// default — a required field here would fail the whole camp's config
9669 /// load, not just the service missing it.
9670 #[test]
9671 fn a_service_without_a_description_still_loads() {
9672 let tmp = tempfile::TempDir::new().unwrap();
9673 let root = tmp.path();
9674 let dir = root.join(".yah/services/api");
9675 std::fs::create_dir_all(&dir).unwrap();
9676 std::fs::write(
9677 dir.join("service.toml"),
9678 "schema_version = 1\nname = \"api\"\n[address]\nkind = \"front-door\"\ndomain = \"api.noisetable.com\"\n",
9679 )
9680 .unwrap();
9681
9682 let cfg = CloudConfig::load(root).unwrap();
9683 assert_eq!(cfg.service("api").unwrap().service.description, None);
9684 }
9685
9686 /// R926-F1. The push relay has no HTTP surface at all: it binds an
9687 /// iroh endpoint and serves one ALPN, so a `domain` for it could only
9688 /// ever be a placeholder, and a placeholder cannot be probed. This is
9689 /// the shape that replaced that dead end.
9690 #[test]
9691 fn a_node_addressed_service_loads_and_reports_no_domain() {
9692 let tmp = tempfile::TempDir::new().unwrap();
9693 let root = tmp.path();
9694 let dir = root.join(".yah/services/push-relay");
9695 std::fs::create_dir_all(&dir).unwrap();
9696 let node_id = "a".repeat(64);
9697 std::fs::write(
9698 dir.join("service.toml"),
9699 format!(
9700 "schema_version = 1\nname = \"push-relay\"\n\
9701 description = \"Mobile push fanout.\"\n\
9702 [address]\nkind = \"node\"\n\
9703 node_id = \"{node_id}\"\nalpn = \"yah/push-relay/1\"\n"
9704 ),
9705 )
9706 .unwrap();
9707
9708 let cfg = CloudConfig::load(root).unwrap();
9709 let svc = &cfg.service("push-relay").unwrap().service;
9710 assert_eq!(svc.domain(), None, "a node has no domain, not a fake one");
9711 assert_eq!(svc.health_path(), None);
9712 assert_eq!(
9713 svc.address,
9714 ServiceAddress::node(&node_id, "yah/push-relay/1")
9715 );
9716 // The label is what a status table prints. It must never be blank
9717 // and must never be a domain-shaped lie.
9718 assert_eq!(svc.address.label(), "node:aaaaaaaa/yah/push-relay/1");
9719 }
9720
9721 /// A caller that genuinely needs a domain gets a sentence naming the
9722 /// service and what it wanted one for — not an `unwrap` panic and not
9723 /// an empty string silently probing the wrong origin.
9724 #[test]
9725 fn require_domain_on_a_node_service_names_the_service_and_the_caller() {
9726 let svc = ServiceConfig {
9727 schema_version: 1,
9728 name: "push-relay".into(),
9729 description: None,
9730 address: ServiceAddress::node("b".repeat(64), "yah/push-relay/1"),
9731 components: vec![],
9732 db: DbCatalog::default(),
9733 };
9734 let err = svc
9735 .require_domain("a DNS record")
9736 .expect_err("a node service has no domain");
9737 let msg = err.to_string();
9738 assert!(msg.contains("push-relay"), "{msg}");
9739 assert!(msg.contains("a DNS record"), "{msg}");
9740 assert!(msg.contains("node:bbbbbbbb"), "{msg}");
9741 }
9742
9743 /// A truncated paste is the realistic way a `node_id` goes wrong, and
9744 /// it would otherwise surface as a dial timeout naming neither the
9745 /// file nor the field.
9746 #[test]
9747 fn a_truncated_node_id_is_refused_at_load() {
9748 let tmp = tempfile::TempDir::new().unwrap();
9749 let root = tmp.path();
9750 let dir = root.join(".yah/services/push-relay");
9751 std::fs::create_dir_all(&dir).unwrap();
9752 std::fs::write(
9753 dir.join("service.toml"),
9754 "schema_version = 1\nname = \"push-relay\"\n\
9755 [address]\nkind = \"node\"\n\
9756 node_id = \"abc123\"\nalpn = \"yah/push-relay/1\"\n",
9757 )
9758 .unwrap();
9759
9760 let err = CloudConfig::load(root).expect_err("a short node_id must not load");
9761 let msg = format!("{err:#}");
9762 assert!(msg.contains("node_id"), "{msg}");
9763 assert!(msg.contains("64 hex"), "{msg}");
9764 }
9765
9766 /// A NodeId with no ALPN names a process, not a service — there would
9767 /// be nothing to dial.
9768 #[test]
9769 fn a_node_address_without_an_alpn_is_refused_at_load() {
9770 let tmp = tempfile::TempDir::new().unwrap();
9771 let root = tmp.path();
9772 let dir = root.join(".yah/services/push-relay");
9773 std::fs::create_dir_all(&dir).unwrap();
9774 std::fs::write(
9775 dir.join("service.toml"),
9776 format!(
9777 "schema_version = 1\nname = \"push-relay\"\n\
9778 [address]\nkind = \"node\"\n\
9779 node_id = \"{}\"\nalpn = \"\"\n",
9780 "c".repeat(64)
9781 ),
9782 )
9783 .unwrap();
9784
9785 let err = CloudConfig::load(root).expect_err("an empty alpn must not load");
9786 assert!(format!("{err:#}").contains("alpn"), "{err:#}");
9787 }
9788
9789 /// `save` writes TOML that `load` accepts, for both address kinds. The
9790 /// ordering trap is real: an `[address]` table emitted before a scalar
9791 /// field would produce a file `toml` cannot parse back.
9792 #[test]
9793 fn both_address_kinds_survive_a_save_load_round_trip() {
9794 for address in [
9795 ServiceAddress::front_door_at("api.example", "/healthz"),
9796 ServiceAddress::node("d".repeat(64), "yah/push-relay/1"),
9797 ] {
9798 let tmp = tempfile::TempDir::new().unwrap();
9799 let root = tmp.path();
9800 let svc = ServiceConfig {
9801 schema_version: 1,
9802 name: "round-trip".into(),
9803 description: Some("described".into()),
9804 address: address.clone(),
9805 components: vec![],
9806 db: DbCatalog::default(),
9807 };
9808 svc.save(root).unwrap();
9809 let cfg = CloudConfig::load(root).unwrap();
9810 let back = &cfg.service("round-trip").unwrap().service;
9811 assert_eq!(back.address, address);
9812 assert_eq!(back.description.as_deref(), Some("described"));
9813 }
9814 }
9815
9816 /// R926. `.yah/services/headscale/service.toml` is the first service in
9817 /// the tree with NO components and NO `mirrors/` directory — it is
9818 /// registered to be described and probed, not deployed (the mesh leader
9819 /// places the appliance, not `yah cloud apply`). That shape has to load,
9820 /// because the alternative is that adding an observability-only service
9821 /// fails the config load for the whole camp.
9822 ///
9823 /// Also pins the query string: the coordination probe is `/key?v=138`,
9824 /// and a validator that got stricter about what follows the leading `/`
9825 /// would silently un-register the one service whose false-green took the
9826 /// mesh down for 37 hours.
9827 #[test]
9828 fn an_observability_only_service_loads_without_components_or_mirrors() {
9829 let tmp = tempfile::TempDir::new().unwrap();
9830 let root = tmp.path();
9831 let dir = root.join(".yah/services/headscale");
9832 std::fs::create_dir_all(&dir).unwrap();
9833 std::fs::write(
9834 dir.join("service.toml"),
9835 "schema_version = 1\n\
9836 name = \"headscale\"\n\
9837 description = \"Mesh coordination server.\"\n\
9838 [address]\n\
9839 kind = \"front-door\"\n\
9840 domain = \"cloud.mesh.yah.dev\"\n\
9841 health_path = \"/key?v=138\"\n",
9842 )
9843 .unwrap();
9844
9845 let cfg = CloudConfig::load(root).unwrap();
9846 let svc = cfg.service("headscale").unwrap();
9847 assert_eq!(svc.service.health_path(), Some("/key?v=138"));
9848 assert_eq!(
9849 svc.service.description.as_deref(),
9850 Some("Mesh coordination server.")
9851 );
9852 assert!(svc.service.components.is_empty());
9853 assert!(svc.mirrors.is_empty());
9854 }
9855
9856 #[test]
9857 fn a_relative_health_path_is_refused_at_load() {
9858 // Not a style rule. A relative path joins onto the origin differently
9859 // depending on which URL builder gets it, so the probe would ask a
9860 // question the file does not read as asking — and it would paint a
9861 // confident dot either way.
9862 let tmp = tempfile::TempDir::new().unwrap();
9863 let root = tmp.path();
9864
9865 ServiceConfig {
9866 schema_version: 1,
9867 name: "api".into(),
9868 address: ServiceAddress::front_door_at("api.noisetable.com", "api/v1/status"),
9869 description: None,
9870 components: vec![],
9871 db: DbCatalog::default(),
9872 }
9873 .save(root)
9874 .unwrap();
9875
9876 let err = CloudConfig::load(root).expect_err("a relative health_path must not load");
9877 let msg = format!("{err:#}");
9878 assert!(
9879 msg.contains("health_path") && msg.contains("/api/v1/status"),
9880 "the error must name the field and the fix, got: {msg}"
9881 );
9882 }
9883
9884 #[test]
9885 fn service_config_save_overwrites_in_place() {
9886 let tmp = tempfile::TempDir::new().unwrap();
9887 let root = tmp.path();
9888
9889 let mut svc = ServiceConfig {
9890 schema_version: 1,
9891 name: "dev-yah".into(),
9892 address: ServiceAddress::front_door("yah.dev"),
9893 description: None,
9894 components: vec![],
9895 db: DbCatalog::default(),
9896 };
9897 svc.save(root).unwrap();
9898 svc.address = ServiceAddress::front_door("yah.example");
9899 svc.save(root).unwrap();
9900
9901 let cfg = CloudConfig::load(root).unwrap();
9902 assert_eq!(
9903 cfg.service("dev-yah").unwrap().service.domain(),
9904 Some("yah.example")
9905 );
9906 }
9907
9908 #[test]
9909 fn mirror_config_save_round_trips_reference_and_inline_slots() {
9910 let tmp = tempfile::TempDir::new().unwrap();
9911 let root = tmp.path();
9912
9913 // A service must exist so the loader walks the mirrors/ dir.
9914 ServiceConfig {
9915 schema_version: 1,
9916 name: "dev-yah".into(),
9917 address: ServiceAddress::front_door("yah.dev"),
9918 description: None,
9919 components: vec![],
9920 db: DbCatalog::default(),
9921 }
9922 .save(root)
9923 .unwrap();
9924
9925 // The cloudflare provider the reference slot points at must resolve,
9926 // or CloudConfig::load's cross-ref check rejects the tree.
9927 let providers = crate::paths::providers_dir(root);
9928 std::fs::create_dir_all(&providers).unwrap();
9929 std::fs::write(
9930 providers.join("cloudflare.toml"),
9931 "schema_version = 1\nid = \"cloudflare\"\nkind = \"cloudflare\"\n",
9932 )
9933 .unwrap();
9934
9935 let mut providers_map = BTreeMap::new();
9936 providers_map.insert(
9937 "static".to_string(),
9938 MirrorProviderSlot::Reference {
9939 provider_id: "cloudflare".into(),
9940 fields: {
9941 let mut f = BTreeMap::new();
9942 f.insert("bucket".to_string(), toml::Value::String("yah-dev".into()));
9943 f
9944 },
9945 },
9946 );
9947 providers_map.insert(
9948 "compute".to_string(),
9949 MirrorProviderSlot::Inline {
9950 kind: Provider::MiniflareNative,
9951 fields: {
9952 let mut f = BTreeMap::new();
9953 f.insert("port".to_string(), toml::Value::Integer(4321));
9954 f
9955 },
9956 },
9957 );
9958 let mirror = MirrorConfig {
9959 schema_version: 1,
9960 shape: MirrorShape::SingleMachine,
9961 providers: providers_map,
9962 ingress: Default::default(),
9963 ingress_machines: Vec::new(),
9964 drivers: Default::default(),
9965 asset_aliases: Default::default(),
9966 build: Default::default(),
9967 };
9968 // Save with canonical name; legacy "cloud" is normalised to "prod" on load.
9969 mirror.save(root, "dev-yah", "prod").unwrap();
9970
9971 let path = crate::paths::service_mirror_toml(root, "dev-yah", "prod");
9972 assert!(
9973 path.exists(),
9974 "mirror toml should exist at {}",
9975 path.display()
9976 );
9977
9978 let cfg = CloudConfig::load(root).unwrap();
9979 let loaded = &cfg.service("dev-yah").unwrap().mirrors["prod"];
9980 assert_eq!(loaded.shape, MirrorShape::SingleMachine);
9981 assert_eq!(loaded.providers["static"].provider_id(), Some("cloudflare"));
9982 assert_eq!(
9983 loaded.providers["compute"].inline_kind(),
9984 Some(Provider::MiniflareNative)
9985 );
9986 }
9987
9988 #[test]
9989 fn service_delete_removes_dir_and_mirrors() {
9990 let tmp = tempfile::TempDir::new().unwrap();
9991 let root = tmp.path();
9992
9993 let svc = ServiceConfig {
9994 schema_version: 1,
9995 name: "dev-yah".into(),
9996 address: ServiceAddress::front_door("yah.dev"),
9997 description: None,
9998 components: vec![],
9999 db: DbCatalog::default(),
10000 };
10001 svc.save(root).unwrap();
10002 MirrorConfig {
10003 schema_version: 1,
10004 shape: MirrorShape::Local,
10005 providers: BTreeMap::new(),
10006 ingress: Default::default(),
10007 ingress_machines: Vec::new(),
10008 drivers: Default::default(),
10009 asset_aliases: Default::default(),
10010 build: Default::default(),
10011 }
10012 .save(root, "dev-yah", "local")
10013 .unwrap();
10014
10015 assert!(
10016 ServiceConfig::delete(root, "dev-yah").unwrap(),
10017 "first delete reports true"
10018 );
10019 assert!(!crate::paths::service_dir(root, "dev-yah").exists());
10020 // Idempotent: deleting again is a no-op that reports false.
10021 assert!(!ServiceConfig::delete(root, "dev-yah").unwrap());
10022
10023 let cfg = CloudConfig::load(root).unwrap();
10024 assert!(cfg.service("dev-yah").is_none());
10025 }
10026
10027 #[test]
10028 fn mirror_delete_leaves_other_mirrors_and_service_intact() {
10029 let tmp = tempfile::TempDir::new().unwrap();
10030 let root = tmp.path();
10031
10032 ServiceConfig {
10033 schema_version: 1,
10034 name: "dev-yah".into(),
10035 address: ServiceAddress::front_door("yah.dev"),
10036 description: None,
10037 components: vec![],
10038 db: DbCatalog::default(),
10039 }
10040 .save(root)
10041 .unwrap();
10042 for env in ["prod", "local"] {
10043 MirrorConfig {
10044 schema_version: 1,
10045 shape: MirrorShape::Local,
10046 providers: BTreeMap::new(),
10047 ingress: Default::default(),
10048 ingress_machines: Vec::new(),
10049 drivers: Default::default(),
10050 asset_aliases: Default::default(),
10051 build: Default::default(),
10052 }
10053 .save(root, "dev-yah", env)
10054 .unwrap();
10055 }
10056
10057 assert!(MirrorConfig::delete(root, "dev-yah", "prod").unwrap());
10058 assert!(!MirrorConfig::delete(root, "dev-yah", "prod").unwrap());
10059
10060 let cfg = CloudConfig::load(root).unwrap();
10061 let svc = cfg
10062 .service("dev-yah")
10063 .expect("service survives mirror delete");
10064 // "prod" is canonical and deletes directly; "local" normalises to "dev" on load.
10065 assert!(!svc.mirrors.contains_key("prod"));
10066 assert!(svc.mirrors.contains_key("dev"));
10067 }
10068
10069 // ─── DomainConfig (R347-F2) ────────────────────────────────────────────
10070
10071 fn write_marketing_service(root: &Path) {
10072 let svc = ServiceConfig {
10073 schema_version: 1,
10074 name: "yah-marketing".into(),
10075 address: ServiceAddress::front_door("yah.dev"),
10076 description: None,
10077 db: DbCatalog::default(),
10078 components: vec![ServiceComponent {
10079 mount: None,
10080 id: "site".into(),
10081 kind: "mesofact-static".into(),
10082 path: "app/yah/web".into(),
10083 role: "static".into(),
10084 publishes: None,
10085 wave: 0,
10086 git: None,
10087 deploy: Default::default(),
10088 }],
10089 };
10090 svc.save(root).unwrap();
10091 }
10092
10093 #[test]
10094 fn round_trip_domain_with_each_route_mode() {
10095 let dom = DomainConfig {
10096 schema_version: 1,
10097 name: "yah-dev".into(),
10098 domain: "yah.dev".into(),
10099 front_door: FrontDoor::Worker,
10100 cdn_bucket: "yah-dev".into(),
10101 worker_bundle_path: Some(".yah/workers/yah-dev/".into()),
10102 routes: vec![
10103 DomainRoute {
10104 headers: Default::default(),
10105 path: "/".into(),
10106 mode: RouteMode::Static {
10107 component: "yah-marketing/site".into(),
10108 },
10109 },
10110 DomainRoute {
10111 headers: Default::default(),
10112 path: "/dashboard/api/*".into(),
10113 mode: RouteMode::Backend {
10114 component: "yah-dashboard/api".into(),
10115 origin: "https://api.dashboard.yah.dev".into(),
10116 origin_path: None,
10117 },
10118 },
10119 DomainRoute {
10120 headers: Default::default(),
10121 path: "/old".into(),
10122 mode: RouteMode::Redirect {
10123 target: "https://yah.dev/blog".into(),
10124 status: 308,
10125 },
10126 },
10127 ],
10128 };
10129 let s = toml::to_string(&dom).unwrap();
10130 let back: DomainConfig = toml::from_str(&s).unwrap();
10131 assert_eq!(back.name, "yah-dev");
10132 assert_eq!(back.routes.len(), 3);
10133 assert!(matches!(back.routes[0].mode, RouteMode::Static { .. }));
10134 assert!(matches!(back.routes[1].mode, RouteMode::Backend { .. }));
10135 assert!(matches!(back.routes[2].mode, RouteMode::Redirect { .. }));
10136 }
10137
10138 /// A one-route `cdn.noisetable.com`-shaped manifest whose static route body
10139 /// is `body`.
10140 fn static_route_manifest(body: &str) -> String {
10141 format!(
10142 "schema_version = 1\nname = \"cdn-noisetable-com\"\ndomain = \"cdn.noisetable.com\"\n\
10143 front_door = \"worker\"\ncdn_bucket = \"noisetable-marketing\"\n\n\
10144 [[routes]]\npath = \"/engine/*\"\nmode = \"static\"\n{body}\n"
10145 )
10146 }
10147
10148 /// R560-F13 — a static route names a `component` OR a `bucket`, never both
10149 /// and never neither, and a bucket has to be a real R2 bucket name.
10150 #[test]
10151 fn a_static_route_names_exactly_one_of_component_or_bucket() {
10152 let dom: DomainConfig =
10153 toml::from_str(&static_route_manifest("bucket = \"noisetable-releases\"")).unwrap();
10154 assert!(matches!(
10155 &dom.routes[0].mode,
10156 RouteMode::StaticBucket { bucket } if bucket == "noisetable-releases"
10157 ));
10158 dom.validate_front_door().unwrap();
10159
10160 let dom: DomainConfig =
10161 toml::from_str(&static_route_manifest("component = \"svc/site\"")).unwrap();
10162 assert!(matches!(
10163 &dom.routes[0].mode,
10164 RouteMode::Static { component } if component == "svc/site"
10165 ));
10166
10167 for (body, needle) in [
10168 (
10169 "component = \"svc/site\"\nbucket = \"noisetable-releases\"",
10170 "both",
10171 ),
10172 ("", "neither"),
10173 ("bucket = \"Noisetable_Releases\"", "not an R2 bucket name"),
10174 ] {
10175 let err = toml::from_str::<DomainConfig>(&static_route_manifest(body))
10176 .unwrap_err()
10177 .to_string();
10178 assert!(err.contains(needle), "{body:?}: {err}");
10179 }
10180 }
10181
10182 /// The Rust variant is `StaticBucket`; the manifest spelling stays
10183 /// `mode = "static"` + `bucket`, through a save as well as a load.
10184 #[test]
10185 fn a_bucket_route_round_trips_as_mode_static_with_a_bucket_key() {
10186 let dom: DomainConfig =
10187 toml::from_str(&static_route_manifest("bucket = \"noisetable-releases\"")).unwrap();
10188 let s = toml::to_string(&dom).unwrap();
10189 assert!(s.contains("mode = \"static\""), "{s}");
10190 assert!(s.contains("bucket = \"noisetable-releases\""), "{s}");
10191 assert!(!s.contains("component"), "{s}");
10192 let back: DomainConfig = toml::from_str(&s).unwrap();
10193 assert!(matches!(back.routes[0].mode, RouteMode::StaticBucket { .. }));
10194 }
10195
10196 /// Passway has no R2 read path, so a bucket route on it is refused at load
10197 /// naming the route, not compiled into an entry that door cannot serve.
10198 #[test]
10199 fn passway_refuses_a_bucket_route() {
10200 let mut dom: DomainConfig =
10201 toml::from_str(&static_route_manifest("bucket = \"noisetable-releases\"")).unwrap();
10202 dom.front_door = FrontDoor::Passway;
10203 let err = dom.validate_front_door().unwrap_err().to_string();
10204 assert!(err.contains("passway") && err.contains("/engine/*"), "{err}");
10205 }
10206
10207 #[test]
10208 fn redirect_status_defaults_to_308() {
10209 let src = r#"
10210schema_version = 1
10211name = "yah-dev"
10212domain = "yah.dev"
10213front_door = "worker"
10214cdn_bucket = "yah-dev"
10215
10216[[routes]]
10217path = "/old"
10218mode = "redirect"
10219target = "https://yah.dev/blog"
10220"#;
10221 let dom: DomainConfig = toml::from_str(src).unwrap();
10222 let RouteMode::Redirect { status, .. } = &dom.routes[0].mode else {
10223 panic!("expected redirect");
10224 };
10225 assert_eq!(*status, 308);
10226 }
10227
10228 #[test]
10229 fn missing_domains_dir_is_empty() {
10230 let tmp = tempfile::TempDir::new().unwrap();
10231 // R844-B7: `.yah/` must exist or this is a wrong-root error rather
10232 // than an empty tree. The absent directory under test is `domains/`.
10233 std::fs::create_dir_all(tmp.path().join(".yah")).unwrap();
10234 let cfg = CloudConfig::load(tmp.path()).unwrap();
10235 assert!(cfg.domains.is_empty());
10236 }
10237
10238 #[test]
10239 fn save_reload_roundtrip() {
10240 let tmp = tempfile::TempDir::new().unwrap();
10241 let root = tmp.path();
10242 write_marketing_service(root);
10243
10244 let dom = DomainConfig {
10245 schema_version: 1,
10246 name: "yah-dev".into(),
10247 domain: "yah.dev".into(),
10248 front_door: FrontDoor::Worker,
10249 cdn_bucket: "yah-dev".into(),
10250 worker_bundle_path: None,
10251 routes: vec![DomainRoute {
10252 headers: Default::default(),
10253 path: "/".into(),
10254 mode: RouteMode::Static {
10255 component: "yah-marketing/site".into(),
10256 },
10257 }],
10258 };
10259 dom.save(root).unwrap();
10260
10261 let cfg = CloudConfig::load(root).unwrap();
10262 let loaded = cfg.domain("yah-dev").expect("yah-dev domain");
10263 assert_eq!(loaded.domain, "yah.dev");
10264 assert_eq!(loaded.routes.len(), 1);
10265 }
10266
10267 #[test]
10268 fn delete_returns_false_when_absent() {
10269 let tmp = tempfile::TempDir::new().unwrap();
10270 assert!(!DomainConfig::delete(tmp.path(), "no-such-domain").unwrap());
10271 }
10272
10273 #[test]
10274 fn delete_returns_true_first_time() {
10275 let tmp = tempfile::TempDir::new().unwrap();
10276 let root = tmp.path();
10277 let dom = DomainConfig {
10278 schema_version: 1,
10279 name: "yah-dev".into(),
10280 domain: "yah.dev".into(),
10281 front_door: FrontDoor::BucketDirect,
10282 cdn_bucket: "yah-dev".into(),
10283 worker_bundle_path: None,
10284 routes: vec![],
10285 };
10286 dom.save(root).unwrap();
10287 assert!(DomainConfig::delete(root, "yah-dev").unwrap());
10288 assert!(!DomainConfig::delete(root, "yah-dev").unwrap());
10289 }
10290
10291 // ---- R594-F12: front-door discriminator ------------------------------
10292
10293 /// Write a raw domain manifest so the tests exercise the deserialize +
10294 /// validate path, not a hand-built struct that skipped serde.
10295 fn write_domain_toml(root: &Path, stem: &str, body: &str) {
10296 let dir = root.join(".yah").join("domains");
10297 std::fs::create_dir_all(&dir).unwrap();
10298 std::fs::write(dir.join(format!("{stem}.toml")), body).unwrap();
10299 }
10300
10301 #[test]
10302 fn front_door_is_required() {
10303 let tmp = tempfile::TempDir::new().unwrap();
10304 let root = tmp.path();
10305 write_marketing_service(root);
10306 write_domain_toml(
10307 root,
10308 "yah-dev",
10309 r#"
10310schema_version = 1
10311name = "yah-dev"
10312domain = "yah.dev"
10313cdn_bucket = "yah-dev"
10314[[routes]]
10315path = "/*"
10316mode = "static"
10317component = "yah-marketing/site"
10318"#,
10319 );
10320 let err = CloudConfig::load(root).unwrap_err().to_string();
10321 // serde's own missing-field message; the point is that omitting the
10322 // discriminator is not a silently-defaulted state.
10323 assert!(err.contains("yah-dev.toml"), "{err}");
10324 }
10325
10326 #[test]
10327 fn bucket_direct_with_routes_is_rejected() {
10328 let tmp = tempfile::TempDir::new().unwrap();
10329 let root = tmp.path();
10330 write_marketing_service(root);
10331 write_domain_toml(
10332 root,
10333 "cdn-yah-dev",
10334 r#"
10335schema_version = 1
10336name = "cdn-yah-dev"
10337domain = "cdn.yah.dev"
10338front_door = "bucket-direct"
10339cdn_bucket = "yah-dev"
10340[[routes]]
10341path = "/docs/*"
10342mode = "static"
10343component = "yah-marketing/site"
10344"#,
10345 );
10346 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10347 assert!(err.contains("front_door"), "{err}");
10348 assert!(err.contains("/docs/*"), "{err}");
10349 }
10350
10351 #[test]
10352 fn bucket_direct_with_worker_bundle_path_is_rejected() {
10353 let tmp = tempfile::TempDir::new().unwrap();
10354 let root = tmp.path();
10355 write_domain_toml(
10356 root,
10357 "cdn-yah-dev",
10358 r#"
10359schema_version = 1
10360name = "cdn-yah-dev"
10361domain = "cdn.yah.dev"
10362front_door = "bucket-direct"
10363cdn_bucket = "yah-dev"
10364worker_bundle_path = ".yah/workers/cdn-yah-dev/"
10365"#,
10366 );
10367 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10368 assert!(err.contains("worker_bundle_path"), "{err}");
10369 }
10370
10371 // ── R746: per-route response headers + component mounts ──────────────────
10372
10373 /// A two-component service: `site` at the root, `app` mounted at `/app`
10374 /// with isolation headers on its route. This is the noisetable.com shape
10375 /// the primitive was built for.
10376 fn write_two_component_service(root: &Path) {
10377 let svc = ServiceConfig {
10378 schema_version: 1,
10379 name: "yah-marketing".into(),
10380 address: ServiceAddress::front_door("yah.dev"),
10381 description: None,
10382 db: DbCatalog::default(),
10383 components: vec![
10384 ServiceComponent {
10385 mount: None,
10386 id: "site".into(),
10387 kind: "mesofact-static".into(),
10388 path: "app/yah/web".into(),
10389 role: "static".into(),
10390 publishes: None,
10391 wave: 0,
10392 git: None,
10393 deploy: Default::default(),
10394 },
10395 ServiceComponent {
10396 mount: Some("/app".into()),
10397 id: "app".into(),
10398 kind: "mesofact-static".into(),
10399 path: "app/browser".into(),
10400 role: "static".into(),
10401 publishes: None,
10402 wave: 0,
10403 git: None,
10404 deploy: Default::default(),
10405 },
10406 ],
10407 };
10408 svc.save(root).unwrap();
10409 }
10410
10411 const MOUNTED_DOMAIN: &str = r#"
10412schema_version = 1
10413name = "yah-dev"
10414domain = "yah.dev"
10415front_door = "worker"
10416cdn_bucket = "yah-dev"
10417
10418[[routes]]
10419path = "/app/*"
10420mode = "static"
10421component = "yah-marketing/app"
10422headers = { "Cross-Origin-Opener-Policy" = "same-origin", "Cross-Origin-Embedder-Policy" = "require-corp" }
10423
10424[[routes]]
10425path = "/*"
10426mode = "static"
10427component = "yah-marketing/site"
10428"#;
10429
10430 #[test]
10431 fn a_mounted_component_routed_at_its_mount_loads() {
10432 let tmp = tempfile::TempDir::new().unwrap();
10433 let root = tmp.path();
10434 write_two_component_service(root);
10435 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
10436 let cfg = CloudConfig::load(root).unwrap();
10437 let dom = cfg.domain("yah-dev").unwrap();
10438 assert_eq!(dom.routes.len(), 2);
10439 assert_eq!(
10440 dom.routes[0].headers.get("Cross-Origin-Opener-Policy").map(String::as_str),
10441 Some("same-origin")
10442 );
10443 assert!(dom.routes[1].headers.is_empty());
10444 }
10445
10446 /// The header table reaches the Worker in MANIFEST order with headerless
10447 /// routes dropped. Order is the whole contract — the front door applies the
10448 /// first match, so `/app/*` before `/*` is what isolates the app without
10449 /// isolating the marketing site.
10450 #[test]
10451 fn route_headers_json_preserves_order_and_drops_headerless_routes() {
10452 let tmp = tempfile::TempDir::new().unwrap();
10453 let root = tmp.path();
10454 write_two_component_service(root);
10455 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
10456 let cfg = CloudConfig::load(root).unwrap();
10457 let json = cfg.domain("yah-dev").unwrap().route_headers_json();
10458
10459 let parsed: serde_json::Value = serde_json::from_str(&json).unwrap();
10460 let rules = parsed.as_array().unwrap();
10461 assert_eq!(rules.len(), 1, "the headerless catch-all is dropped: {json}");
10462 assert_eq!(rules[0]["path"], "/app/*");
10463 assert_eq!(rules[0]["headers"]["Cross-Origin-Embedder-Policy"], "require-corp");
10464 }
10465
10466 #[test]
10467 fn route_headers_json_is_an_empty_array_when_nothing_declares_headers() {
10468 let tmp = tempfile::TempDir::new().unwrap();
10469 let root = tmp.path();
10470 write_marketing_service(root);
10471 write_domain_toml(
10472 root,
10473 "yah-dev",
10474 r#"
10475schema_version = 1
10476name = "yah-dev"
10477domain = "yah.dev"
10478front_door = "worker"
10479cdn_bucket = "yah-dev"
10480
10481[[routes]]
10482path = "/*"
10483mode = "static"
10484component = "yah-marketing/site"
10485"#,
10486 );
10487 let cfg = CloudConfig::load(root).unwrap();
10488 assert_eq!(cfg.domain("yah-dev").unwrap().route_headers_json(), "[]");
10489 }
10490
10491 /// The reconciler's own entry point: given a workspace root and a service
10492 /// name, produce the binding value. `"[]"` when nothing routes the service.
10493 #[test]
10494 fn route_headers_for_service_reads_the_workspace_domains() {
10495 let tmp = tempfile::TempDir::new().unwrap();
10496 let root = tmp.path();
10497 write_two_component_service(root);
10498 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
10499 assert!(route_headers_for_service(root, "yah-marketing")
10500 .unwrap()
10501 .contains("require-corp"));
10502 assert_eq!(route_headers_for_service(root, "some-other-svc").unwrap(), "[]");
10503 }
10504
10505 // ---- R749-T5: a broken table fails the DEPLOY, not the edge -----------
10506
10507 /// The manifest's `headers` map is hand-written TOML, so a header name with
10508 /// spaces in it is one keystroke away — and it survives serialization into
10509 /// a structurally-valid table that neither front door can apply. Fail at
10510 /// load, naming the domain, the route and the header, instead of shipping a
10511 /// binding the Worker throws on and an origin that refuses to boot.
10512 #[test]
10513 fn a_route_header_name_that_is_not_a_header_name_fails_the_load() {
10514 let tmp = tempfile::TempDir::new().unwrap();
10515 let root = tmp.path();
10516 write_marketing_service(root);
10517 write_domain_toml(
10518 root,
10519 "yah-dev",
10520 r#"
10521schema_version = 1
10522name = "yah-dev"
10523domain = "yah.dev"
10524front_door = "worker"
10525cdn_bucket = "yah-dev"
10526
10527[[routes]]
10528path = "/*"
10529mode = "static"
10530component = "yah-marketing/site"
10531headers = { "Cross Origin Opener Policy" = "same-origin" }
10532"#,
10533 );
10534 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10535 assert!(err.contains("yah-dev"), "{err}");
10536 assert!(err.contains("/*"), "{err}");
10537 assert!(err.contains("Cross Origin Opener Policy"), "{err}");
10538 assert!(err.contains("not a valid HTTP header name"), "{err}");
10539 }
10540
10541 /// A newline in a value is header injection if it ever reached the wire, so
10542 /// both doors reject it and so does this.
10543 #[test]
10544 fn a_route_header_value_that_is_not_a_header_value_fails_the_load() {
10545 let tmp = tempfile::TempDir::new().unwrap();
10546 let root = tmp.path();
10547 write_marketing_service(root);
10548 write_domain_toml(
10549 root,
10550 "yah-dev",
10551 r#"
10552schema_version = 1
10553name = "yah-dev"
10554domain = "yah.dev"
10555front_door = "worker"
10556cdn_bucket = "yah-dev"
10557
10558[[routes]]
10559path = "/*"
10560mode = "static"
10561component = "yah-marketing/site"
10562headers = { "X-Frame-Options" = "DENY\nSet-Cookie: pwned=1" }
10563"#,
10564 );
10565 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10566 assert!(err.contains("X-Frame-Options"), "{err}");
10567 assert!(err.contains("not a valid HTTP header value"), "{err}");
10568 }
10569
10570 /// The invariant this gate exists to hold: everything `route_headers_json`
10571 /// emits is applicable. A headerless route contributes no rule, so its path
10572 /// is not the table's business — only rules that ship are checked.
10573 #[test]
10574 fn a_headerless_route_is_not_subject_to_the_route_header_gate() {
10575 let tmp = tempfile::TempDir::new().unwrap();
10576 let root = tmp.path();
10577 write_two_component_service(root);
10578 write_domain_toml(root, "yah-dev", MOUNTED_DOMAIN);
10579 let cfg = CloudConfig::load(root).unwrap();
10580 cfg.domain("yah-dev")
10581 .unwrap()
10582 .validate_route_headers()
10583 .unwrap();
10584 }
10585
10586 /// A `bucket-direct` domain has no front door to set headers on, so it must
10587 /// not be picked up as a service's header source.
10588 #[test]
10589 fn route_headers_ignores_domains_that_are_not_route_driven() {
10590 let doms: BTreeMap<String, DomainConfig> = [(
10591 "cdn".to_string(),
10592 DomainConfig {
10593 schema_version: 1,
10594 name: "cdn".into(),
10595 domain: "cdn.yah.dev".into(),
10596 front_door: FrontDoor::BucketDirect,
10597 cdn_bucket: "yah-dev".into(),
10598 worker_bundle_path: None,
10599 routes: vec![],
10600 },
10601 )]
10602 .into_iter()
10603 .collect();
10604 assert!(domain_serving_service(&doms, "yah-marketing").is_none());
10605 }
10606
10607 #[test]
10608 fn a_mount_that_disagrees_with_its_route_path_is_rejected() {
10609 let tmp = tempfile::TempDir::new().unwrap();
10610 let root = tmp.path();
10611 write_two_component_service(root);
10612 write_domain_toml(
10613 root,
10614 "yah-dev",
10615 r#"
10616schema_version = 1
10617name = "yah-dev"
10618domain = "yah.dev"
10619front_door = "worker"
10620cdn_bucket = "yah-dev"
10621
10622[[routes]]
10623path = "/studio/*"
10624mode = "static"
10625component = "yah-marketing/app"
10626"#,
10627 );
10628 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10629 assert!(err.contains("mount = \"/app\""), "{err}");
10630 assert!(err.contains("/studio/*"), "{err}");
10631 }
10632
10633 /// The other direction: routing an unmounted component under a sub-path
10634 /// points requests at a prefix nothing published to.
10635 #[test]
10636 fn routing_an_unmounted_component_under_a_subpath_is_rejected() {
10637 let tmp = tempfile::TempDir::new().unwrap();
10638 let root = tmp.path();
10639 write_marketing_service(root);
10640 write_domain_toml(
10641 root,
10642 "yah-dev",
10643 r#"
10644schema_version = 1
10645name = "yah-dev"
10646domain = "yah.dev"
10647front_door = "worker"
10648cdn_bucket = "yah-dev"
10649
10650[[routes]]
10651path = "/docs/*"
10652mode = "static"
10653component = "yah-marketing/site"
10654"#,
10655 );
10656 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10657 assert!(err.contains("no `mount`"), "{err}");
10658 assert!(err.contains("/docs"), "{err}");
10659 }
10660
10661 /// R931-B2: the mount/route cross-check must fire for `mode = "backend"`
10662 /// too, not just `mode = "static"` — [`RouteMode::component`] resolves a
10663 /// component reference for both, and a mounted component routed under the
10664 /// wrong prefix is the same silent-404 defect regardless of which mode
10665 /// serves it. Mutation-proved before this fix: `yah cloud validate`
10666 /// exited 0 against exactly this mismatch.
10667 #[test]
10668 fn a_backend_mode_mount_that_disagrees_with_its_route_path_is_rejected() {
10669 let tmp = tempfile::TempDir::new().unwrap();
10670 let root = tmp.path();
10671 write_two_component_service(root);
10672 write_domain_toml(
10673 root,
10674 "yah-dev",
10675 r#"
10676schema_version = 1
10677name = "yah-dev"
10678domain = "yah.dev"
10679front_door = "worker"
10680cdn_bucket = "yah-dev"
10681
10682[[routes]]
10683path = "/studio/*"
10684mode = "backend"
10685component = "yah-marketing/app"
10686origin = "http://127.0.0.1:9000"
10687"#,
10688 );
10689 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10690 assert!(err.contains("mount = \"/app\""), "{err}");
10691 assert!(err.contains("/studio/*"), "{err}");
10692 }
10693
10694 /// The positive control for the same widening: a `backend` route at its
10695 /// component's actual mount must still load cleanly.
10696 #[test]
10697 fn a_backend_mode_route_matching_its_mount_loads() {
10698 let tmp = tempfile::TempDir::new().unwrap();
10699 let root = tmp.path();
10700 write_two_component_service(root);
10701 write_domain_toml(
10702 root,
10703 "yah-dev",
10704 r#"
10705schema_version = 1
10706name = "yah-dev"
10707domain = "yah.dev"
10708front_door = "worker"
10709cdn_bucket = "yah-dev"
10710
10711[[routes]]
10712path = "/app/*"
10713mode = "backend"
10714component = "yah-marketing/app"
10715origin = "http://127.0.0.1:9000"
10716"#,
10717 );
10718 let cfg = CloudConfig::load(root).unwrap();
10719 assert_eq!(cfg.domain("yah-dev").unwrap().routes.len(), 1);
10720 }
10721
10722 #[test]
10723 fn mount_and_route_prefix_normalization_agree() {
10724 for m in ["/app", "app", "app/", "/app/"] {
10725 assert_eq!(normalize_mount(m), "app", "mount {m:?}");
10726 }
10727 assert_eq!(normalize_mount("/"), "");
10728 assert_eq!(route_path_prefix("/*"), "");
10729 assert_eq!(route_path_prefix("/app/*"), "app");
10730 assert_eq!(route_path_prefix("/app"), "app");
10731 assert_eq!(route_path_prefix("/"), "");
10732 }
10733
10734 // ── R870-B11: a mount is owned by exactly one bundle-tier component ────
10735
10736 /// Two bundle-tier components at the same explicit mount would stage into
10737 /// the same `app/dist/<mount>/` prefix inside one assembled bundle and
10738 /// silently clobber each other — reject at load, before that happens.
10739 #[test]
10740 fn two_bundle_components_at_the_same_mount_are_rejected() {
10741 let tmp = tempfile::TempDir::new().unwrap();
10742 let root = tmp.path();
10743 let svc = ServiceConfig {
10744 schema_version: 1,
10745 name: "noisetable-marketing".into(),
10746 address: ServiceAddress::front_door("noisetable.com"),
10747 description: None,
10748 db: DbCatalog::default(),
10749 components: vec![
10750 ServiceComponent {
10751 mount: Some("/app".into()),
10752 id: "app".into(),
10753 kind: "mesofact-static".into(),
10754 path: "app/browser".into(),
10755 role: "static".into(),
10756 publishes: None,
10757 wave: 0,
10758 git: None,
10759 deploy: Default::default(),
10760 },
10761 ServiceComponent {
10762 mount: Some("app/".into()),
10763 id: "app2".into(),
10764 kind: "mesofact-spa".into(),
10765 path: "app/other".into(),
10766 role: "static".into(),
10767 publishes: None,
10768 wave: 0,
10769 git: None,
10770 deploy: Default::default(),
10771 },
10772 ],
10773 };
10774 svc.save(root).unwrap();
10775 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10776 assert!(err.contains("\"app\""), "{err}");
10777 assert!(err.contains("\"app2\""), "{err}");
10778 assert!(err.contains("mount = \"/app\""), "{err}");
10779 }
10780
10781 /// The unmounted case: two bundle-tier components both leaving `mount`
10782 /// unset both claim the service root, which collides exactly the same
10783 /// way — this is the noisetable shape the ticket was filed against, if
10784 /// `app`'s mount had been forgotten instead of declared.
10785 #[test]
10786 fn two_bundle_components_with_no_mount_are_rejected() {
10787 let tmp = tempfile::TempDir::new().unwrap();
10788 let root = tmp.path();
10789 let svc = ServiceConfig {
10790 schema_version: 1,
10791 name: "noisetable-marketing".into(),
10792 address: ServiceAddress::front_door("noisetable.com"),
10793 description: None,
10794 db: DbCatalog::default(),
10795 components: vec![
10796 ServiceComponent {
10797 mount: None,
10798 id: "site".into(),
10799 kind: "mesofact-spa".into(),
10800 path: "web/landing".into(),
10801 role: "static".into(),
10802 publishes: None,
10803 wave: 0,
10804 git: None,
10805 deploy: Default::default(),
10806 },
10807 ServiceComponent {
10808 mount: None,
10809 id: "app".into(),
10810 kind: "mesofact-static".into(),
10811 path: "app/browser".into(),
10812 role: "static".into(),
10813 publishes: None,
10814 wave: 0,
10815 git: None,
10816 deploy: Default::default(),
10817 },
10818 ],
10819 };
10820 svc.save(root).unwrap();
10821 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10822 assert!(err.contains("\"site\""), "{err}");
10823 assert!(err.contains("\"app\""), "{err}");
10824 assert!(err.contains("the service root"), "{err}");
10825 }
10826
10827 /// R870-F23 widened the same loop to the workload tier. Two components in
10828 /// DIFFERENT tiers at one mount is the same clobber read from the routing
10829 /// side: the inner door's table names one upstream for that prefix, so a
10830 /// request reaches either the bundle or the workload and nothing says
10831 /// which.
10832 #[test]
10833 fn a_bundle_component_and_a_workload_component_at_one_mount_are_rejected() {
10834 let tmp = tempfile::TempDir::new().unwrap();
10835 let root = tmp.path();
10836 let svc = ServiceConfig {
10837 schema_version: 1,
10838 name: "noisetable".into(),
10839 address: ServiceAddress::front_door("noisetable.com"),
10840 description: None,
10841 db: DbCatalog::default(),
10842 components: vec![
10843 ServiceComponent {
10844 mount: Some("app".into()),
10845 id: "app-bundle".into(),
10846 kind: "mesofact-spa".into(),
10847 path: "app/browser".into(),
10848 role: "static".into(),
10849 publishes: None,
10850 wave: 0,
10851 git: None,
10852 deploy: DeployTier::Bundle,
10853 },
10854 ServiceComponent {
10855 mount: Some("/app/".into()),
10856 id: "app-service".into(),
10857 kind: "container".into(),
10858 path: "app/server".into(),
10859 role: "compute".into(),
10860 publishes: None,
10861 wave: 0,
10862 git: None,
10863 deploy: DeployTier::Workload,
10864 },
10865 ],
10866 };
10867 svc.save(root).unwrap();
10868 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10869 assert!(err.contains("\"app-bundle\""), "{err}");
10870 assert!(err.contains("\"app-service\""), "{err}");
10871 assert!(err.contains("deploys as its own workload"), "{err}");
10872 }
10873
10874 /// And two WORKLOAD-tier components at one mount, which the pre-R870-F23
10875 /// loop skipped entirely (it filtered on `kind`, and a workload-tier
10876 /// component need not be a mesofact kind at all).
10877 #[test]
10878 fn two_workload_components_at_one_mount_are_rejected() {
10879 let tmp = tempfile::TempDir::new().unwrap();
10880 let root = tmp.path();
10881 let svc = ServiceConfig {
10882 schema_version: 1,
10883 name: "noisetable".into(),
10884 address: ServiceAddress::front_door("noisetable.com"),
10885 description: None,
10886 db: DbCatalog::default(),
10887 components: vec![
10888 ServiceComponent {
10889 mount: None,
10890 id: "api".into(),
10891 kind: "container".into(),
10892 path: "svc/api".into(),
10893 role: "compute".into(),
10894 publishes: None,
10895 wave: 0,
10896 git: None,
10897 deploy: DeployTier::Workload,
10898 },
10899 ServiceComponent {
10900 mount: Some("/".into()),
10901 id: "api2".into(),
10902 kind: "container".into(),
10903 path: "svc/api2".into(),
10904 role: "compute".into(),
10905 publishes: None,
10906 wave: 0,
10907 git: None,
10908 deploy: DeployTier::Workload,
10909 },
10910 ],
10911 };
10912 svc.save(root).unwrap();
10913 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10914 assert!(err.contains("inner-door route table"), "{err}");
10915 assert!(err.contains("the service root"), "{err}");
10916 }
10917
10918 /// The legitimate shape (distinct mounts) is untouched — regression guard
10919 /// so the new check does not become the next silent-overwrite bug.
10920 #[test]
10921 fn bundle_components_at_distinct_mounts_still_load() {
10922 let tmp = tempfile::TempDir::new().unwrap();
10923 let root = tmp.path();
10924 write_two_component_service(root);
10925 assert!(CloudConfig::load(root).is_ok());
10926 }
10927
10928 #[test]
10929 fn worker_with_no_routes_is_rejected() {
10930 let tmp = tempfile::TempDir::new().unwrap();
10931 let root = tmp.path();
10932 write_domain_toml(
10933 root,
10934 "yah-dev",
10935 r#"
10936schema_version = 1
10937name = "yah-dev"
10938domain = "yah.dev"
10939front_door = "worker"
10940cdn_bucket = "yah-dev"
10941"#,
10942 );
10943 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10944 assert!(err.contains("front_door = \"worker\""), "{err}");
10945 assert!(err.contains("404"), "{err}");
10946 }
10947
10948 #[test]
10949 fn passway_with_no_routes_is_rejected_too() {
10950 let tmp = tempfile::TempDir::new().unwrap();
10951 let root = tmp.path();
10952 write_domain_toml(
10953 root,
10954 "yah-dev",
10955 r#"
10956schema_version = 1
10957name = "yah-dev"
10958domain = "yah.dev"
10959front_door = "passway"
10960cdn_bucket = "yah-dev"
10961"#,
10962 );
10963 let err = format!("{:#}", CloudConfig::load(root).unwrap_err());
10964 assert!(err.contains("front_door = \"passway\""), "{err}");
10965 }
10966
10967 #[test]
10968 fn bucket_direct_without_routes_loads() {
10969 let tmp = tempfile::TempDir::new().unwrap();
10970 let root = tmp.path();
10971 // Exactly the shape .yah/domains/cdn-yah-dev.toml ships (W175: a pure
10972 // asset tier deliberately has no Worker behaviours).
10973 write_domain_toml(
10974 root,
10975 "cdn-yah-dev",
10976 r#"
10977schema_version = 1
10978name = "cdn-yah-dev"
10979domain = "cdn.yah.dev"
10980front_door = "bucket-direct"
10981cdn_bucket = "yah-dev"
10982"#,
10983 );
10984 let cfg = CloudConfig::load(root).unwrap();
10985 let dom = cfg.domain("cdn-yah-dev").expect("cdn-yah-dev domain");
10986 assert_eq!(dom.front_door, FrontDoor::BucketDirect);
10987 assert!(!dom.front_door.is_route_driven());
10988 }
10989
10990 #[test]
10991 fn front_door_round_trips_through_save() {
10992 let tmp = tempfile::TempDir::new().unwrap();
10993 let root = tmp.path();
10994 write_marketing_service(root);
10995 let dom = DomainConfig {
10996 schema_version: 1,
10997 name: "yah-dev".into(),
10998 domain: "yah.dev".into(),
10999 front_door: FrontDoor::Passway,
11000 cdn_bucket: "yah-dev".into(),
11001 worker_bundle_path: None,
11002 routes: vec![DomainRoute {
11003 headers: Default::default(),
11004 path: "/*".into(),
11005 mode: RouteMode::Static {
11006 component: "yah-marketing/site".into(),
11007 },
11008 }],
11009 };
11010 dom.save(root).unwrap();
11011 let cfg = CloudConfig::load(root).unwrap();
11012 assert_eq!(
11013 cfg.domain("yah-dev").unwrap().front_door,
11014 FrontDoor::Passway
11015 );
11016 }
11017
11018 // The four manifests this repo actually ships are asserted in
11019 // `tests/live_workspace_smoke.rs` — that's the only place with a
11020 // depth-agnostic path to the live `.yah/` tree and a skip path for the
11021 // standalone mirror checkout.
11022
11023 #[test]
11024 fn cross_ref_bails_on_missing_service() {
11025 let tmp = tempfile::TempDir::new().unwrap();
11026 let root = tmp.path();
11027 // No services declared at all — component ref must fail to resolve.
11028 let dom = DomainConfig {
11029 schema_version: 1,
11030 name: "yah-dev".into(),
11031 domain: "yah.dev".into(),
11032 front_door: FrontDoor::Worker,
11033 cdn_bucket: "yah-dev".into(),
11034 worker_bundle_path: None,
11035 routes: vec![DomainRoute {
11036 headers: Default::default(),
11037 path: "/".into(),
11038 mode: RouteMode::Static {
11039 component: "yah-marketing/site".into(),
11040 },
11041 }],
11042 };
11043 dom.save(root).unwrap();
11044
11045 let err = CloudConfig::load(root).unwrap_err();
11046 let msg = format!("{err:#}");
11047 assert!(msg.contains("no such service"), "got: {msg}");
11048 assert!(msg.contains("yah-marketing"), "got: {msg}");
11049 }
11050
11051 #[test]
11052 fn cross_ref_bails_on_missing_component() {
11053 let tmp = tempfile::TempDir::new().unwrap();
11054 let root = tmp.path();
11055 write_marketing_service(root); // has component id "site", not "elsewhere"
11056
11057 let dom = DomainConfig {
11058 schema_version: 1,
11059 name: "yah-dev".into(),
11060 domain: "yah.dev".into(),
11061 front_door: FrontDoor::Worker,
11062 cdn_bucket: "yah-dev".into(),
11063 worker_bundle_path: None,
11064 routes: vec![DomainRoute {
11065 headers: Default::default(),
11066 path: "/".into(),
11067 mode: RouteMode::Static {
11068 component: "yah-marketing/elsewhere".into(),
11069 },
11070 }],
11071 };
11072 dom.save(root).unwrap();
11073
11074 let err = CloudConfig::load(root).unwrap_err();
11075 let msg = format!("{err:#}");
11076 assert!(msg.contains("no component with id"), "got: {msg}");
11077 assert!(msg.contains("elsewhere"), "got: {msg}");
11078 }
11079
11080 #[test]
11081 fn cross_ref_bails_on_malformed_ref() {
11082 let tmp = tempfile::TempDir::new().unwrap();
11083 let root = tmp.path();
11084 write_marketing_service(root);
11085
11086 let dom = DomainConfig {
11087 schema_version: 1,
11088 name: "yah-dev".into(),
11089 domain: "yah.dev".into(),
11090 front_door: FrontDoor::Worker,
11091 cdn_bucket: "yah-dev".into(),
11092 worker_bundle_path: None,
11093 routes: vec![DomainRoute {
11094 headers: Default::default(),
11095 path: "/".into(),
11096 mode: RouteMode::Static {
11097 component: "no-slash-here".into(),
11098 },
11099 }],
11100 };
11101 dom.save(root).unwrap();
11102
11103 let err = CloudConfig::load(root).unwrap_err();
11104 let msg = format!("{err:#}");
11105 assert!(msg.contains("expected"), "got: {msg}");
11106 }
11107
11108 #[test]
11109 fn redirect_routes_skip_component_validation() {
11110 let tmp = tempfile::TempDir::new().unwrap();
11111 let root = tmp.path();
11112 // No services at all — redirect must still load cleanly because it
11113 // references nothing.
11114 let dom = DomainConfig {
11115 schema_version: 1,
11116 name: "yah-dev".into(),
11117 domain: "yah.dev".into(),
11118 front_door: FrontDoor::Worker,
11119 cdn_bucket: "yah-dev".into(),
11120 worker_bundle_path: None,
11121 routes: vec![DomainRoute {
11122 headers: Default::default(),
11123 path: "/old".into(),
11124 mode: RouteMode::Redirect {
11125 target: "https://yah.dev/blog".into(),
11126 status: 308,
11127 },
11128 }],
11129 };
11130 dom.save(root).unwrap();
11131
11132 let cfg = CloudConfig::load(root).unwrap();
11133 assert!(cfg.domain("yah-dev").is_some());
11134 }
11135
11136 #[test]
11137 fn name_must_match_file_stem() {
11138 let tmp = tempfile::TempDir::new().unwrap();
11139 let root = tmp.path();
11140 // Hand-write a file whose stem disagrees with its `name`.
11141 let dir = root.join(".yah").join("domains");
11142 std::fs::create_dir_all(&dir).unwrap();
11143 std::fs::write(
11144 dir.join("yah-dev.toml"),
11145 r#"schema_version = 1
11146name = "different-name"
11147domain = "yah.dev"
11148front_door = "bucket-direct"
11149cdn_bucket = "yah-dev"
11150"#,
11151 )
11152 .unwrap();
11153
11154 let err = CloudConfig::load(root).unwrap_err();
11155 let msg = format!("{err:#}");
11156 assert!(msg.contains("must match the file stem"), "got: {msg}");
11157 }
11158
11159 #[test]
11160 fn net_alias_tier_subdomain_manifest_loads_and_cross_refs() {
11161 // R561-F2: a per-tenant subdomain manifest on the net.yah.dev wildcard
11162 // alias tier is just a DomainConfig whose `domain` is `<name>.net.yah.dev`
11163 // and whose static route cross-refs the tenant's service component.
11164 // This is exactly the shape .yah/domains/scrabcake-net-yah-dev.toml ships.
11165 let tmp = tempfile::TempDir::new().unwrap();
11166 let root = tmp.path();
11167 write_marketing_service(root); // service "yah-marketing", component "site"
11168
11169 let dom = DomainConfig {
11170 schema_version: 1,
11171 name: "tenant-net-yah-dev".into(),
11172 domain: "tenant.net.yah.dev".into(),
11173 front_door: FrontDoor::Worker,
11174 cdn_bucket: "net-yah-dev".into(), // shared per-tier bucket
11175 worker_bundle_path: None,
11176 routes: vec![DomainRoute {
11177 headers: Default::default(),
11178 path: "/*".into(),
11179 mode: RouteMode::Static {
11180 component: "yah-marketing/site".into(),
11181 },
11182 }],
11183 };
11184 dom.save(root).unwrap();
11185
11186 let cfg = CloudConfig::load(root).unwrap();
11187 let dom = cfg
11188 .domain("tenant-net-yah-dev")
11189 .expect("net-tier subdomain manifest should load");
11190 assert_eq!(dom.domain, "tenant.net.yah.dev");
11191 assert_eq!(dom.cdn_bucket, "net-yah-dev");
11192 }
11193
11194 // ─── R572-F3: NodeAllocatable + taints ──────────────────────────────────
11195
11196 #[test]
11197 fn machine_allocatable_round_trips() {
11198 let toml_src = r#"
11199name = "us-west-001"
11200provider = "static"
11201mesh_tags = ["tag:cloud-runner"]
11202[allocatable]
11203memory_mb = 3800
11204cpu_millis = 2000
11205"#;
11206 let m: MachineConfig = toml::from_str(toml_src).unwrap();
11207 let a = m.allocatable.as_ref().expect("allocatable should parse");
11208 assert_eq!(a.memory_mb, 3800);
11209 assert_eq!(a.cpu_millis, 2000);
11210
11211 let s = toml::to_string(&m).unwrap();
11212 let back: MachineConfig = toml::from_str(&s).unwrap();
11213 let a2 = back.allocatable.as_ref().unwrap();
11214 assert_eq!(a2.memory_mb, 3800);
11215 assert_eq!(a2.cpu_millis, 2000);
11216 }
11217
11218 #[test]
11219 fn machine_taints_round_trips() {
11220 let toml_src = r#"
11221name = "us-south-001"
11222provider = "static"
11223mesh_tags = ["tag:cloud-runner"]
11224taints = ["no-appliance"]
11225"#;
11226 let m: MachineConfig = toml::from_str(toml_src).unwrap();
11227 assert_eq!(m.taints, vec!["no-appliance"]);
11228
11229 let s = toml::to_string(&m).unwrap();
11230 let back: MachineConfig = toml::from_str(&s).unwrap();
11231 assert_eq!(back.taints, vec!["no-appliance"]);
11232 }
11233
11234 #[test]
11235 fn machine_allocatable_absent_is_none() {
11236 let toml_src = "name = \"node\"\nprovider = \"static\"\nmesh_tags = []\n";
11237 let m: MachineConfig = toml::from_str(toml_src).unwrap();
11238 assert!(m.allocatable.is_none());
11239 assert!(m.taints.is_empty());
11240 }
11241
11242 #[test]
11243 fn machine_allocatable_skipped_when_none() {
11244 let m = make_machine("node", vec![]);
11245 let s = toml::to_string(&m).unwrap();
11246 assert!(
11247 !s.contains("allocatable"),
11248 "None allocatable must be omitted: {s}"
11249 );
11250 assert!(!s.contains("taints"), "empty taints must be omitted: {s}");
11251 }
11252
11253 #[test]
11254 fn machine_multiple_taints_round_trip() {
11255 let toml_src = r#"
11256name = "quarantined"
11257provider = "static"
11258mesh_tags = ["tag:build-worker"]
11259taints = ["no-server", "no-appliance", "no-job"]
11260"#;
11261 let m: MachineConfig = toml::from_str(toml_src).unwrap();
11262 assert_eq!(m.taints.len(), 3);
11263 assert!(m.taints.contains(&"no-server".to_string()));
11264 assert!(m.taints.contains(&"no-appliance".to_string()));
11265 assert!(m.taints.contains(&"no-job".to_string()));
11266 // R742-T4: every key here is one the scheduler reads. This fixture
11267 // used to carry `no-voter`, which none of them is.
11268 assert!(m.inert_taints().is_empty());
11269 }
11270
11271 // ─── R742-T4 (W305): inert-taint classification ─────────────────────────
11272
11273 #[test]
11274 fn every_archetype_repel_key_is_live() {
11275 for arch in LifecycleArchetype::ALL {
11276 let key = format!("no-{}", arch.taint_key());
11277 assert_eq!(
11278 taint_effect(&key),
11279 TaintEffect::Repels(arch),
11280 "{key} must repel {arch:?}"
11281 );
11282 }
11283 }
11284
11285 #[test]
11286 fn public_ip_is_an_affinity_key_not_an_inert_one() {
11287 assert_eq!(
11288 taint_effect(workload_spec::PUBLIC_IP_TAINT),
11289 TaintEffect::Attracts
11290 );
11291 }
11292
11293 #[test]
11294 fn a_free_form_taint_is_inert_and_says_so() {
11295 // W305's headline example: `taints = ["qa"]` parsed clean and did
11296 // nothing. Environment is not expressible as a taint.
11297 assert_eq!(taint_effect("qa"), TaintEffect::Inert);
11298 // And the one that actually cost fleet state: `no-voter` reads as an
11299 // exclusion and excludes nothing — "voter" is not an archetype.
11300 assert_eq!(taint_effect("no-voter"), TaintEffect::Inert);
11301 // A near-miss on a real key is inert too, not silently forgiven.
11302 assert_eq!(taint_effect("no-servers"), TaintEffect::Inert);
11303
11304 let m = make_machine_with_capacity(
11305 "dev-pi",
11306 8192,
11307 4000,
11308 vec!["no-appliance", "no-voter", "qa"],
11309 );
11310 assert_eq!(m.inert_taints(), vec!["no-voter", "qa"]);
11311 }
11312
11313 // ─── R876-B7: repel-by-default + declarable toleration ──────────────────
11314
11315 /// The headline inversion. A bare `RequiredSpec` — which is exactly what
11316 /// deserializing a mirror's `required = { regions, mesh_tags }` produces,
11317 /// since no TOML in the tree writes `tolerates` — is now repelled by a
11318 /// repelling taint. Before B7 it matched, because repulsion was conditional
11319 /// on a `#[serde(skip)]` field that this path could never fill.
11320 #[test]
11321 fn an_undeclared_spec_is_repelled_by_a_repelling_taint() {
11322 let tainted = make_machine_with_capacity("n", 8192, 4000, vec!["no-server"]);
11323 assert!(!RequiredSpec::default().matches(&tainted));
11324
11325 // And it is the DESERIALIZED shape that matters, not a hand-built one:
11326 // this is the mirror path reproduced exactly.
11327 let from_toml: RequiredSpec =
11328 toml::from_str("regions = [\"us-east\"]\n").expect("a mirror-shaped required parses");
11329 assert!(from_toml.tolerates.is_empty());
11330 let mut in_region = tainted.clone();
11331 in_region.region = Some("us-east".to_string());
11332 assert!(
11333 !from_toml.matches(&in_region),
11334 "a mirror-declared placement must now read machine.taints"
11335 );
11336 }
11337
11338 /// The opt-back-in half, and the one an operator writes by hand.
11339 #[test]
11340 fn an_explicit_toleration_admits_the_tainted_machine_again() {
11341 let tainted = make_machine_with_capacity("n", 8192, 4000, vec!["no-server"]);
11342 let spec = RequiredSpec {
11343 tolerates: vec!["no-server".to_string()],
11344 ..Default::default()
11345 };
11346 assert!(spec.matches(&tainted));
11347
11348 // Per-key, not a blanket pass: tolerating one repelling key says nothing
11349 // about another.
11350 let both = make_machine_with_capacity("n", 8192, 4000, vec!["no-server", "no-appliance"]);
11351 assert!(!spec.matches(&both));
11352
11353 // And it deserializes — the whole point of replacing a `#[serde(skip)]`
11354 // field is that a mirror can now declare this.
11355 let from_toml: RequiredSpec = toml::from_str("tolerates = [\"no-server\"]\n")
11356 .expect("a slot can declare a toleration");
11357 assert!(from_toml.matches(&tainted));
11358 }
11359
11360 /// An untainted machine is unaffected, which is what makes the migration
11361 /// bounded: six of the nine fleet machines carry no repelling taint at all.
11362 #[test]
11363 fn an_untainted_machine_matches_exactly_as_before() {
11364 let clean = make_machine_with_capacity("n", 8192, 4000, vec![]);
11365 assert!(RequiredSpec::default().matches(&clean));
11366 assert!(RequiredSpec {
11367 tolerates: vec!["no-server".to_string()],
11368 ..Default::default()
11369 }
11370 .matches(&clean));
11371 }
11372
11373 /// THE MIGRATION'S LOAD-BEARING FACT. `public-ip` is on three fleet nodes
11374 /// including us-east-001, the only origin serving the yah.dev apex. It is an
11375 /// *affinity* key, so repel-by-default must not touch it — reading every
11376 /// taint as repulsion would evict the apex on the next apply.
11377 #[test]
11378 fn an_affinity_taint_does_not_repel() {
11379 let public = make_machine_with_capacity("us-east-001", 8192, 4000, vec!["public-ip"]);
11380 assert!(
11381 RequiredSpec::default().matches(&public),
11382 "public-ip attracts; it must never be read as repulsion"
11383 );
11384 }
11385
11386 /// `select_matching` filters on the same predicate, so a tainted machine
11387 /// leaves the candidate set rather than being silently placed onto.
11388 #[test]
11389 fn select_matching_drops_a_tainted_candidate_and_keeps_the_rest() {
11390 let drained = make_machine_with_capacity("drained", 8192, 4000, vec!["no-server"]);
11391 let healthy = make_machine_with_capacity("healthy", 8192, 4000, vec![]);
11392 let pool = [&drained, &healthy];
11393
11394 let picked = select_matching(&pool, &RequiredSpec::default(), 1, "test pool", "empty")
11395 .expect("one candidate remains");
11396 assert_eq!(
11397 picked.iter().map(|m| m.name.as_str()).collect::<Vec<_>>(),
11398 vec!["healthy"],
11399 "tainting the first candidate moves the placement to the second"
11400 );
11401
11402 // Asking for both is a shortfall, not a half-placement.
11403 let err = select_matching(&pool, &RequiredSpec::default(), 2, "test pool", "empty")
11404 .expect_err("only one of two matches");
11405 assert!(format!("{err:#}").contains("only 1 of 2 machines match"));
11406 }
11407
11408 /// The `admit_workload` path must be behaviourally unchanged: its spec is
11409 /// built by `admission_spec`, which now emits the complementary tolerations.
11410 #[test]
11411 fn admission_preserves_archetype_scoped_repulsion_across_the_inversion() {
11412 let ws = minimal_spec("srv", 1); // a Server
11413 let req = admission_spec(&ws, &[]).unwrap();
11414 assert_eq!(ws.effective_archetype(), LifecycleArchetype::Server);
11415
11416 let no_server = make_machine_with_capacity("n", 8192, 4000, vec!["no-server"]);
11417 let no_appliance = make_machine_with_capacity("n", 8192, 4000, vec!["no-appliance"]);
11418 assert!(!req.matches(&no_server), "its own class still repels it");
11419 assert!(
11420 req.matches(&no_appliance),
11421 "another class's taint still does not — this is the pre-B7 answer"
11422 );
11423 }
11424
11425 #[test]
11426 fn describe_names_the_toleration_so_a_refusal_is_readable() {
11427 let spec = RequiredSpec {
11428 regions: vec!["us-west".to_string()],
11429 tolerates: vec!["no-appliance".to_string()],
11430 ..Default::default()
11431 };
11432 assert_eq!(
11433 spec.describe(),
11434 "required.regions=[us-west] + required.tolerates=[no-appliance]"
11435 );
11436 }
11437
11438 /// A toleration widens; it must not make an underspecified slot look
11439 /// specified, or the deploy side stops refusing one.
11440 #[test]
11441 fn a_toleration_alone_is_still_an_unconstrained_spec() {
11442 assert!(RequiredSpec {
11443 tolerates: vec!["no-server".to_string()],
11444 ..Default::default()
11445 }
11446 .is_unconstrained());
11447 }
11448
11449 #[test]
11450 fn an_inert_taint_changes_no_placement_decision() {
11451 // The reason this is a lint and not a behaviour change: the guard's
11452 // whole premise is that these keys are invisible to `matches`.
11453 let clean = make_machine_with_capacity("n", 8192, 4000, vec![]);
11454 let noisy = make_machine_with_capacity("n", 8192, 4000, vec!["no-voter", "qa"]);
11455 // R876-B7: still true under repel-by-default, and for a sharper reason —
11456 // `matches` now walks `machine.taints` itself, so an unclassifiable key
11457 // is skipped by `taint_effect` rather than merely never looked up.
11458 for arch in LifecycleArchetype::ALL {
11459 let req = RequiredSpec {
11460 tolerates: tolerations_excluding(&[arch]),
11461 ..Default::default()
11462 };
11463 assert_eq!(req.matches(&clean), req.matches(&noisy));
11464 assert!(req.matches(&noisy), "neither key repels");
11465 }
11466 }
11467
11468 /// The toleration set [`admission_spec`] derives for a group of `archetypes`
11469 /// — every repelling key that is not the group's own class.
11470 fn tolerations_excluding(archetypes: &[LifecycleArchetype]) -> Vec<String> {
11471 LifecycleArchetype::ALL
11472 .into_iter()
11473 .filter(|a| !archetypes.contains(a))
11474 .map(|a| format!("no-{}", a.taint_key()))
11475 .collect()
11476 }
11477
11478 #[test]
11479 fn live_taint_keys_lists_the_whole_legal_vocabulary() {
11480 assert_eq!(
11481 live_taint_keys(),
11482 vec!["no-appliance", "no-job", "no-server", "public-ip"]
11483 );
11484 }
11485
11486 // ─── R742-F1 (W305): sovereign groups ───────────────────────────────────
11487
11488 /// A machine in `group`, with the role left unwritten — which is the state
11489 /// of every machine TOML that predates R605-F12 and resolves to `voter`.
11490 fn in_group(name: &str, group: Option<&str>) -> MachineConfig {
11491 MachineConfig {
11492 sovereign_group: group.map(String::from),
11493 ..make_machine(name, vec![])
11494 }
11495 }
11496
11497 /// A machine in `group` with its quorum eligibility stated (R605-F12).
11498 fn in_group_as(name: &str, group: &str, role: SovereignRole) -> MachineConfig {
11499 MachineConfig {
11500 sovereign_group: Some(group.to_string()),
11501 sovereign_role: Some(role),
11502 ..make_machine(name, vec![])
11503 }
11504 }
11505
11506 #[test]
11507 fn a_join_within_one_sovereign_group_is_permitted() {
11508 assert_eq!(
11509 judge_join(
11510 &in_group("us-west-013", Some("dev")),
11511 &in_group("us-west-011", Some("dev")),
11512 ),
11513 JoinVerdict::Permit
11514 );
11515 }
11516
11517 /// The case the field exists for: before it, the only thing standing
11518 /// between a dev Pi and the prod quorum was a comment in a TOML.
11519 #[test]
11520 fn a_cross_group_join_is_refused_naming_both_groups() {
11521 let verdict = judge_join(
11522 &in_group("us-west-011", Some("dev")),
11523 &in_group("us-west-001", Some("prod")),
11524 );
11525 let JoinVerdict::Refuse(msg) = verdict else {
11526 panic!("a dev node joining prod must be refused: {verdict:?}");
11527 };
11528 // A refusal that does not name what it saw is one the operator has to
11529 // go and reconstruct, so it gets worked around instead of fixed.
11530 assert!(msg.contains("us-west-011") && msg.contains("us-west-001"), "{msg}");
11531 assert!(msg.contains("dev") && msg.contains("prod"), "{msg}");
11532 }
11533
11534 /// `None` is a declaration ("standalone, in no group"), not a gap — so
11535 /// growing prod with an unstamped box is a cross-group join too, and the
11536 /// refusal has to say which file makes it legal.
11537 #[test]
11538 fn an_undeclared_node_cannot_join_a_declared_group() {
11539 let verdict = judge_join(
11540 &in_group("us-west-002", None),
11541 &in_group("us-west-001", Some("prod")),
11542 );
11543 let JoinVerdict::Refuse(msg) = verdict else {
11544 panic!("an unstamped node joining prod must be refused: {verdict:?}");
11545 };
11546 assert!(
11547 msg.contains(".yah/infra/machines/us-west-002.toml"),
11548 "the refusal must name the file to stamp: {msg}"
11549 );
11550 }
11551
11552 #[test]
11553 fn a_declared_node_cannot_join_a_standalone_target() {
11554 // us-west-003 is `mode: standalone` on purpose; it is not a group of
11555 // one waiting to be grown.
11556 let verdict = judge_join(
11557 &in_group("us-west-001", Some("prod")),
11558 &in_group("us-west-003", None),
11559 );
11560 assert!(matches!(verdict, JoinVerdict::Refuse(msg) if msg.contains("us-west-003")));
11561 }
11562
11563 #[test]
11564 fn two_undeclared_nodes_cannot_form_an_undeclared_group() {
11565 let verdict = judge_join(
11566 &in_group("us-west-002", None),
11567 &in_group("us-west-015", None),
11568 );
11569 assert!(
11570 matches!(&verdict, JoinVerdict::Refuse(msg) if msg.contains("us-west-002")
11571 && msg.contains("us-west-015")),
11572 "forming a group nobody declared must be refused, naming both: {verdict:?}"
11573 );
11574 }
11575
11576 // ─── R605-F12: the voting axis ──────────────────────────────────────────
11577
11578 /// The whole ticket in one assertion. us-west-003 is a member of prod —
11579 /// same secrets, same upgrade cadence, same destruction — and must never
11580 /// hold a prod raft seat. Before the role axis, the only thing refusing it
11581 /// was its *absent* group stamp, so writing down the truth above would have
11582 /// removed the guard.
11583 #[test]
11584 fn a_non_voting_member_is_refused_into_its_own_group() {
11585 let verdict = judge_join(
11586 &in_group_as("us-west-003", "prod", SovereignRole::NonVoter),
11587 &in_group_as("us-west-001", "prod", SovereignRole::Voter),
11588 );
11589 let JoinVerdict::Refuse(msg) = verdict else {
11590 panic!("a non-voting prod member must not join the prod quorum: {verdict:?}");
11591 };
11592 assert!(msg.contains("us-west-003") && msg.contains("NON-VOTING"), "{msg}");
11593 // The refusal must not blame the group: both sides say "prod", and a
11594 // cross-group message here would read as a bug in the check itself.
11595 assert!(!msg.contains("cross-group"), "{msg}");
11596 assert!(
11597 msg.contains(".yah/infra/machines/us-west-003.toml"),
11598 "the refusal must name the file that decides it: {msg}"
11599 );
11600 }
11601
11602 /// Read from the other end: a box declared non-voting has no quorum seat to
11603 /// be grown, so it cannot be a join target either.
11604 #[test]
11605 fn a_non_voting_target_has_no_quorum_to_grow() {
11606 let verdict = judge_join(
11607 &in_group_as("us-west-001", "prod", SovereignRole::Voter),
11608 &in_group_as("us-west-003", "prod", SovereignRole::NonVoter),
11609 );
11610 assert!(
11611 matches!(&verdict, JoinVerdict::Refuse(msg) if msg.contains("the target")
11612 && msg.contains("us-west-003")),
11613 "{verdict:?}"
11614 );
11615 }
11616
11617 /// A non-voter joining a *standalone* target is refused for two reasons at
11618 /// once, and the message must pick the one whose fix would actually work.
11619 /// Naming the role here would send the operator to flip `sovereign_role`
11620 /// and come back to the same refusal.
11621 #[test]
11622 fn a_refusal_names_the_group_when_fixing_the_role_would_not_help() {
11623 let verdict = judge_join(
11624 &in_group_as("us-west-003", "prod", SovereignRole::NonVoter),
11625 &in_group("us-west-002", None),
11626 );
11627 let JoinVerdict::Refuse(msg) = verdict else {
11628 panic!("a standalone target has no group to join: {verdict:?}");
11629 };
11630 assert!(
11631 msg.contains(".yah/infra/machines/us-west-002.toml"),
11632 "the refusal must point at the target's missing group stamp: {msg}"
11633 );
11634 assert!(!msg.contains("NON-VOTING"), "{msg}");
11635 }
11636
11637 /// The back-compat seam, pinned: the six nodes stamped before R605-F12
11638 /// write no role, and an absent role means what declaring a group has
11639 /// always meant. If this flips, the live prod and dev quorums stop being
11640 /// growable on a config the operator never edited.
11641 #[test]
11642 fn an_unwritten_role_still_joins_its_group() {
11643 let joiner = in_group("us-west-013", Some("dev"));
11644 assert_eq!(joiner.sovereign_role, None);
11645 assert_eq!(
11646 judge_join(&joiner, &in_group("us-west-011", Some("dev"))),
11647 JoinVerdict::Permit
11648 );
11649 assert_eq!(
11650 judge_join(
11651 &joiner,
11652 &in_group_as("us-west-011", "dev", SovereignRole::Voter)
11653 ),
11654 JoinVerdict::Permit
11655 );
11656 }
11657
11658 /// A non-voting member is still a *member*, and the two claims must not be
11659 /// conflated: `sovereign_membership()` reports the group either way, so a
11660 /// consumer asking "is this box in prod's blast radius" gets yes.
11661 #[test]
11662 fn a_non_voter_is_still_in_the_group_it_names() {
11663 let m = in_group_as("us-west-003", "prod", SovereignRole::NonVoter);
11664 assert_eq!(m.sovereign_membership().group, Some("prod"));
11665 assert!(!m.sovereign_membership().role.is_voter());
11666
11667 // …and the group-membership query the fleet reads is unaffected by it.
11668 let cfg = make_empty_cfg(vec![
11669 m,
11670 in_group_as("us-west-001", "prod", SovereignRole::Voter),
11671 ]);
11672 assert_eq!(
11673 cfg.machines_in_group("prod")
11674 .iter()
11675 .map(|m| m.name.as_str())
11676 .collect::<Vec<_>>(),
11677 vec!["us-west-003", "us-west-001"]
11678 );
11679 }
11680
11681 /// The role travels through TOML in one spelling, and an absent one stays
11682 /// absent on the way back out — otherwise every machine file would grow a
11683 /// `sovereign_role = "voter"` line the operator never wrote, and the
11684 /// unroled-member lint would have nothing left to find.
11685 #[test]
11686 fn sovereign_role_round_trips_and_is_omitted_when_unwritten() {
11687 let m: MachineConfig = toml::from_str(
11688 r#"
11689name = "us-west-003"
11690provider = "static"
11691region = "us-west"
11692arch = "x86_64"
11693mesh_tags = []
11694sovereign_group = "prod"
11695sovereign_role = "non-voter"
11696"#,
11697 )
11698 .unwrap();
11699 assert_eq!(m.sovereign_role, Some(SovereignRole::NonVoter));
11700 assert!(toml::to_string(&m)
11701 .unwrap()
11702 .contains(r#"sovereign_role = "non-voter""#));
11703
11704 let unwritten = MachineConfig {
11705 sovereign_role: None,
11706 ..m
11707 };
11708 assert!(!toml::to_string(&unwritten)
11709 .unwrap()
11710 .contains("sovereign_role"));
11711 }
11712
11713 /// The invariant the ticket is most explicit about: a sovereign group is a
11714 /// blast radius, not a filter. If this ever fails, `matches` has grown an
11715 /// axis it must not have and dev-mode workloads have silently become
11716 /// unschedulable on the dev group.
11717 #[test]
11718 fn sovereign_group_is_not_a_placement_input() {
11719 let standalone = in_group("n", None);
11720 let grouped = in_group("n", Some("dev"));
11721 let other = in_group("n", Some("prod"));
11722
11723 for spec in [
11724 RequiredSpec::default(),
11725 RequiredSpec {
11726 regions: vec!["us-west".into()],
11727 ..Default::default()
11728 },
11729 RequiredSpec {
11730 tolerates: tolerations_excluding(&[LifecycleArchetype::Appliance]),
11731 ..Default::default()
11732 },
11733 ] {
11734 let baseline = spec.matches(&standalone);
11735 assert_eq!(spec.matches(&grouped), baseline);
11736 assert_eq!(spec.matches(&other), baseline);
11737 }
11738 }
11739
11740 // ─── R742-F3 (W305): group → machine set, and group-scoped admission ────
11741
11742 /// The primitive `migrate --to` needs and `rollout plan` still lacks
11743 /// (W314 gap 1): a group exists only as the set of machines naming it, so
11744 /// membership has to be derived rather than declared anywhere.
11745 #[test]
11746 fn machines_in_group_derives_membership_from_the_declarations() {
11747 let cfg = make_empty_cfg(vec![
11748 in_group("us-west-001", Some("prod")),
11749 in_group("us-west-011", Some("dev")),
11750 in_group("us-west-013", Some("dev")),
11751 in_group("us-west-002", None),
11752 ]);
11753
11754 let dev: Vec<&str> = cfg
11755 .machines_in_group("dev")
11756 .iter()
11757 .map(|m| m.name.as_str())
11758 .collect();
11759 assert_eq!(dev, vec!["us-west-011", "us-west-013"]);
11760 assert_eq!(cfg.machines_in_group("prod").len(), 1);
11761
11762 // Standalone is "in no group", not "in a group called none" — so an
11763 // unstamped box is never swept into a migration target.
11764 assert!(cfg.machines_in_group("").is_empty());
11765 assert!(cfg.machines_in_group("staging").is_empty());
11766 }
11767
11768 #[test]
11769 fn declared_sovereign_groups_is_the_vocabulary_a_bad_target_is_named_against() {
11770 let cfg = make_empty_cfg(vec![
11771 in_group("a", Some("prod")),
11772 in_group("b", Some("dev")),
11773 in_group("c", Some("prod")),
11774 in_group("d", None),
11775 ]);
11776 // Sorted + deduped, and standalone contributes nothing.
11777 assert_eq!(cfg.declared_sovereign_groups(), vec!["dev", "prod"]);
11778 assert!(make_empty_cfg(vec![in_group("a", None)])
11779 .declared_sovereign_groups()
11780 .is_empty());
11781 }
11782
11783 /// Group-scoped admission must be the SAME predicate as unscoped
11784 /// admission, only over fewer candidates. If it ever diverges, `migrate`
11785 /// becomes a way to place a workload somewhere `yah cloud apply` would
11786 /// refuse — which is exactly the silent routing-around W305 exists to stop.
11787 #[test]
11788 fn admit_workload_in_group_narrows_candidates_without_changing_the_predicate() {
11789 let mut prod = in_group("us-west-001", Some("prod"));
11790 prod.mesh_tags = vec!["tag:cloud-runner".into()];
11791 let mut dev_repels = in_group("us-west-011", Some("dev"));
11792 dev_repels.taints = vec!["no-appliance".into()];
11793 let mut dev_ok = in_group("us-west-013", Some("dev"));
11794 dev_ok.mesh_tags = vec!["tag:cloud-runner".into()];
11795
11796 let cfg = make_empty_cfg(vec![prod, dev_repels, dev_ok]);
11797
11798 let mut ws = ws_with_selector(None);
11799 ws.archetype = Some(LifecycleArchetype::Appliance);
11800
11801 // Unscoped picks the first match in declaration order.
11802 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "us-west-001");
11803 // Scoped skips the repelling dev node and lands on the other one —
11804 // the taint is honoured, not bypassed.
11805 assert_eq!(
11806 cfg.admit_workload_in_group(&ws, "dev").unwrap().name,
11807 "us-west-013"
11808 );
11809 }
11810
11811 #[test]
11812 fn admit_workload_in_group_distinguishes_an_empty_group_from_a_repelling_one() {
11813 let mut dev = in_group("us-west-011", Some("dev"));
11814 dev.taints = vec!["no-appliance".into()];
11815 let cfg = make_empty_cfg(vec![in_group("us-west-001", Some("prod")), dev]);
11816
11817 let mut ws = ws_with_selector(None);
11818 ws.archetype = Some(LifecycleArchetype::Appliance);
11819
11820 // A group nobody declares names the legal vocabulary, because a typo
11821 // is the realistic cause and "no candidates" would send the operator
11822 // hunting for a placement problem that does not exist.
11823 let missing = cfg.admit_workload_in_group(&ws, "stagng").unwrap_err().to_string();
11824 assert!(missing.contains("no machine declares"), "{missing}");
11825 assert!(missing.contains("dev") && missing.contains("prod"), "{missing}");
11826
11827 // A group that exists but refuses names the machines it tried.
11828 let repelled = cfg.admit_workload_in_group(&ws, "dev").unwrap_err().to_string();
11829 assert!(repelled.contains("us-west-011"), "{repelled}");
11830 }
11831
11832 #[test]
11833 fn sovereign_group_round_trips_and_is_omitted_when_standalone() {
11834 let src = r#"
11835name = "us-west-011"
11836provider = "static"
11837mesh_tags = []
11838sovereign_group = "dev"
11839"#;
11840 let m: MachineConfig = toml::from_str(src).unwrap();
11841 assert_eq!(m.sovereign_group.as_deref(), Some("dev"));
11842 assert!(toml::to_string(&m).unwrap().contains("sovereign_group"));
11843
11844 // A machine that predates the field parses as standalone and does not
11845 // grow the key back on write.
11846 let legacy: MachineConfig =
11847 toml::from_str("name = \"us-west-002\"\nprovider = \"static\"\nmesh_tags = []\n")
11848 .unwrap();
11849 assert_eq!(legacy.sovereign_group, None);
11850 assert!(!toml::to_string(&legacy).unwrap().contains("sovereign_group"));
11851 }
11852
11853 // ─── R572-F5: capacity floor + absolute (untolerable) taints ────────────
11854
11855 fn make_machine_with_capacity(
11856 name: &str,
11857 memory_mb: u32,
11858 cpu_millis: u32,
11859 taints: Vec<&str>,
11860 ) -> MachineConfig {
11861 MachineConfig {
11862 allocatable: Some(NodeAllocatable {
11863 memory_mb,
11864 cpu_millis,
11865 }),
11866 taints: taints.into_iter().map(String::from).collect(),
11867 ..make_machine(name, vec![])
11868 }
11869 }
11870
11871 fn server_spec(memory_mb: u32, cpu_millis: u32) -> WorkloadSpec {
11872 use workload_spec::{ImageRef, LifecycleArchetype, TierTag};
11873 let mut ws = WorkloadSpec::for_forge(
11874 "f5-test",
11875 ImageRef {
11876 registry: "localhost".into(),
11877 repository: "test".into(),
11878 tag: "latest".into(),
11879 digest: workload_spec::testing::test_digest(),
11880 },
11881 TierTag("infra".into()),
11882 vec![],
11883 );
11884 ws.archetype = Some(LifecycleArchetype::Server);
11885 ws.resources.memory_mb = memory_mb;
11886 ws.resources.cpu_millis = cpu_millis;
11887 // These are SERVER specs that borrow `for_forge` as a constructor
11888 // shortcut, so drop the forge memory request it stamps on — otherwise
11889 // every spec here silently requests the forge default instead of the
11890 // `memory_mb` the caller passed, and the capacity-floor tests below
11891 // stop testing their own argument. A server workload declares no
11892 // request, which is the documented fall-back-to-`resources.memory_mb`
11893 // path (`WorkloadSpec::memory_request_mb`).
11894 ws.resources.memory_request_mb = None;
11895 ws
11896 }
11897
11898 fn appliance_spec_ws(memory_mb: u32, cpu_millis: u32) -> WorkloadSpec {
11899 use workload_spec::LifecycleArchetype;
11900 let mut ws = server_spec(memory_mb, cpu_millis);
11901 ws.archetype = Some(LifecycleArchetype::Appliance);
11902 ws
11903 }
11904
11905 #[test]
11906 fn capacity_floor_rejects_undersized_node() {
11907 let cfg = make_empty_cfg(vec![make_machine_with_capacity("small", 256, 500, vec![])]);
11908 let ws = server_spec(512, 1000); // demands more than available
11909 assert!(cfg.admit_workload(&ws).is_err());
11910 }
11911
11912 #[test]
11913 fn capacity_floor_accepts_exact_fit() {
11914 let cfg = make_empty_cfg(vec![make_machine_with_capacity("exact", 512, 1000, vec![])]);
11915 let ws = server_spec(512, 1000);
11916 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "exact");
11917 }
11918
11919 #[test]
11920 fn capacity_floor_passes_when_allocatable_absent() {
11921 // A machine with no allocatable block skips the capacity check (no data).
11922 let cfg = make_empty_cfg(vec![make_machine("no-alloc", vec![])]);
11923 let ws = server_spec(99999, 99999); // would exceed any real node
11924 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "no-alloc");
11925 }
11926
11927 #[test]
11928 fn taint_repulsion_blocks_appliance_on_no_appliance_node() {
11929 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
11930 "south",
11931 1024,
11932 2000,
11933 vec!["no-appliance"],
11934 )]);
11935 let ws = appliance_spec_ws(256, 500);
11936 assert!(
11937 cfg.admit_workload(&ws).is_err(),
11938 "appliance must be repelled by no-appliance taint"
11939 );
11940 }
11941
11942 #[test]
11943 fn taint_repulsion_allows_server_on_no_appliance_node() {
11944 // "no-appliance" only repels Appliance workloads; servers are unaffected.
11945 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
11946 "south",
11947 1024,
11948 2000,
11949 vec!["no-appliance"],
11950 )]);
11951 let ws = server_spec(256, 500);
11952 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "south");
11953 }
11954
11955 #[test]
11956 fn taint_repulsion_job_not_blocked_by_no_server() {
11957 use workload_spec::LifecycleArchetype;
11958 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
11959 "build-box",
11960 8192,
11961 4000,
11962 vec!["no-server", "no-appliance"],
11963 )]);
11964 let mut ws = server_spec(256, 500);
11965 ws.archetype = Some(LifecycleArchetype::Job);
11966 // Job only repelled by "no-job"; "no-server" and "no-appliance" don't affect it.
11967 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "build-box");
11968 }
11969
11970 #[test]
11971 fn requires_taint_affinity_blocks_placement_without_it() {
11972 use workload_spec::{LifecycleArchetype, PUBLIC_IP_TAINT, REQUIRES_TAINT_ANNOTATION};
11973 // Simulate the passway ingress appliance: requires "public-ip" taint.
11974 let mut ws = appliance_spec_ws(256, 512);
11975 ws.archetype = Some(LifecycleArchetype::Appliance);
11976 ws.annotations
11977 .insert(REQUIRES_TAINT_ANNOTATION.into(), PUBLIC_IP_TAINT.into());
11978
11979 // Node without the taint: rejected.
11980 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
11981 "no-pip",
11982 2048,
11983 2000,
11984 vec![],
11985 )]);
11986 assert!(cfg.admit_workload(&ws).is_err());
11987
11988 // Node with the taint: accepted.
11989 let cfg = make_empty_cfg(vec![make_machine_with_capacity(
11990 "pub-node",
11991 2048,
11992 2000,
11993 vec!["public-ip"],
11994 )]);
11995 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "pub-node");
11996 }
11997
11998 #[test]
11999 fn w244_fleet_scenario_appliance_rejected_from_south_and_west002() {
12000 // Full W244 fleet table scenario:
12001 // us-west-001/east-001: no taints, large capacity → appliance lands here
12002 // us-south-001: no-appliance taint → appliance rejected
12003 // us-west-002: no-server, no-appliance → appliance rejected
12004 let cfg = make_empty_cfg(vec![
12005 make_machine_with_capacity("us-south-001", 512, 1000, vec!["no-appliance"]),
12006 make_machine_with_capacity(
12007 "us-west-002",
12008 16384,
12009 8000,
12010 vec!["no-server", "no-appliance"],
12011 ),
12012 make_machine_with_capacity("us-west-001", 4096, 4000, vec![]),
12013 ]);
12014 let ws = appliance_spec_ws(256, 500);
12015 // Skips south (no-appliance) and west-002 (no-appliance), lands on west-001.
12016 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "us-west-001");
12017 }
12018
12019 #[test]
12020 fn w244_fleet_scenario_job_lands_on_west002_first() {
12021 use workload_spec::LifecycleArchetype;
12022 // Jobs should prefer (or at least land on) the job-only box.
12023 let cfg = make_empty_cfg(vec![
12024 make_machine_with_capacity("us-west-001", 4096, 4000, vec![]),
12025 make_machine_with_capacity(
12026 "us-west-002",
12027 16384,
12028 8000,
12029 vec!["no-server", "no-appliance"],
12030 ),
12031 ]);
12032 let mut ws = server_spec(256, 500);
12033 ws.archetype = Some(LifecycleArchetype::Job);
12034 // No fleet node declares `no-job`, so a Job is repelled by nothing;
12035 // west-001 comes first in declaration order (greedy, no preference),
12036 // which is the expected tie-break. Note this is *absence of a repel
12037 // key*, not toleration — no workload can tolerate a taint (W305).
12038 let picked = cfg.admit_workload(&ws).unwrap();
12039 // Both are eligible.
12040 assert!(
12041 picked.name == "us-west-001" || picked.name == "us-west-002",
12042 "job must land on an eligible node, got {}",
12043 picked.name
12044 );
12045 }
12046
12047 #[test]
12048 fn r569_f4_macos_node_taints_keep_cloud_critical_off_but_admit_build_jobs() {
12049 use workload_spec::LifecycleArchetype;
12050 // R569-F4: the headless M2 (us-west-015) joins the fleet as a
12051 // build-worker but must never take cloud-critical load. It carries the
12052 // same repel set as the x86 build-worker (`no-server, no-appliance` —
12053 // see .yah/infra/machines/us-west-015.toml). This pins that intent:
12054 // with a plain cloud node available beside the Mac, every
12055 // cloud-critical archetype lands on the cloud node and never the Mac;
12056 // build Jobs (the Mac's actual purpose) remain eligible on it.
12057 //
12058 // R742-T4: `no-voter` used to sit in this set and in the TOML. It was
12059 // never read here — there is no "voter" workload archetype — and
12060 // R569-F3's learner-only join is what actually keeps the box out of
12061 // quorum. It is now rejected by `yah cloud validate` as inert.
12062 let mac_taints = vec!["no-server", "no-appliance"];
12063 let fleet = || {
12064 make_empty_cfg(vec![
12065 make_machine_with_capacity("us-west-015", 24576, 8000, mac_taints.clone()),
12066 make_machine_with_capacity("us-west-001", 4096, 4000, vec![]),
12067 ])
12068 };
12069
12070 // A cloud-critical Server workload is repelled from the Mac and lands
12071 // on the untainted cloud node.
12072 let cfg = fleet();
12073 assert_eq!(
12074 cfg.admit_workload(&server_spec(256, 500)).unwrap().name,
12075 "us-west-001",
12076 "a Server workload must never land on the no-server Mac node"
12077 );
12078
12079 // Same for an Appliance (pinned/stateful cloud-critical) workload.
12080 let cfg = fleet();
12081 assert_eq!(
12082 cfg.admit_workload(&appliance_spec_ws(256, 500))
12083 .unwrap()
12084 .name,
12085 "us-west-001",
12086 "an Appliance workload must never land on the no-appliance Mac node"
12087 );
12088
12089 // Sharpest repulsion proof: with ONLY the Mac in the fleet, a
12090 // cloud-critical Server workload is rejected outright — the taint keeps
12091 // it off even when that means nowhere to run.
12092 let mac_only = make_empty_cfg(vec![make_machine_with_capacity(
12093 "us-west-015",
12094 24576,
12095 8000,
12096 mac_taints.clone(),
12097 )]);
12098 assert!(
12099 mac_only.admit_workload(&server_spec(256, 500)).is_err(),
12100 "a Server workload must be repelled from a Mac-only fleet, not admitted"
12101 );
12102
12103 // But the Mac's real job — build/forge workloads — IS admitted on it:
12104 // it tolerates every fleet taint (there is no `no-job`).
12105 let mut job = server_spec(256, 500);
12106 job.archetype = Some(LifecycleArchetype::Job);
12107 assert_eq!(
12108 mac_only.admit_workload(&job).unwrap().name,
12109 "us-west-015",
12110 "a build Job must still be admitted on the Mac build-worker"
12111 );
12112 }
12113
12114 // ─── R615-F1: linked infra sources (`.yah/infra/sources.toml`) ─────────
12115
12116 #[test]
12117 fn sources_load_is_empty_when_the_file_is_absent() {
12118 // "Every camp without linked infra has none" — which today is every
12119 // camp — must not be an error.
12120 let tmp = tempfile::TempDir::new().unwrap();
12121 let cfg = SourcesConfig::load(tmp.path()).unwrap();
12122 assert_eq!(cfg, SourcesConfig::default());
12123 assert!(cfg.source.is_empty());
12124 assert_eq!(cfg.schema_version, 1);
12125 }
12126
12127 #[test]
12128 fn sources_parses_a_path_kind_exactly_like_w274s_example() {
12129 let tmp = tempfile::TempDir::new().unwrap();
12130 std::fs::write(
12131 tmp.path().join("sources.toml"),
12132 r#"
12133schema_version = 1
12134
12135[[source]]
12136owner = "yah"
12137kind = "path"
12138path = "../yah"
12139mode = "read-only"
12140"#,
12141 )
12142 .unwrap();
12143 let cfg = SourcesConfig::load(tmp.path()).unwrap();
12144 assert_eq!(cfg.source.len(), 1);
12145 let s = &cfg.source[0];
12146 assert_eq!(s.owner, "yah");
12147 assert_eq!(s.mode, SourceMode::ReadOnly);
12148 assert!(s.select.is_empty());
12149 match &s.kind {
12150 InfraSourceKind::Path { path } => assert_eq!(path, "../yah"),
12151 other => panic!("expected Path, got {other:?}"),
12152 }
12153 }
12154
12155 #[test]
12156 fn sources_parses_a_git_kind_reusing_gitsource_verbatim() {
12157 let tmp = tempfile::TempDir::new().unwrap();
12158 std::fs::write(
12159 tmp.path().join("sources.toml"),
12160 r#"
12161schema_version = 1
12162
12163[[source]]
12164owner = "yah"
12165kind = "git"
12166repo = "git@github.com:yah-ai/infra.git"
12167ref = "main"
12168subdir = "infra"
12169select = ["tag:cloud-runner"]
12170mode = "read-only"
12171"#,
12172 )
12173 .unwrap();
12174 let cfg = SourcesConfig::load(tmp.path()).unwrap();
12175 assert_eq!(cfg.source.len(), 1);
12176 let s = &cfg.source[0];
12177 assert_eq!(s.select, vec!["tag:cloud-runner".to_string()]);
12178 match &s.kind {
12179 InfraSourceKind::Git(git) => {
12180 assert_eq!(git.repo, "git@github.com:yah-ai/infra.git");
12181 assert_eq!(git.r#ref, "main");
12182 assert_eq!(git.subdir.as_deref(), Some("infra"));
12183 }
12184 other => panic!("expected Git, got {other:?}"),
12185 }
12186 }
12187
12188 #[test]
12189 fn sources_mode_defaults_to_read_only_and_manage_is_explicit() {
12190 let tmp = tempfile::TempDir::new().unwrap();
12191 std::fs::write(
12192 tmp.path().join("sources.toml"),
12193 r#"
12194schema_version = 1
12195
12196[[source]]
12197owner = "a"
12198kind = "path"
12199path = "../a"
12200
12201[[source]]
12202owner = "b"
12203kind = "path"
12204path = "../b"
12205mode = "manage"
12206"#,
12207 )
12208 .unwrap();
12209 let cfg = SourcesConfig::load(tmp.path()).unwrap();
12210 assert_eq!(cfg.source[0].mode, SourceMode::ReadOnly, "omitted mode = read-only");
12211 assert_eq!(cfg.source[1].mode, SourceMode::Manage);
12212 }
12213
12214 #[test]
12215 fn sources_preserves_declaration_order() {
12216 // Overlay order matters (R615-F2) when two sources name the same
12217 // machine — the list must round-trip in file order, not be reordered
12218 // by owner or kind.
12219 let tmp = tempfile::TempDir::new().unwrap();
12220 std::fs::write(
12221 tmp.path().join("sources.toml"),
12222 r#"
12223schema_version = 1
12224
12225[[source]]
12226owner = "second"
12227kind = "path"
12228path = "../second"
12229
12230[[source]]
12231owner = "first"
12232kind = "path"
12233path = "../first"
12234"#,
12235 )
12236 .unwrap();
12237 let cfg = SourcesConfig::load(tmp.path()).unwrap();
12238 let owners: Vec<&str> = cfg.source.iter().map(|s| s.owner.as_str()).collect();
12239 assert_eq!(owners, vec!["second", "first"]);
12240 }
12241
12242 #[test]
12243 fn sources_round_trips_through_serialize() {
12244 let cfg = SourcesConfig {
12245 schema_version: 1,
12246 source: vec![
12247 InfraSource {
12248 owner: "yah".into(),
12249 kind: InfraSourceKind::Path {
12250 path: "../yah".into(),
12251 },
12252 mode: SourceMode::ReadOnly,
12253 select: vec![],
12254 },
12255 InfraSource {
12256 owner: "yah".into(),
12257 kind: InfraSourceKind::Git(GitSource {
12258 repo: "git@github.com:yah-ai/infra.git".into(),
12259 r#ref: "main".into(),
12260 subdir: Some("infra".into()),
12261 }),
12262 mode: SourceMode::Manage,
12263 select: vec!["tag:cloud-runner".into()],
12264 },
12265 ],
12266 };
12267 let toml_str = toml::to_string_pretty(&cfg).unwrap();
12268 let reloaded: SourcesConfig = toml::from_str(&toml_str).unwrap();
12269 assert_eq!(reloaded, cfg, "round-trip through TOML must be lossless:\n{toml_str}");
12270 }
12271
12272 // ─── R615-F2: overlay loader in CloudConfig::load ───────────────────────
12273
12274 fn write_min_machine(dir: &Path, name: &str, extra_toml: &str) {
12275 std::fs::create_dir_all(dir).unwrap();
12276 // `extra_toml` supplies `mesh_tags` when the caller cares about it;
12277 // otherwise default to the empty list. Never hardcode `mesh_tags`
12278 // here as well as in `extra_toml` -- TOML rejects a duplicate key.
12279 let mesh_tags = if extra_toml.contains("mesh_tags") {
12280 String::new()
12281 } else {
12282 "mesh_tags = []\n".to_string()
12283 };
12284 std::fs::write(
12285 dir.join(format!("{name}.toml")),
12286 format!("name = \"{name}\"\nprovider = \"static\"\n{mesh_tags}{extra_toml}"),
12287 )
12288 .unwrap();
12289 }
12290
12291 fn write_min_provider(dir: &Path, id: &str) {
12292 std::fs::create_dir_all(dir).unwrap();
12293 std::fs::write(
12294 dir.join(format!("{id}.toml")),
12295 format!("schema_version = 1\nid = \"{id}\"\nkind = \"static\"\n"),
12296 )
12297 .unwrap();
12298 }
12299
12300 fn write_sources_toml(camp_root: &Path, body: &str) {
12301 let dir = camp_root.join(".yah/infra");
12302 std::fs::create_dir_all(&dir).unwrap();
12303 std::fs::write(dir.join("sources.toml"), body).unwrap();
12304 }
12305
12306 #[test]
12307 fn load_with_no_sources_toml_is_unchanged() {
12308 let tmp = tempfile::TempDir::new().unwrap();
12309 write_min_machine(&tmp.path().join(".yah/infra/machines"), "local-1", "");
12310 let cfg = CloudConfig::load(tmp.path()).unwrap();
12311 assert_eq!(cfg.machines.len(), 1);
12312 assert!(cfg.machine_origins.is_empty());
12313 assert!(cfg.provider_origins.is_empty());
12314 }
12315
12316 #[test]
12317 fn path_source_overlays_machines_and_providers_tagged_with_origin() {
12318 let camp = tempfile::TempDir::new().unwrap();
12319 let other = tempfile::TempDir::new().unwrap();
12320 write_min_machine(&other.path().join(".yah/infra/machines"), "borrowed-1", "");
12321 write_min_provider(&other.path().join(".yah/infra/providers"), "borrowed-provider");
12322 write_sources_toml(
12323 camp.path(),
12324 &format!(
12325 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
12326 other.path().display()
12327 ),
12328 );
12329
12330 let cfg = CloudConfig::load(camp.path()).unwrap();
12331 assert_eq!(cfg.machines.len(), 1);
12332 assert_eq!(cfg.machines[0].name, "borrowed-1");
12333 assert_eq!(cfg.providers.len(), 1);
12334 assert_eq!(cfg.providers[0].id, "borrowed-provider");
12335
12336 let origin = cfg.machine_origins.get("borrowed-1").expect("origin recorded");
12337 assert_eq!(origin.owner, "other");
12338 assert_eq!(origin.mode, SourceMode::ReadOnly);
12339 assert!(origin.source.starts_with("path:"));
12340 assert_eq!(
12341 cfg.provider_origins.get("borrowed-provider").unwrap().owner,
12342 "other"
12343 );
12344 }
12345
12346 #[test]
12347 fn camp_local_wins_on_name_collision_and_carries_no_origin() {
12348 let camp = tempfile::TempDir::new().unwrap();
12349 let other = tempfile::TempDir::new().unwrap();
12350 // Both declare a machine named "shared" -- camp-local's copy must win,
12351 // and it must never gain an origin tag.
12352 write_min_machine(&camp.path().join(".yah/infra/machines"), "shared", "");
12353 write_min_machine(
12354 &other.path().join(".yah/infra/machines"),
12355 "shared",
12356 "nickname = \"the borrowed one\"\n",
12357 );
12358 write_sources_toml(
12359 camp.path(),
12360 &format!(
12361 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
12362 other.path().display()
12363 ),
12364 );
12365
12366 let cfg = CloudConfig::load(camp.path()).unwrap();
12367 assert_eq!(cfg.machines.len(), 1, "the name collides, so exactly one entry");
12368 assert_eq!(cfg.machines[0].nickname, None, "camp-local's copy, not the borrowed one");
12369 assert!(
12370 !cfg.machine_origins.contains_key("shared"),
12371 "camp-local entries never carry an origin tag"
12372 );
12373 }
12374
12375 #[test]
12376 fn an_earlier_source_wins_over_a_later_one_on_collision() {
12377 let camp = tempfile::TempDir::new().unwrap();
12378 let first = tempfile::TempDir::new().unwrap();
12379 let second = tempfile::TempDir::new().unwrap();
12380 write_min_machine(&first.path().join(".yah/infra/machines"), "dup", "");
12381 write_min_machine(&second.path().join(".yah/infra/machines"), "dup", "");
12382 write_sources_toml(
12383 camp.path(),
12384 &format!(
12385 "schema_version = 1\n\n[[source]]\nowner = \"first\"\nkind = \"path\"\npath = \"{}\"\n\n[[source]]\nowner = \"second\"\nkind = \"path\"\npath = \"{}\"\n",
12386 first.path().display(),
12387 second.path().display()
12388 ),
12389 );
12390
12391 let cfg = CloudConfig::load(camp.path()).unwrap();
12392 assert_eq!(cfg.machines.len(), 1);
12393 assert_eq!(cfg.machine_origins.get("dup").unwrap().owner, "first");
12394 }
12395
12396 #[test]
12397 fn select_filters_borrowed_machines_by_name_or_mesh_tag() {
12398 let camp = tempfile::TempDir::new().unwrap();
12399 let other = tempfile::TempDir::new().unwrap();
12400 write_min_machine(&other.path().join(".yah/infra/machines"), "runner-1", "mesh_tags = [\"tag:cloud-runner\"]\n");
12401 write_min_machine(&other.path().join(".yah/infra/machines"), "excluded-1", "");
12402 write_sources_toml(
12403 camp.path(),
12404 &format!(
12405 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\nselect = [\"tag:cloud-runner\"]\n",
12406 other.path().display()
12407 ),
12408 );
12409
12410 let cfg = CloudConfig::load(camp.path()).unwrap();
12411 assert_eq!(cfg.machines.len(), 1);
12412 assert_eq!(cfg.machines[0].name, "runner-1");
12413 }
12414
12415 #[test]
12416 fn one_unparseable_foreign_machine_does_not_sink_the_rest_of_the_directory_or_the_load() {
12417 let camp = tempfile::TempDir::new().unwrap();
12418 let other = tempfile::TempDir::new().unwrap();
12419 let dir = other.path().join(".yah/infra/machines");
12420 write_min_machine(&dir, "good", "");
12421 // Schema-skew gotcha: a foreign machine this binary's MachineConfig
12422 // can't parse at all (not just an unknown field -- MachineConfig has
12423 // no deny_unknown_fields, so this has to fail on a TYPE, not a name).
12424 std::fs::write(dir.join("bad.toml"), "name = 1\nprovider = 2\n").unwrap();
12425 write_sources_toml(
12426 camp.path(),
12427 &format!(
12428 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
12429 other.path().display()
12430 ),
12431 );
12432
12433 // Must not error at all -- camp-local load must never fail because a
12434 // source it doesn't own has one bad file.
12435 let cfg = CloudConfig::load(camp.path()).unwrap();
12436 assert_eq!(cfg.machines.len(), 1, "the good entry still loads");
12437 assert_eq!(cfg.machines[0].name, "good");
12438 }
12439
12440 #[test]
12441 fn an_unsynced_git_source_overlays_nothing_and_is_not_an_error() {
12442 // No `yah infra sync` (R615-T3) has ever run, so the cache dir this
12443 // resolves to doesn't exist. Must be silent, not fatal.
12444 let camp = tempfile::TempDir::new().unwrap();
12445 write_sources_toml(
12446 camp.path(),
12447 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"git\"\nrepo = \"git@github.com:yah-ai/infra.git\"\nref = \"main\"\n",
12448 );
12449 let cfg = CloudConfig::load(camp.path()).unwrap();
12450 assert!(cfg.machines.is_empty());
12451 assert!(cfg.machine_origins.is_empty());
12452 }
12453
12454 #[test]
12455 fn a_synced_git_source_reads_from_the_cache_dir_not_the_repo_path() {
12456 // No `subdir` declared -- the checkout ROOT is the infra root.
12457 let camp = tempfile::TempDir::new().unwrap();
12458 let cache = crate::paths::infra_source_cache_dir(camp.path(), "yah");
12459 write_min_machine(&cache.join("machines"), "synced-1", "");
12460 write_sources_toml(
12461 camp.path(),
12462 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"git\"\nrepo = \"git@github.com:yah-ai/infra.git\"\nref = \"main\"\n",
12463 );
12464 let cfg = CloudConfig::load(camp.path()).unwrap();
12465 assert_eq!(cfg.machines.len(), 1);
12466 assert_eq!(cfg.machines[0].name, "synced-1");
12467 assert!(cfg.machine_origins.get("synced-1").unwrap().source.starts_with("git:"));
12468 }
12469
12470 #[test]
12471 fn a_git_sources_subdir_is_honoured_like_the_component_case() {
12472 // W274's own example declares `subdir = "infra"` for a monorepo whose
12473 // registry lives under a subdirectory of the clone rather than at its
12474 // root -- prove `infra_root` actually reads it, not just `.subdir` on
12475 // GitSource parsing (R615-F1 already covers that half).
12476 let camp = tempfile::TempDir::new().unwrap();
12477 let cache = crate::paths::infra_source_cache_dir(camp.path(), "yah");
12478 write_min_machine(&cache.join("infra").join("machines"), "subdir-1", "");
12479 // Also plant a decoy at the checkout root to prove the root itself is
12480 // NOT read when a subdir is declared.
12481 write_min_machine(&cache.join("machines"), "root-decoy", "");
12482 write_sources_toml(
12483 camp.path(),
12484 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"git\"\nrepo = \"git@github.com:yah-ai/infra.git\"\nref = \"main\"\nsubdir = \"infra\"\n",
12485 );
12486 let cfg = CloudConfig::load(camp.path()).unwrap();
12487 assert_eq!(cfg.machines.len(), 1);
12488 assert_eq!(cfg.machines[0].name, "subdir-1");
12489 }
12490
12491 #[test]
12492 fn load_from_config_dir_never_applies_sources_overlay() {
12493 // R615-F2's explicit decision: multi-root sibling trees don't inherit
12494 // the classic .yah/infra/sources.toml. Prove it rather than assert it
12495 // silently -- a sources.toml sitting at workspace_root/.yah/infra/
12496 // must NOT leak into a load_from_config_dir call even though both
12497 // share the same workspace_root.
12498 let camp = tempfile::TempDir::new().unwrap();
12499 let other = tempfile::TempDir::new().unwrap();
12500 write_min_machine(&other.path().join(".yah/infra/machines"), "borrowed-1", "");
12501 write_sources_toml(
12502 camp.path(),
12503 &format!(
12504 "schema_version = 1\n\n[[source]]\nowner = \"other\"\nkind = \"path\"\npath = \"{}\"\n",
12505 other.path().display()
12506 ),
12507 );
12508 let sibling_config_dir = camp.path().join(".noisetable");
12509 std::fs::create_dir_all(&sibling_config_dir).unwrap();
12510
12511 let cfg = CloudConfig::load_from_config_dir(&sibling_config_dir, camp.path()).unwrap();
12512 assert!(cfg.machines.is_empty(), "sources.toml must not apply here");
12513 assert!(cfg.machine_origins.is_empty());
12514 }
12515
12516 // ─── R615-T5: `inherit_machines` retirement — cutover proof ────────────
12517
12518 /// The successor to R615-T5's parity proof. That earlier pair of tests
12519 /// asserted the legacy `[infra].inherit_machines` redirect and an
12520 /// equivalent `kind = "path"` source resolved the same machine set, and
12521 /// that the two coexisted without duplicating rows. Both claims were about
12522 /// a mechanism that no longer exists, so they retired with it — what has
12523 /// to hold *now* is the other half of the same guarantee: a camp that
12524 /// declares only `sources.toml` resolves the shared root exactly as the
12525 /// redirect used to, and a stale `inherit_machines` key left behind in
12526 /// `camp.toml` changes nothing.
12527 ///
12528 /// That stale-key case is not hypothetical: it is precisely the state a
12529 /// camp is in between the code cutover and someone tidying its
12530 /// `camp.toml`, and a silent re-resolution there would double-count the
12531 /// borrowed nodes or hide their origin badge.
12532 #[test]
12533 fn a_stale_inherit_machines_key_does_not_change_what_sources_toml_resolves() {
12534 let shared = tempfile::TempDir::new().unwrap();
12535 write_min_machine(&shared.path().join(".yah/infra/machines"), "shared-node-1", "");
12536 write_min_machine(&shared.path().join(".yah/infra/machines"), "shared-node-2", "");
12537
12538 let sources_toml = format!(
12539 "schema_version = 1\n\n[[source]]\nowner = \"yah\"\nkind = \"path\"\npath = \"{}\"\nmode = \"read-only\"\n",
12540 shared.path().display()
12541 );
12542
12543 // Camp A: migrated cleanly — sources.toml only.
12544 let clean = tempfile::TempDir::new().unwrap();
12545 write_sources_toml(clean.path(), &sources_toml);
12546
12547 // Camp B: mid-migration — same source, plus the retired key still
12548 // sitting in camp.toml pointing at the same root.
12549 let stale = tempfile::TempDir::new().unwrap();
12550 std::fs::create_dir_all(stale.path().join(".yah")).unwrap();
12551 std::fs::write(
12552 stale.path().join(".yah/camp.toml"),
12553 format!(
12554 "[infra]\ninherit_machines = \"{}\"\n",
12555 shared.path().display()
12556 ),
12557 )
12558 .unwrap();
12559 write_sources_toml(stale.path(), &sources_toml);
12560
12561 let via_clean = CloudConfig::load(clean.path()).unwrap();
12562 let via_stale = CloudConfig::load(stale.path()).unwrap();
12563
12564 let names = |cfg: &CloudConfig| {
12565 let mut v: Vec<String> = cfg.machines.iter().map(|m| m.name.clone()).collect();
12566 v.sort();
12567 v
12568 };
12569 assert_eq!(
12570 names(&via_clean),
12571 names(&via_stale),
12572 "a leftover inherit_machines key must be inert — the retired redirect is gone"
12573 );
12574 assert_eq!(names(&via_clean), vec!["shared-node-1", "shared-node-2"]);
12575
12576 // And both are *borrowed*, not camp-local. This is the operator-facing
12577 // win the stopgap could never deliver: under the old redirect these
12578 // resolved with no origin at all, indistinguishable from locally-owned
12579 // nodes.
12580 assert_eq!(via_clean.machine_origins.len(), 2);
12581 assert_eq!(via_stale.machine_origins.len(), 2);
12582 for origin in via_stale.machine_origins.values() {
12583 assert_eq!(origin.owner, "yah");
12584 assert_eq!(origin.mode, SourceMode::ReadOnly);
12585 }
12586 }
12587
12588 // ─── R860-T4 (W338): placement groups ───────────────────────────────────
12589
12590 /// One requirement edge, written the way a spec author writes it.
12591 fn requirement(ident: &str, locality: Locality) -> workload_spec::Requirement {
12592 workload_spec::Requirement {
12593 ident: workload_spec::MeshIdent(ident.into()),
12594 locality,
12595 supply: workload_spec::Supply::Wait,
12596 provides: None,
12597 }
12598 }
12599
12600 /// A `minimal_spec` (256 MiB / 250 millicores, Server by inference) that
12601 /// requires the given edges.
12602 fn spec_requiring(name: &str, requires: Vec<workload_spec::Requirement>) -> WorkloadSpec {
12603 WorkloadSpec {
12604 requires,
12605 ..minimal_spec(name, 1)
12606 }
12607 }
12608
12609 /// The declared inventory an ident is resolved against — `.yah/infra/workloads/`.
12610 fn declared(specs: Vec<WorkloadSpec>) -> Vec<WorkloadConfig> {
12611 specs.into_iter().map(|spec| WorkloadConfig { spec }).collect()
12612 }
12613
12614 fn member_names(group: &[WorkloadSpec]) -> Vec<&str> {
12615 group.iter().map(|s| s.name.as_str()).collect()
12616 }
12617
12618 /// The headline case: `local` means "same node", so the two specs are one
12619 /// placement unit and admission has to reason about both.
12620 #[test]
12621 fn a_local_edge_binds_the_provider_into_the_placement_group() {
12622 let inventory = declared(vec![minimal_spec("headscale-replicator", 1)]);
12623 let requirer = spec_requiring(
12624 "headscale",
12625 vec![requirement("headscale-replicator", Locality::Local)],
12626 );
12627
12628 assert_eq!(
12629 member_names(&placement_group(&requirer, &inventory)),
12630 vec!["headscale", "headscale-replicator"]
12631 );
12632 }
12633
12634 /// The edge that must NOT bind. `prefer-local` "never blocks placement"
12635 /// (W338's locality table), and `anywhere` — which is what every legacy
12636 /// `depends_on` folds into — is an ordinary service dependency. Binding
12637 /// either would silently make every dependency in the tree a co-scheduling
12638 /// constraint and start summing unrelated workloads into the capacity floor.
12639 #[test]
12640 fn prefer_local_and_anywhere_edges_do_not_bind_the_group() {
12641 let inventory = declared(vec![
12642 minimal_spec("headscale-db", 1),
12643 minimal_spec("metrics", 1),
12644 minimal_spec("legacy-dep", 1),
12645 ]);
12646
12647 let requirer = WorkloadSpec {
12648 depends_on: vec![workload_spec::MeshIdent("legacy-dep".into())],
12649 ..spec_requiring(
12650 "headscale",
12651 vec![
12652 requirement("headscale-db", Locality::PreferLocal),
12653 requirement("metrics", Locality::Anywhere),
12654 ],
12655 )
12656 };
12657
12658 assert_eq!(
12659 member_names(&placement_group(&requirer, &inventory)),
12660 vec!["headscale"]
12661 );
12662 }
12663
12664 /// Transitive, and via the inline spec a `supply = "self"` requirement
12665 /// carries rather than via an ident lookup — the sidecar shape W338's
12666 /// worked example is built on.
12667 #[test]
12668 fn the_group_is_the_transitive_closure_and_traverses_inline_provides() {
12669 let inline = workload_spec::Requirement {
12670 supply: workload_spec::Supply::SelfProvision,
12671 provides: Some(Box::new(minimal_spec("headscale-restore", 1))),
12672 ..requirement("headscale-restore", Locality::Local)
12673 };
12674 let middle = WorkloadSpec {
12675 requires: vec![requirement("wal-shipper", Locality::Local)],
12676 ..minimal_spec("headscale-replicator", 1)
12677 };
12678 let inventory = declared(vec![middle, minimal_spec("wal-shipper", 1)]);
12679
12680 let requirer = spec_requiring(
12681 "headscale",
12682 vec![
12683 inline,
12684 requirement("headscale-replicator", Locality::Local),
12685 ],
12686 );
12687
12688 assert_eq!(
12689 member_names(&placement_group(&requirer, &inventory)),
12690 vec![
12691 "headscale",
12692 "headscale-restore",
12693 "headscale-replicator",
12694 "wal-shipper"
12695 ]
12696 );
12697 }
12698
12699 /// `validate::check_requires` bounds `provides` nesting to depth 1 but
12700 /// cannot stop two separately-declared specs from naming each other. Without
12701 /// the visited set this closure never terminates, so admission would hang
12702 /// rather than refuse — the worst failure shape for a deploy gate.
12703 #[test]
12704 fn an_ident_cycle_closes_the_group_instead_of_looping_forever() {
12705 let b = spec_requiring("b", vec![requirement("a", Locality::Local)]);
12706 let a = spec_requiring("a", vec![requirement("b", Locality::Local)]);
12707 let inventory = declared(vec![a.clone(), b]);
12708
12709 assert_eq!(member_names(&placement_group(&a, &inventory)), vec!["a", "b"]);
12710 }
12711
12712 /// An unresolvable ident is skipped, not fatal: admission is a pure function
12713 /// of the declared inventory, and refusing every deploy whose provider is
12714 /// not yet declared would make `requires` unusable before R860-T6 lands.
12715 #[test]
12716 fn an_unresolvable_local_ident_is_skipped_rather_than_refused() {
12717 let requirer = spec_requiring("headscale", vec![requirement("not-declared", Locality::Local)]);
12718 assert_eq!(
12719 member_names(&placement_group(&requirer, &[])),
12720 vec!["headscale"]
12721 );
12722 }
12723
12724 /// W338 §Placement consequences 1: the capacity floor is the group's sum.
12725 /// A node that fits the requirer alone must refuse the group — placing it
12726 /// there would oversubscribe the node the moment the provider follows.
12727 #[test]
12728 fn the_capacity_floor_is_the_sum_of_the_group_not_the_requirer_alone() {
12729 let provider = minimal_spec("headscale-replicator", 1);
12730 let requirer = spec_requiring(
12731 "headscale",
12732 vec![requirement("headscale-replicator", Locality::Local)],
12733 );
12734 // Two `minimal_spec`s: 256 MiB + 250 millicores each.
12735 let inventory = declared(vec![provider]);
12736
12737 let too_small = CloudConfig {
12738 workloads: inventory.clone(),
12739 ..make_empty_cfg(vec![make_machine_with_capacity("small", 300, 4000, vec![])])
12740 };
12741 let err = too_small.admit_workload(&requirer).unwrap_err().to_string();
12742 assert!(
12743 err.contains("memory_mb>=512"),
12744 "the floor must name the group's summed demand, got: {err}"
12745 );
12746
12747 let big_enough = CloudConfig {
12748 workloads: inventory,
12749 ..make_empty_cfg(vec![make_machine_with_capacity("roomy", 512, 4000, vec![])])
12750 };
12751 assert_eq!(
12752 big_enough.admit_workload(&requirer).unwrap().name,
12753 "roomy",
12754 "a node covering the sum must still admit the group"
12755 );
12756 }
12757
12758 /// W338 §Placement consequences 2, and the reason repulsion is computed over
12759 /// a set at all: the requirer is a `Server`, so the pre-R860 axis would have
12760 /// let it onto a `no-appliance` dev Pi and dragged its Appliance provider
12761 /// there with it.
12762 #[test]
12763 fn a_server_requiring_an_appliance_locally_is_repelled_by_no_appliance() {
12764 let appliance = WorkloadSpec {
12765 archetype: Some(LifecycleArchetype::Appliance),
12766 ..minimal_spec("headscale", 1)
12767 };
12768 let requirer = spec_requiring("headscale-ui", vec![requirement("headscale", Locality::Local)]);
12769 assert_eq!(
12770 requirer.effective_archetype(),
12771 LifecycleArchetype::Server,
12772 "precondition: the requirer itself must not be an Appliance"
12773 );
12774
12775 let cfg = CloudConfig {
12776 workloads: declared(vec![appliance]),
12777 ..make_empty_cfg(vec![
12778 make_machine_with_capacity("dev-pi", 8192, 4000, vec!["no-appliance"]),
12779 make_machine_with_capacity("us-west-001", 8192, 4000, vec![]),
12780 ])
12781 };
12782
12783 assert_eq!(
12784 cfg.admit_workload(&requirer).unwrap().name,
12785 "us-west-001",
12786 "the dev Pi repels the group's Appliance member"
12787 );
12788
12789 // And with the Appliance gone from the group, the same requirer is
12790 // admissible on the same Pi — proving the repulsion came from the edge.
12791 let alone = minimal_spec("headscale-ui", 1);
12792 assert_eq!(cfg.admit_workload(&alone).unwrap().name, "dev-pi");
12793 }
12794
12795 /// The set-valued form of the per-workload drain skip the node makes in
12796 /// `drain_workloads`: one Appliance member pins the whole group.
12797 #[test]
12798 fn a_group_containing_an_appliance_is_not_drainable() {
12799 let server = minimal_spec("headscale-ui", 1);
12800 let appliance = WorkloadSpec {
12801 archetype: Some(LifecycleArchetype::Appliance),
12802 ..minimal_spec("headscale", 1)
12803 };
12804
12805 assert!(group_is_drainable(std::slice::from_ref(&server)));
12806 assert!(!group_is_drainable(&[server, appliance]));
12807 }
12808
12809 /// The regression that matters most: nothing in the tree declares
12810 /// `requires` yet, so every existing spec's group is exactly itself and its
12811 /// admission axes must be bit-identical to the pre-R860 derivation.
12812 #[test]
12813 fn a_spec_with_no_local_edges_admits_exactly_as_it_did_before() {
12814 let ws = ws_with_selector(Some("tag:build-worker,arch:x86"));
12815 let req = admission_spec(&ws, &[]).unwrap();
12816
12817 assert_eq!(req.mesh_tags, vec!["tag:build-worker", "arch:x86"]);
12818 assert_eq!(req.memory_mb, ws.memory_request_mb());
12819 assert_eq!(req.cpu_millis, ws.resources.cpu_millis);
12820 // R876-B7: the axis is now the complement — every repelling key EXCEPT
12821 // this spec's own class, which is the same predicate stated from the
12822 // other side. Asserted against the derivation rather than a literal so
12823 // it stays true if a fourth archetype is added.
12824 assert_eq!(
12825 req.tolerates,
12826 tolerations_excluding(&[ws.effective_archetype()])
12827 );
12828 let own = format!("no-{}", ws.effective_archetype().taint_key());
12829 assert!(
12830 !req.tolerates.contains(&own),
12831 "a spec never tolerates the taint aimed at its own class"
12832 );
12833 }
12834
12835 // ─── R860-T5 (W338 §Placement consequences 3): native-exec capability ────
12836
12837 /// A `minimal_spec` carrying the `yah.exec = native` marker — the only way
12838 /// a workload says "fork+exec me on the host" (`WorkloadSpec::
12839 /// wants_native_exec`). It stays a Container workload on the wire; the
12840 /// marker is the whole difference.
12841 fn native_spec(name: &str) -> WorkloadSpec {
12842 let mut ws = minimal_spec(name, 1);
12843 ws.annotations.insert(
12844 workload_spec::NATIVE_EXEC_ANNOTATION.to_string(),
12845 workload_spec::NATIVE_EXEC_VALUE.to_string(),
12846 );
12847 assert!(ws.wants_native_exec(), "precondition: the marker must read back");
12848 ws
12849 }
12850
12851 /// The R858 failure, now caught at placement instead of at dispatch: a node
12852 /// whose kamaji has no `--native-exec-dir` accepted the election and then
12853 /// refused the deploy, and nothing upstream could see it coming.
12854 #[test]
12855 fn a_node_without_the_native_exec_capability_cannot_host_a_native_workload() {
12856 let native = native_spec("headscale");
12857
12858 let incapable = make_empty_cfg(vec![make_machine("us-south-001", vec![])]);
12859 let err = incapable.admit_workload(&native).unwrap_err().to_string();
12860 assert!(
12861 err.contains(NATIVE_EXEC_MESH_TAG),
12862 "the refusal must name the missing capability, got: {err}"
12863 );
12864
12865 let capable = make_empty_cfg(vec![
12866 make_machine("us-south-001", vec![]),
12867 make_machine("us-west-001", vec![NATIVE_EXEC_MESH_TAG]),
12868 ]);
12869 assert_eq!(
12870 capable.admit_workload(&native).unwrap().name,
12871 "us-west-001",
12872 "a node declaring the capability admits the native workload"
12873 );
12874 }
12875
12876 /// W338's actual sentence: a `supply = "self"` spec "must be placeable where
12877 /// its requirer lands". The requirer here is an ordinary container workload
12878 /// — it is the *provider* reached by a `local` edge that needs the host
12879 /// backend, so the capability has to be required of the group, not of the
12880 /// spec being deployed.
12881 #[test]
12882 fn a_local_edge_to_a_native_provider_makes_the_requirer_need_the_capability() {
12883 let requirer = spec_requiring(
12884 "headscale-ui",
12885 vec![requirement("headscale", Locality::Local)],
12886 );
12887 assert!(
12888 !requirer.wants_native_exec(),
12889 "precondition: the requirer itself is an ordinary container workload"
12890 );
12891
12892 let cfg = CloudConfig {
12893 workloads: declared(vec![native_spec("headscale")]),
12894 ..make_empty_cfg(vec![
12895 make_machine("plain", vec![]),
12896 make_machine("us-west-001", vec![NATIVE_EXEC_MESH_TAG]),
12897 ])
12898 };
12899
12900 assert_eq!(
12901 cfg.admit_workload(&requirer).unwrap().name,
12902 "us-west-001",
12903 "the group's native member pulls the requirer onto a capable node"
12904 );
12905
12906 // Without the edge the same requirer is admissible on the plain node,
12907 // so the constraint provably came from the group and not from the spec.
12908 let alone = minimal_spec("headscale-ui", 1);
12909 assert_eq!(cfg.admit_workload(&alone).unwrap().name, "plain");
12910 }
12911
12912 /// The regression guard: nothing in the tree is native-marked today, so
12913 /// every existing spec's axes must be untouched by this ticket.
12914 #[test]
12915 fn a_group_with_no_native_member_does_not_require_the_capability() {
12916 let inventory = declared(vec![minimal_spec("headscale-replicator", 1)]);
12917 let requirer = spec_requiring(
12918 "headscale",
12919 vec![requirement("headscale-replicator", Locality::Local)],
12920 );
12921
12922 let req = admission_spec(&requirer, &inventory).unwrap();
12923 assert!(
12924 !req.mesh_tags.iter().any(|t| t == NATIVE_EXEC_MESH_TAG),
12925 "no native member ⇒ no capability axis, got: {:?}",
12926 req.mesh_tags
12927 );
12928
12929 // And it still lands on a node that declares nothing at all.
12930 let cfg = CloudConfig {
12931 workloads: inventory,
12932 ..make_empty_cfg(vec![make_machine("plain", vec![])])
12933 };
12934 assert_eq!(cfg.admit_workload(&requirer).unwrap().name, "plain");
12935 }
12936
12937 // ─── R894-F1: trust declares a minimum isolation substrate ───────────────
12938
12939 /// A `minimal_spec` marked untrusted **without** raising its substrate —
12940 /// i.e. the incoherent pairing this ticket exists to refuse.
12941 ///
12942 /// It deliberately does not go through `WorkloadSpec::stamp_untrusted`,
12943 /// which raises the substrate as it stamps: the point of these tests is the
12944 /// admission-side gate, so the spec has to arrive in the state a buggy or
12945 /// hostile producer would leave it in, not the state the correct producer
12946 /// guarantees.
12947 fn untrusted_spec(name: &str, exec: Option<&str>) -> WorkloadSpec {
12948 let mut ws = minimal_spec(name, 1);
12949 ws.annotations.insert(
12950 workload_spec::TRUST_ANNOTATION.to_string(),
12951 workload_spec::TRUST_UNTRUSTED_VALUE.to_string(),
12952 );
12953 if let Some(v) = exec {
12954 ws.annotations.insert(
12955 workload_spec::NATIVE_EXEC_ANNOTATION.to_string(),
12956 v.to_string(),
12957 );
12958 }
12959 assert_eq!(
12960 ws.trust().unwrap(),
12961 workload_spec::TrustLevel::Untrusted,
12962 "precondition: the trust marker must read back"
12963 );
12964 ws
12965 }
12966
12967 /// THE ACCEPTANCE GATE. Both refusals happen inside `admit_workload` — the
12968 /// real dispatch path `MeshYubabaClient::elect_node` calls — against a fleet
12969 /// that contains a microVM-capable node, so the refusal is provably the
12970 /// trust rule and not "nothing admits this".
12971 #[test]
12972 fn admit_workload_refuses_untrusted_code_on_a_shared_kernel() {
12973 let cfg = make_empty_cfg(vec![
12974 make_machine("plain", vec![]),
12975 make_machine("us-west-003", vec![MICROVM_MESH_TAG]),
12976 ]);
12977
12978 // (1) Untrusted + `yah.exec = native`: the widest possible substrate.
12979 let native = untrusted_spec("tenant-camp", Some(workload_spec::NATIVE_EXEC_VALUE));
12980 let err = cfg.admit_workload(&native).unwrap_err().to_string();
12981 assert!(
12982 err.contains("tenant-camp") && err.contains("native") && err.contains("microvm"),
12983 "the refusal must name the workload, what it asked for and the floor, got: {err}"
12984 );
12985
12986 // (2) Untrusted with no substrate marker at all — the container default.
12987 // This is the one a "absent means trusted for ALL origins" spelling
12988 // would have admitted.
12989 let container = untrusted_spec("tenant-camp", None);
12990 assert_eq!(
12991 container.exec_substrate(),
12992 workload_spec::ExecSubstrate::Container,
12993 "precondition: no marker means the container backend"
12994 );
12995 let err = cfg.admit_workload(&container).unwrap_err().to_string();
12996 assert!(
12997 err.contains("container") && err.contains("microvm"),
12998 "an unmarked untrusted spec is refused for the same reason, got: {err}"
12999 );
13000
13001 // (3) The same spec asking for a microVM is admitted, onto the node that
13002 // declares the capability. Same fleet, same workload name — so (1) and
13003 // (2) provably failed on the substrate and nothing else.
13004 let vm = untrusted_spec("tenant-camp", Some(workload_spec::MICROVM_EXEC_VALUE));
13005 assert_eq!(cfg.admit_workload(&vm).unwrap().name, "us-west-003");
13006 }
13007
13008 /// A caller may always request *more* isolation than it needs. A trusted
13009 /// workload asking for a microVM is not "exceeding" anything — the rule is
13010 /// one-directional.
13011 #[test]
13012 fn a_trusted_workload_may_still_request_a_stricter_substrate() {
13013 let mut ws = minimal_spec("forge-build", 1);
13014 ws.annotations.insert(
13015 workload_spec::NATIVE_EXEC_ANNOTATION.to_string(),
13016 workload_spec::MICROVM_EXEC_VALUE.to_string(),
13017 );
13018 assert_eq!(ws.trust().unwrap(), workload_spec::TrustLevel::Trusted);
13019
13020 let cfg = make_empty_cfg(vec![
13021 make_machine("plain", vec![]),
13022 make_machine("us-west-003", vec![MICROVM_MESH_TAG]),
13023 ]);
13024 assert_eq!(cfg.admit_workload(&ws).unwrap().name, "us-west-003");
13025 }
13026
13027 /// An untrusted member reached by a `local` edge is still on the node, so
13028 /// the group is refused even though the spec being deployed is fine — the
13029 /// same group-wide reasoning every other axis in `admission_spec` uses.
13030 #[test]
13031 fn an_untrusted_provider_refuses_the_whole_placement_group() {
13032 let requirer = spec_requiring(
13033 "camp-front",
13034 vec![requirement("tenant-camp", Locality::Local)],
13035 );
13036 assert_eq!(
13037 requirer.trust().unwrap(),
13038 workload_spec::TrustLevel::Trusted,
13039 "precondition: the requirer itself is an ordinary trusted workload"
13040 );
13041
13042 let cfg = CloudConfig {
13043 workloads: declared(vec![untrusted_spec("tenant-camp", None)]),
13044 ..make_empty_cfg(vec![make_machine("us-west-003", vec![MICROVM_MESH_TAG])])
13045 };
13046 let err = cfg.admit_workload(&requirer).unwrap_err().to_string();
13047 assert!(
13048 err.contains("tenant-camp"),
13049 "the refusal must name the offending MEMBER, not the requirer, got: {err}"
13050 );
13051
13052 // The same requirer without the edge admits fine, so the refusal
13053 // provably came from the group.
13054 let alone = minimal_spec("camp-front", 1);
13055 assert_eq!(cfg.admit_workload(&alone).unwrap().name, "us-west-003");
13056 }
13057
13058 /// A typo in the trust value is refused, not resolved to either side.
13059 /// Reading it as trusted would turn the microVM floor off by misspelling.
13060 #[test]
13061 fn an_unreadable_trust_declaration_is_refused_rather_than_defaulted() {
13062 let mut ws = minimal_spec("tenant-camp", 1);
13063 ws.annotations.insert(
13064 workload_spec::TRUST_ANNOTATION.to_string(),
13065 "untrused".to_string(),
13066 );
13067
13068 let cfg = make_empty_cfg(vec![make_machine("plain", vec![])]);
13069 let err = cfg.admit_workload(&ws).unwrap_err().to_string();
13070 assert!(
13071 err.contains("untrused") && err.contains("tenant-camp"),
13072 "the refusal must quote the unreadable value, got: {err}"
13073 );
13074 }
13075
13076 /// The regression guard for the new mesh-tag axis, in the shape R860-T5's
13077 /// own guard uses: nothing in the fleet is microVM-marked today, so every
13078 /// existing spec's axes must be untouched.
13079 #[test]
13080 fn a_group_with_no_microvm_member_does_not_require_the_capability() {
13081 let ws = minimal_spec("headscale-ui", 1);
13082 let req = admission_spec(&ws, &[]).unwrap();
13083 assert!(
13084 !req.mesh_tags.iter().any(|t| t == MICROVM_MESH_TAG),
13085 "no microVM member ⇒ no capability axis, got: {:?}",
13086 req.mesh_tags
13087 );
13088
13089 // And a microVM-marked spec is refused by a node declaring nothing,
13090 // naming the tag — the dispatch-time surprise R858 paid for.
13091 let mut vm = minimal_spec("forge-build", 1);
13092 vm.annotations.insert(
13093 workload_spec::NATIVE_EXEC_ANNOTATION.to_string(),
13094 workload_spec::MICROVM_EXEC_VALUE.to_string(),
13095 );
13096 let cfg = make_empty_cfg(vec![make_machine("plain", vec![])]);
13097 let err = cfg.admit_workload(&vm).unwrap_err().to_string();
13098 assert!(
13099 err.contains(MICROVM_MESH_TAG),
13100 "the refusal must name the missing capability, got: {err}"
13101 );
13102 }
13103
13104 /// The gate is structural, not a convention followed at three call sites:
13105 /// every admission entry point goes through `admission_spec`, so all three
13106 /// refuse the same spec.
13107 #[test]
13108 fn every_admission_entry_point_enforces_the_trust_floor() {
13109 let ws = untrusted_spec("tenant-camp", Some(workload_spec::NATIVE_EXEC_VALUE));
13110 let mut machine = make_machine("us-west-003", vec![MICROVM_MESH_TAG]);
13111 machine.sovereign_group = Some("home".to_string());
13112 let cfg = make_empty_cfg(vec![machine]);
13113
13114 assert!(cfg.admit_workload(&ws).is_err());
13115 assert!(cfg.admit_workload_candidates(&ws).is_err());
13116 assert!(cfg.admit_workload_in_group(&ws, "home").is_err());
13117 }
13118
13119 // ── R892-B1: a declared key must never be dropped in silence ─────────────
13120
13121 /// The real shape, not a hand-minimised one: a key check that passes on a
13122 /// toy spec and trips on a production file is worse than no check.
13123 /// Modelled on `.yah/infra/workloads/yah-cloud-admin.toml`.
13124 const REAL_WORKLOAD_TOML: &str = r#"
13125name = "yah-cloud-admin"
13126tier = "infra"
13127archetype = "server"
13128replicas = 1
13129restart_policy = "always"
13130
13131[image]
13132registry = "cr.yah.dev"
13133repository = "yah-cloud-admin"
13134tag = "20260903-amd64"
13135digest = "sha256:efaa7824ebf1654b226e4bf9cf4f86f1ce39b17900895b1f9d83e4d987f58733"
13136
13137[resources]
13138memory_mb = 256
13139cpu_millis = 250
13140
13141[annotations]
13142"yah.network" = "host"
13143
13144[expose.mesh]
13145identity = "yah-cloud-admin"
13146ports = [4325]
13147allow_from = []
13148
13149[[env]]
13150name = "YAH_CLOUD_ADMIN_ADDR"
13151value = { literal = { value = "100.64.0.1:4325" } }
13152
13153[[secrets]]
13154source = { cluster = { name = "cheers/cloud-admin/verify-key" } }
13155target = { file = { path = "/run/secrets/cheers-verify.key", mode = 0o400 } }
13156
13157[healthcheck]
13158probe = { http_get = { path = "/__mesofact/health", port = 4325, expect_status = 200 } }
13159interval = 15000
13160timeout = 5000
13161initial_delay = 20000
13162failure_threshold = 3
13163
13164[stop_policy]
13165signal = 15
13166grace_period = 10000
13167"#;
13168
13169 fn check_keys(src: &str) -> Result<()> {
13170 let spec: WorkloadSpec = toml::from_str(src).expect("fixture must parse");
13171 refuse_dropped_keys(src, &spec, "fixture.toml")
13172 }
13173
13174 /// Non-vacuity, and the guard against a check that flags everything: a file
13175 /// whose every key the schema knows must load clean.
13176 #[test]
13177 fn a_workload_file_whose_keys_all_survive_the_parse_is_accepted() {
13178 check_keys(REAL_WORKLOAD_TOML).expect("a fully-known file must not be refused");
13179 }
13180
13181 /// The 2026-09-11 outage, reproduced in one line of TOML: R885-T6 deleted
13182 /// `ResourceLimits::ephemeral_storage_mb`, every camp's file still declared
13183 /// it, and serde discarded it without a word.
13184 #[test]
13185 fn the_key_that_caused_the_outage_is_now_refused_by_name() {
13186 let src = REAL_WORKLOAD_TOML.replace(
13187 "cpu_millis = 250",
13188 "cpu_millis = 250\nephemeral_storage_mb = 256",
13189 );
13190 let err = check_keys(&src).expect_err("a dropped key must refuse the file");
13191 let msg = format!("{err:#}");
13192 assert!(
13193 msg.contains("resources.ephemeral_storage_mb"),
13194 "the refusal must name the key by its full path, got: {msg}"
13195 );
13196 }
13197
13198 /// Nested and top-level keys are both reported, so a misspelling anywhere in
13199 /// the file is as loud as a removed field.
13200 #[test]
13201 fn every_unknown_key_is_named_not_just_the_first() {
13202 let src = format!("{REAL_WORKLOAD_TOML}\nreplica_count = 3\n\n[healthcheck_typo]\nx = 1\n");
13203 let err = check_keys(&src).expect_err("unknown keys must refuse the file");
13204 let msg = format!("{err:#}");
13205 assert!(msg.contains("replica_count"), "got: {msg}");
13206 assert!(msg.contains("healthcheck_typo"), "got: {msg}");
13207 }
13208
13209 /// A value the serialiser re-spells (an enum, a duration) is not a dropped
13210 /// key. Only a missing KEY is, which is what keeps this check from firing on
13211 /// every legitimate file in the fleet.
13212 #[test]
13213 fn a_canonicalised_value_is_not_reported_as_dropped() {
13214 let declared = toml::Value::try_from(toml::toml! {
13215 restart = "always"
13216 [nested]
13217 list = [1, 2]
13218 })
13219 .unwrap();
13220 let kept = toml::Value::try_from(toml::toml! {
13221 restart = "Always"
13222 [nested]
13223 list = [9, 9]
13224 })
13225 .unwrap();
13226 let mut out = Vec::new();
13227 collect_dropped_keys(&declared, &kept, "", &mut out);
13228 assert!(out.is_empty(), "values differ, keys do not: {out:?}");
13229 }
13230
13231 /// R896-B5: a key whose field was deleted as inert loads with a warning
13232 /// instead of refusing the file, so deleting the field does not break every
13233 /// camp whose committed TOML still declares it. Driven off the registry, so
13234 /// the next retired key is covered the moment it is listed.
13235 #[test]
13236 fn every_retired_key_is_ignored_not_refused() {
13237 for retired in workload_spec::RETIRED_KEYS {
13238 let mut declared: toml::Value = toml::from_str(REAL_WORKLOAD_TOML).unwrap();
13239 let (parents, leaf) = match retired.path.rsplit_once('.') {
13240 Some((p, l)) => (p.split('.').collect::<Vec<_>>(), l),
13241 None => (vec![], retired.path),
13242 };
13243 let mut table = declared.as_table_mut().unwrap();
13244 for p in parents {
13245 table = table
13246 .entry(p)
13247 .or_insert_with(|| toml::Value::Table(Default::default()))
13248 .as_table_mut()
13249 .unwrap();
13250 }
13251 table.insert(leaf.into(), toml::Value::Integer(1));
13252 let src = toml::to_string(&declared).unwrap();
13253 check_keys(&src)
13254 .unwrap_or_else(|e| panic!("retired `{}` must not refuse: {e:#}", retired.path));
13255 }
13256 }
13257
13258 /// The registry is an exemption from R892-B1, not a hole in it: a retired
13259 /// key sitting beside an unknown one still refuses the file, naming only the
13260 /// unknown key.
13261 #[test]
13262 fn a_retired_key_does_not_excuse_an_unknown_one() {
13263 let src = format!("schema_version = 1\n{REAL_WORKLOAD_TOML}\nreplica_count = 3\n");
13264 let msg = format!("{:#}", check_keys(&src).expect_err("unknown key must still refuse"));
13265 assert!(msg.contains("replica_count"), "got: {msg}");
13266 assert!(!msg.contains("schema_version"), "retired key must not be named: {msg}");
13267 }
13268}