boatramp_node/node.rs
1//! The node-graph assembly: given a built store (blobs + KV), a configured
2//! [`Auth`](boatramp_server::Auth), and resolved
3//! [`ServerOptions`](boatramp_server::ServerOptions), wire the deploy store,
4//! handler runtime, compute reconcile loop, and domain-verify reconcile loop
5//! into a [`RunningNode`] ready to hand to a transport (`serve_with` & friends).
6//!
7//! This is the headline extraction of `PLAN-node-library`: the binary's
8//! `serve::run` used to inline this wiring, so no embedder or in-process test
9//! could exercise the same graph the `boatramp serve` binary runs. `run` now
10//! resolves the *environment* (args -> backends -> store, signal handlers,
11//! migration, auth) and calls [`assemble`]; the cluster path keeps its own inline
12//! copy until a later step converges it here.
13
14use std::path::Path;
15use std::sync::Arc;
16
17use boatramp_core::Storage;
18use boatramp_core::deploy::DeployStore;
19use boatramp_core::kv::KvStore;
20
21use crate::config::ServerConfig;
22use crate::error::{Error, Result};
23
24/// How often the compute reconcile loop converges desired vs actual workloads.
25/// Defaults to 30s; override with `BOATRAMP_COMPUTE_RECONCILE_TICK_MS` (milliseconds)
26/// so compute-backed tests can converge in a fraction of a second instead of
27/// waiting a full tick for the launch/scale reconcile.
28pub fn compute_reconcile_tick() -> std::time::Duration {
29 std::env::var("BOATRAMP_COMPUTE_RECONCILE_TICK_MS")
30 .ok()
31 .and_then(|s| s.parse::<u64>().ok())
32 .filter(|&ms| ms > 0)
33 .map(std::time::Duration::from_millis)
34 .unwrap_or(std::time::Duration::from_secs(30))
35}
36/// How often the domain-verify reconcile loop re-checks pending challenges.
37pub const DOMAIN_VERIFY_RECONCILE_TICK: std::time::Duration = std::time::Duration::from_secs(60);
38/// How long a compute workload may be idle before scale-to-zero sleeps it.
39pub const COMPUTE_IDLE_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(300);
40
41/// The built store + resolved config handed to [`assemble`]. Owns the blob/KV
42/// backends and the auth/options the caller already resolved; borrows the parsed
43/// config and data directory.
44pub struct NodeInput<'a> {
45 /// The full parsed server config (the handler + compute sections are read here).
46 pub config: &'a ServerConfig,
47 /// The node data directory (per-site SQL, handler state).
48 pub data_dir: &'a Path,
49 /// The object store built by [`crate::blobs::build_blobs`].
50 pub storage: Arc<dyn Storage>,
51 /// The metadata KV, already cache-fronted, built by [`crate::backends::build_kv`].
52 pub kv: Arc<dyn KvStore>,
53 /// The control-plane auth built by [`crate::auth::configure_auth`].
54 pub auth: boatramp_server::Auth,
55 /// Server options, already carrying the resolved posture, daemon runtime, and
56 /// (post-`configure_auth`/`configure_oidc`) issuer / OIDC verifier.
57 pub options: boatramp_server::ServerOptions,
58 /// The public HTTP serve bind address, if known — used (under
59 /// `allow_guest_self_egress`) to let a handler guest's `wasi:http` reach this
60 /// instance's own front door over loopback. `None` (an in-process embedder with no
61 /// listener) disables self-egress.
62 pub serve_addr: Option<std::net::SocketAddr>,
63 /// The cloud blob-change watch provider (FA-5b2), if the backend is a cloud one.
64 pub watch_provider: Option<Arc<dyn boatramp_core::blob_provision::WatchProvider>>,
65 /// The provisioning tier for the watch provider.
66 pub provision_tier: boatramp_core::blob_notify::ProvisionTier,
67 /// The `wasi:messaging` substrate override for the handler runtime. `None` uses
68 /// the single-node default (`LogMessaging` over the same backends); the cluster
69 /// path passes its Raft-backed coordinator.
70 pub messaging: Option<Arc<dyn boatramp_core::messaging::Messaging>>,
71 /// The single leader gate for cron firing + the compute / domain-verify reconcile
72 /// loops. Single-node passes an always-true gate (there is one node); the cluster
73 /// passes its Raft `is_leader` check so a single node drives each sweep.
74 pub is_leader: boatramp_server::CronLeaderGate,
75 /// This node's compute scheduler id (`0` single-node; the cluster node id in a
76 /// fleet, so replicas are tagged to the right node).
77 pub node_id: u64,
78 /// The binary the re-exec'd compute workers run as — the container backend's
79 /// `__sandbox` jailer and the microVM backends' `__vmm-run`/`__vz-run` VM hosts.
80 /// `None` uses this process's own executable (`current_exe`), which is what
81 /// `boatramp serve` wants (the child *is* boatramp). An **embedding harness**
82 /// whose own binary doesn't implement those subcommands should point this at a
83 /// built `boatramp` binary, so it can drive the real container/microVM backends
84 /// in-process (only the per-workload worker re-execs; the serving plane stays
85 /// embedded). The docker backend needs neither — it talks to a daemon.
86 pub worker_exe: Option<std::path::PathBuf>,
87}
88
89/// A fully wired node: the deploy store, handler runtime, auth, and options a
90/// transport consumes, plus the detached reconcile loops kept alive for the
91/// node's serving life. Destructure it and hold `reconcile` across the serve
92/// await so the loops outlive assembly.
93pub struct RunningNode {
94 /// The deploy store (blob + KV) the router serves from.
95 pub deploy: DeployStore,
96 /// The handler runtime for wasm handlers (a disabled build ⇒ a no-op runtime).
97 pub handlers: boatramp_server::HandlerRuntime,
98 /// The control-plane auth.
99 pub auth: boatramp_server::Auth,
100 /// The resolved server options.
101 pub options: boatramp_server::ServerOptions,
102 /// The detached reconcile loops (compute + domain-verify). Tokio `JoinHandle`s
103 /// do not abort on drop, so the loops run for the process life regardless; the
104 /// handles are retained so an embedder can join/abort them on shutdown.
105 pub reconcile: Vec<tokio::task::JoinHandle<()>>,
106}
107
108/// The instance's own serve socket(s) a guest self-call may reach, given the bind `addr` and
109/// whether the posture (`allow_guest_self_egress`) permits it. A wildcard bind
110/// (`0.0.0.0`/`::`) is reachable over loopback, so it normalizes to `127.0.0.1` **and** `::1`
111/// on the serve port; a specific bind is reachable at itself. Empty when disabled or no
112/// listener.
113fn self_egress_addrs(
114 addr: Option<std::net::SocketAddr>,
115 enabled: bool,
116) -> Vec<std::net::SocketAddr> {
117 use std::net::{IpAddr, Ipv4Addr, Ipv6Addr, SocketAddr};
118 let Some(addr) = addr.filter(|_| enabled) else {
119 return Vec::new();
120 };
121 if addr.ip().is_unspecified() {
122 let port = addr.port();
123 vec![
124 SocketAddr::new(IpAddr::V4(Ipv4Addr::LOCALHOST), port),
125 SocketAddr::new(IpAddr::V6(Ipv6Addr::LOCALHOST), port),
126 ]
127 } else {
128 vec![addr]
129 }
130}
131
132/// Whether a node with **no configured control-plane issuer** should auto-provision an ephemeral
133/// in-memory fleet signer (for host session cookies + delegable capabilities) from its serve bind.
134///
135/// The fleet signer is a distinct trust domain from control-plane admin auth, so a DEV / loopback
136/// node with auth disabled should still be able to sign/verify session cookies and capabilities. It
137/// is gated **strictly to a loopback bind** (`127.0.0.1`/`::1`) or an in-process embedder with no
138/// bind address (`None`): NEVER a public or wildcard (`0.0.0.0`/`::`, reachable off-host) bind,
139/// where an ephemeral key would silently invalidate live capabilities across a restart — a
140/// public node that wants guest capabilities without control-plane auth must supply a persistent
141/// key. (`is_loopback()` is already `false` for a wildcard/unspecified address, so a `0.0.0.0` bind
142/// correctly does NOT auto-provision.)
143#[cfg(feature = "handlers")]
144fn should_autoprovision_fleet_signer(serve_addr: Option<std::net::SocketAddr>) -> bool {
145 serve_addr.is_none_or(|a| a.ip().is_loopback())
146}
147
148/// Wire [`NodeInput`] into a [`RunningNode`]: build the handler runtime, the
149/// deploy store (materializing the reserved `default` project), the compute
150/// backends + reconcile loop, and the domain-verify reconcile loop.
151///
152/// The caller has already built the store and configured auth/OIDC on `options`;
153/// this is the pure node-graph wiring, identical to what `boatramp serve` runs.
154pub async fn assemble(input: NodeInput<'_>) -> Result<RunningNode> {
155 let NodeInput {
156 config,
157 data_dir,
158 storage,
159 kv,
160 auth,
161 options,
162 serve_addr,
163 watch_provider,
164 provision_tier,
165 messaging,
166 is_leader,
167 node_id,
168 worker_exe,
169 } = input;
170 // Copy out the posture scalars up front so `options` can be moved into the
171 // returned `RunningNode` without a lingering borrow.
172 let max_handler_blob_bytes = options.posture.max_handler_blob_bytes;
173 let max_component_bytes = options.posture.max_component_bytes;
174 let allow_guest_private_egress = options.posture.allow_guest_private_egress;
175 let allow_env_secret_refs = options.posture.allow_env_secret_refs;
176 let allow_guest_email = options.posture.allow_guest_email;
177 // The instance's own serve socket(s) a guest self-call may reach, when the posture allows
178 // it: a wildcard bind (`0.0.0.0`/`::`) is reachable on loopback, so normalize to
179 // `127.0.0.1`/`::1`; a specific bind is itself.
180 let self_egress_addrs = self_egress_addrs(serve_addr, options.posture.allow_guest_self_egress);
181 let allow_shared_kernel = options.posture.allow_shared_kernel_compute;
182 let domain_verify_allow_private = options.posture.domain_verify_allow_private;
183
184 // The deploy store the router serves from — built up front so the handler
185 // runtime's managed compute-backed `sql` binding can resolve DB endpoints from
186 // the same store the reconcile writes.
187 let compute_storage = storage.clone();
188 let deploy = DeployStore::new(storage, kv.clone());
189 // **Crown-jewel frontier-sync (v0.9.0 KV-recovery, C2).** The sealed-secret / sealed-credential
190 // stores below hold IRREPLACEABLE data (a dropped write = a lost secret with no client copy), so
191 // rather than rely on per-call-site discipline we route them ALL through a `CheckpointKv`: every
192 // write advances the durable frontier before it acks, so a crash's self-heal-on-open trailing-
193 // tail quarantine can never drop one. Wrapping ONCE here is the can't-miss guarantee. (On the
194 // cluster `RaftKv` / `MemoryKv` the added checkpoint is a no-op, so this is inert there.) The
195 // mixed `DeployStore` above keeps the RAW `kv` and selects the frontier-sync per method instead
196 // (its `current/*` pointers, history, invocation records and metering stay async).
197 let secrets_kv: Arc<dyn KvStore> = Arc::new(boatramp_core::kv::CheckpointKv::new(kv.clone()));
198 // The `[secrets]` envelope (local KEK / Vault) that seals a managed SQL
199 // credential at rest. `None` ⇒ no wrapping (a managed DB then fails closed).
200 let secrets_envelope = build_secrets_envelope(config.secrets.as_ref(), data_dir)?;
201 // The project-scoped internal secret store, built from the same KV + `[secrets]`
202 // envelope that seal managed-DB credentials. Backs both the `boatramp:<name>`
203 // resolver (wired into the handler runtime below, when that feature is present)
204 // and the admin secrets API (threaded into `ServerOptions` unconditionally, so it
205 // works on a lean node too). `None` when no envelope is configured — the admin
206 // endpoints then fail closed with a clear 501, never a panic.
207 let secret_store = secrets_envelope.clone().map(|envelope| {
208 Arc::new(boatramp_core::secret_store::SecretStore::new(
209 secrets_kv.clone(),
210 envelope,
211 ))
212 });
213 // The per-TENANT sealed secret store (task #493), sealed with the SAME `[secrets]` envelope.
214 // Backs both the control-plane CRUD (threaded into `ServerOptions` below) and — when the
215 // `tenant-secrets` feature is compiled — the runtime guest binding (the SAME `Arc` handed to
216 // `set_tenant_secret_store`). `None` when no envelope is configured, so the endpoints fail
217 // closed with a clear 501 and the guest binding is not built.
218 let tenant_secret_store = secrets_envelope.clone().map(|envelope| {
219 Arc::new(boatramp_core::secret_store::TenantSecretStore::new(
220 secrets_kv.clone(),
221 envelope,
222 ))
223 });
224 // The project-scoped SMTP email-profile store, built from the same KV + envelope
225 // (the password is sealed at rest). Backs the admin API (`options` below,
226 // unconditionally, so it works on a lean node) and — when the `email` feature +
227 // `allow_guest_email` posture permit — the runtime's host-side profile
228 // resolution (wired inside `build_handler_runtime`). `None` with no envelope, so
229 // the admin email endpoints fail closed with a clear 501.
230 let email_profile_store = secrets_envelope.clone().map(|envelope| {
231 Arc::new(boatramp_core::email_config::EmailProfileStore::new(
232 secrets_kv.clone(),
233 envelope,
234 ))
235 });
236
237 // Dev-posture guest-egress extra CA(s): when the posture permits (off/refused under
238 // multi-tenant) AND the operator pointed `BOATRAMP_GUEST_EGRESS_EXTRA_CA_FILE` at a PEM, parse
239 // it into trust anchors the guest's outbound `wasi:http` TLS client trusts on TOP of the webpki
240 // roots (for a hermetic HTTPS test double). Empty otherwise. A configured-but-unreadable/
241 // unparsable file is a hard config error (fail closed), never a silent no-trust.
242 let guest_egress_extra_roots =
243 load_guest_egress_extra_roots(options.posture.allow_guest_egress_extra_ca)?;
244
245 // The handler runtime reuses the same blob/KV backends (per-site prefixed)
246 // for its wasi:blobstore/keyvalue bindings; the sql binding is selected by
247 // `[handlers.bindings.sql]` (default: per-site libsql files under <data-dir>).
248 let handlers = crate::handlers::build_handler_runtime(
249 kv.clone(),
250 compute_storage.clone(),
251 data_dir,
252 config.handlers.as_ref(),
253 messaging,
254 max_handler_blob_bytes,
255 max_component_bytes,
256 allow_guest_private_egress,
257 self_egress_addrs,
258 guest_egress_extra_roots,
259 allow_env_secret_refs,
260 allow_guest_email,
261 options.posture.require_tenancy_declaration,
262 options.posture.allow_cross_tenant_db,
263 &deploy,
264 secrets_envelope.clone(),
265 )
266 .await?;
267 // JWKS pre-warm (federated-gateway overhead fix): fetch the configured first-party (`own`-tier)
268 // JWKS up front so even the FIRST request per first-party issuer is instant, not a cold fetch.
269 // Best-effort + spawned (a slow/unreachable IdP must never delay node readiness); the token
270 // cache's single-flight lock dedups a pre-warm fetch against a concurrent first request.
271 // `prewarm_own_jwks` internally honors the `[handlers] jwks_prewarm` flag (no-op when off).
272 // Sites deployed AFTER startup simply warm lazily on first request (today's behavior).
273 #[cfg(feature = "handlers")]
274 {
275 let prewarm_deploy = deploy.clone();
276 tokio::spawn(async move {
277 let entries = collect_own_jwks_prewarm(&prewarm_deploy).await;
278 if !entries.is_empty() {
279 boatramp_server::prewarm_own_jwks(&entries).await;
280 }
281 });
282 }
283 // Hand the runtime the SAME per-tenant secret store `Arc` the control-plane routes hold (task
284 // #493), so a guest `tenant-secrets` `get` and a control-plane `PUT` seal/unseal against ONE
285 // store. Unset when no `[secrets]` envelope, so the guest binding is not built (fail-closed).
286 // Handlers-gated: `set_tenant_secret_store` lives on the handler runtime, so a lean (no-handlers)
287 // node has no guest binding to wire (the control-plane routes still work via `ServerOptions`).
288 #[cfg(feature = "handlers")]
289 if let Some(store) = tenant_secret_store.clone() {
290 handlers.set_tenant_secret_store(store);
291 }
292 // Record the node data directory so the node-health endpoint can report `/data` filesystem
293 // headroom (construens node-health-alerting — the disk-full incident signal).
294 #[cfg(feature = "handlers")]
295 handlers.set_data_dir(data_dir.to_path_buf());
296 // Wire the fleet session-cookie signer (R3, PLAN-tenancy-principal): the same issuer that mints
297 // control-plane tokens signs + verifies the host-issued anonymous session cookie AND the
298 // delegable capabilities (PLAN-delegable-capabilities). Handlers-gated: the session-cookie
299 // machinery lives on the handler runtime, so a lean (no-handlers) build has nothing to wire.
300 //
301 // The fleet signer is a DIFFERENT trust domain from control-plane admin auth (signing a
302 // customer's session cookie / an embed capability is not the authority to admit an operator to
303 // the control plane), but production derives it from the control-plane issuer for convenience.
304 // For a DEV / loopback node with control-plane auth disabled (`options.issuer` is `None`),
305 // auto-provision an EPHEMERAL in-memory Ed25519 fleet key so the guest-facing signer just works
306 // — session cookies + capability mint/verify — WITHOUT turning on control-plane auth. Strictly
307 // gated to a loopback bind (or an in-process embedder with no bind address): never on a public
308 // bind, where an ephemeral key would silently invalidate live capabilities across a restart (a
309 // public node that wants guest capabilities without control-plane auth must supply a persistent
310 // key). Ephemeral = issue + verify within one process run; nothing persisted, no cross-process
311 // or cross-deploy trust. Production is byte-identical: a real deploy supplies a control-plane key
312 // ⇒ `issuer` is `Some` ⇒ this fallback is never taken.
313 #[cfg(feature = "handlers")]
314 {
315 let fleet_signer = options.issuer.clone().or_else(|| {
316 should_autoprovision_fleet_signer(serve_addr).then(|| {
317 tracing::warn!(
318 "control-plane auth is disabled and no signer is configured; auto-provisioning \
319 an EPHEMERAL in-memory fleet signer (Ed25519) for host session cookies + \
320 delegable capabilities on this loopback/dev node — regenerated each start, \
321 never persisted. Configure a control-plane key (or a dedicated signer) for \
322 production."
323 );
324 Arc::new(boatramp_core::cose::LocalSigner::generate(
325 boatramp_core::cose::TokenAlg::Ed25519,
326 )) as Arc<dyn boatramp_core::cose::Signer>
327 })
328 });
329 if let Some(issuer) = fleet_signer {
330 handlers.set_session_signer(issuer);
331 }
332 }
333 // Enable guest capability minting (`boatramp:handlers/capability`, PLAN-delegable-capabilities)
334 // when the operator posture allows it. A minted capability is verified against the same fleet
335 // signer as the session cookie (wired just above), so this only enables the mint path + the TTL
336 // ceiling; posture-off (or a zero ceiling) ⇒ not offered (a guest `mint` is access-denied).
337 #[cfg(feature = "capability")]
338 if options.posture.allow_guest_mint_capability {
339 handlers.set_capability_minting(options.posture.max_guest_capability_ttl_secs);
340 }
341 // Per-project tenancy/capability posture overrides (Gap 4a): resolve each
342 // `[security.projects.<p>]` override against the fleet base so one serve process can run a
343 // strict-isolation project beside a looser one on a shared, multi-project machine. Empty ⇒
344 // every project uses the node base wired just above. Only these four in-project knobs are
345 // per-project; cross-project isolation stays structural (project = database).
346 #[cfg(feature = "handlers")]
347 {
348 let base = &options.posture;
349 let overrides: std::collections::BTreeMap<
350 String,
351 boatramp_core::security::ResolvedProjectTenancy,
352 > = config
353 .security
354 .as_ref()
355 .map(|s| {
356 s.projects
357 .iter()
358 .map(|(project, ovr)| (project.clone(), base.project_tenancy(ovr)))
359 .collect()
360 })
361 .unwrap_or_default();
362 handlers.set_project_tenancy_overrides(overrides);
363 }
364 // Wire the guest project self-config capability (`boatramp:handlers/admin`) when the
365 // operator posture enables at least one surface. The controller reuses the same in-process
366 // domain-verify / email-profile / secret / site-config subsystems + the real domain probe;
367 // it's project-scoped per grant and rate-limited + audited. Posture-off ⇒ not offered.
368 #[cfg(feature = "admin")]
369 {
370 use boatramp_handlers::AdminSurface;
371 let p = &options.posture;
372 let mut surfaces = std::collections::BTreeSet::new();
373 if p.allow_guest_admin_domains {
374 surfaces.insert(AdminSurface::Domains);
375 }
376 if p.allow_guest_admin_email {
377 surfaces.insert(AdminSurface::Email);
378 }
379 if p.allow_guest_admin_site {
380 surfaces.insert(AdminSurface::Site);
381 }
382 if p.allow_guest_admin_secrets {
383 surfaces.insert(AdminSurface::Secrets);
384 }
385 if !surfaces.is_empty() {
386 let controller = Arc::new(boatramp_server::ServerAdminController::with_server_probe(
387 deploy.clone(),
388 email_profile_store.clone(),
389 secret_store.clone(),
390 p.domain_verify_allow_private,
391 ));
392 handlers.set_admin(controller, surfaces);
393 }
394 }
395 // Leader-gate cron firing (cluster: only the Raft leader fires; single-node: an
396 // always-true gate, equivalent to the unset default). The same gate drives the
397 // reconcile loops below, so all three converge on one leader per fleet. Only the
398 // handler runtime has a scheduler, so this is a no-op without the `handlers` feature.
399 #[cfg(feature = "handlers")]
400 handlers.set_cron_leader_gate(is_leader.clone());
401 // FA-5b2: on a cloud backend, wire the blob-change notification provisioner +
402 // its tier so adding a `blob` trigger provisions (and removing it retracts).
403 #[cfg(feature = "handlers")]
404 if let Some(provider) = watch_provider {
405 handlers.set_watch_provider(provider);
406 handlers.set_provision_tier(provision_tier);
407 }
408 #[cfg(not(feature = "handlers"))]
409 let _ = (watch_provider, provision_tier);
410
411 // Materialize the reserved `default` project so `project ls` / `project show
412 // default` reflect it on a fresh install, not only after a migration. Best
413 // effort: the reader backstop keeps listings correct even if this write can't
414 // land, so a transient failure must never block serving.
415 match deploy.ensure_default_project().await {
416 Ok(true) => tracing::info!("materialized the reserved `default` project record"),
417 Ok(false) => {}
418 Err(e) => tracing::warn!(
419 error = %e,
420 "could not materialize the `default` project record; readers use the synthesized default"
421 ),
422 }
423 // Wire the function-to-function invoke resolver now the deploy store exists,
424 // so a function granted `invoke` can call a sibling in-process (FI).
425 #[cfg(feature = "handlers")]
426 handlers.set_invoker(deploy.clone());
427
428 // Compute reconcile loop. Single-node is always the "leader". Backends are
429 // built from the `[compute]` config + capability detection; a no-op when none
430 // are registered. Detached for the server's life.
431 let (compute_backends, compute_node) = crate::compute::build_compute(
432 config.compute.as_ref(),
433 compute_storage,
434 data_dir,
435 node_id,
436 !allow_shared_kernel,
437 options.daemon_runtime.clone(),
438 worker_exe.as_deref(),
439 )
440 .await;
441 // Adopt the IPs of already-running replicas into each backend's fresh-on-boot
442 // IP pool BEFORE the reconcile loop starts allocating. A backend with a per-node
443 // pool (the native container backend) rebuilds it empty each process start; without
444 // this the boot reconcile could re-hand a live address to a different workload —
445 // the container-IP collision — or move a replica's endpoint on relaunch. Feeds
446 // every persisted replica's `(workload, replica, endpoint-ip)`; each backend keeps
447 // only the IPs in its own subnet (a cheap no-op for docker/cloudflare/VMM).
448 crate::compute::adopt_running_replica_ips(&deploy, &compute_backends).await;
449 // Per-project internal DNS (service discovery): start the resolver on the bridge
450 // gateway so a guest resolves peers by name within its project. On by default;
451 // starts only when the container backend + bridge are up (Linux). Detached for
452 // the node's serving life (pushed into `reconcile` below). Started before the
453 // reconcile loop consumes `compute_backends` — it borrows the registry to check
454 // the container backend is present.
455 let internal_dns =
456 crate::compute::spawn_internal_dns(config.compute.as_ref(), &compute_backends, &deploy);
457 // Activate the compute sql-shim (PLAN-compute-bindings): bind its listener +
458 // build the resolver when a sql provider and `compute.sql_shim_url` are both present.
459 #[cfg(feature = "handlers")]
460 let sql_resolver = boatramp_server::sql_shim::spawn_sql_shim(
461 handlers.sql_backends(),
462 config.compute.as_ref().and_then(|c| c.sql_shim_url.clone()),
463 )
464 .await;
465 #[cfg(not(feature = "handlers"))]
466 let sql_resolver: Option<Arc<dyn boatramp_core::compute::ComputeBindingResolver>> = None;
467
468 // Managed compute-backed SQL (PLAN-managed-compute-sql P2-b): if the handler
469 // `sql` config declares any managed database, inject its `POSTGRES_*`/`MYSQL_*`
470 // server-init env into the DB workload at launch from the sealed credential.
471 // Reaching here with a managed DB implies an envelope (build_handler_runtime
472 // fails closed otherwise), so the credential store always has one to seal with.
473 // Keep a clone of the secrets envelope for the operator-SQL capability below
474 // (the managed_db_resolver match moves the original).
475 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
476 let operator_envelope = secrets_envelope.clone();
477 // …and a second clone for the tenant-deprovision capability (drops a deleted
478 // tenant's managed DB/role/credential on project/site delete). It needs a real
479 // envelope to seal/unseal + delete per-tenant credentials, so it is wired only
480 // when one is present (same fail-closed gating as the managed-DB paths).
481 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
482 let deprovision_envelope = secrets_envelope.clone();
483 // …and a clone for the provisioning drift-repair capability (owner-model retrofit /
484 // reconcile). Like the migrate path it connects as the sealed owner role and re-seals
485 // credentials, so it needs a real envelope; wired only when one is present.
486 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
487 let repair_envelope = secrets_envelope.clone();
488 // …and a third clone for the soft-delete tombstone reaper (the leader-gated task
489 // that hard-drops a Shared-Postgres tenant once its grace window elapses). It, too,
490 // needs a real envelope to unseal the superuser credential + delete the per-tenant
491 // one on hard-drop.
492 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
493 let reaper_envelope = secrets_envelope.clone();
494 // …and a clone for the declarative managed-database capability (#501 Stage B): it
495 // provisions the declared DB + seals its credential server-side, so it needs a real
496 // envelope; wired only when one is present (fail-closed like every managed-DB path).
497 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
498 let declare_envelope = secrets_envelope.clone();
499 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
500 let managed_db_resolver: Option<Arc<dyn boatramp_core::compute::ManagedDbEnvResolver>> = match (
501 config
502 .handlers
503 .as_ref()
504 .and_then(|h| h.bindings.sql.as_ref()),
505 secrets_envelope,
506 ) {
507 (Some(sql), Some(envelope)) if !sql.databases.is_empty() => {
508 // Crown-jewel (C2): sealed managed-DB credentials go through the frontier-syncing
509 // `secrets_kv`, so a freshly-minted DB credential is past the durable frontier before
510 // the workload sees it — never droppable by the self-heal-on-open trailing-tail quarantine.
511 let creds =
512 crate::managed_sql::ManagedSqlCredentials::new(secrets_kv.clone(), envelope);
513 let privilege = config
514 .compute
515 .as_ref()
516 .map(|c| c.managed_db_privilege)
517 .unwrap_or_default();
518 let env =
519 crate::managed_sql::ManagedDbEnv::from_config(&sql.databases, creds, privilege);
520 (!env.is_empty()).then(|| Arc::new(env) as Arc<_>)
521 }
522 _ => None,
523 };
524 #[cfg(not(any(feature = "sql-postgres", feature = "sql-mysql")))]
525 let managed_db_resolver: Option<Arc<dyn boatramp_core::compute::ManagedDbEnvResolver>> = None;
526
527 // Turnkey managed DB: auto-register the compute workload(s) backing each managed
528 // co-located database that has none yet, so declaring the `databases` binding is
529 // enough to boot the DB (no separate `compute set` / apply). Tenant-aware — a
530 // `Shared` binding registers its one shared server; a `Single` binding registers
531 // nothing at boot (its per-tenant `<compute>-<ident>` is created durably by the lazy
532 // resolve on first `sql` use and relaunched by the reconcile, so a project that never
533 // uses `sql` — e.g. a static-only site — never gets a spurious DB). Non-clobbering +
534 // idempotent; runs before the reconcile loop so its first tick can launch what it
535 // registered.
536 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
537 if let Some(sql) = config
538 .handlers
539 .as_ref()
540 .and_then(|h| h.bindings.sql.as_ref())
541 .filter(|sql| !sql.databases.is_empty())
542 {
543 crate::managed_sql::auto_register_managed_db_workloads(&deploy, &sql.databases).await;
544 }
545
546 // Operator SQL capability (managed-DB migrations/queries via the sealed credential, resolved
547 // server-side) — backs `POST /api/sql/{db}/{exec,query}`. The SAME concrete NodeOperatorSql
548 // also backs the owner-gated schema-migration runner (which reuses its owner + superuser
549 // backends), so build it ONCE and share it.
550 // Build the SAME concrete NodeOperatorSql once (when a managed DB is configured) and share it:
551 // it backs both `operator_sql` (the sql exec/query cap) and the migration runner (which reuses
552 // its owner + superuser backends). Two separate bindings so neither annotation is a complex type.
553 // The migration substrate is a single dispatcher (crate::managed_sql::DispatchMigrationRunner)
554 // routing each `(project, db)` to its engine's substrate: the sqlx NodeMigrationRunner for a
555 // Postgres/MySQL binding, the LibsqlMigrationRunner for a `libsql` binding. It compiles whenever a
556 // sqlx engine OR `migrate` (⇒ libsql) is on, so the embedded-libsql default can migrate even on a
557 // node with no external sqlx engine. `operator_sql` (the `POST /api/sql/{db}/{exec,query}` cap)
558 // stays sqlx-only — a libsql file has no operator-SQL/credential seam.
559 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
560 let operator_sql: Option<Arc<dyn boatramp_core::sql::OperatorSql>>;
561 // Late-init: the match arms assign it (and, under sqlx, `operator_sql` in the same block), and the
562 // arms differ by feature-cfg — so a direct `let … = match {…}` would need cfg'd arm bodies. The
563 // late-init keeps that readable; the value is always assigned before use.
564 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql", feature = "migrate"))]
565 #[allow(clippy::needless_late_init)]
566 let migration_substrate: Option<Arc<dyn boatramp_core::sql::MigrationSubstrate>>;
567 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql", feature = "migrate"))]
568 match config
569 .handlers
570 .as_ref()
571 .and_then(|h| h.bindings.sql.as_ref())
572 .filter(|sql| !sql.databases.is_empty())
573 {
574 Some(sql) => {
575 // The sqlx (Postgres/MySQL) arm — the shared NodeOperatorSql backs both the operator-SQL
576 // cap and the sqlx migration runner. Only built when a sqlx engine is compiled in.
577 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
578 let node_op = Arc::new(crate::managed_sql::NodeOperatorSql::new(
579 sql.databases.clone(),
580 kv.clone(),
581 operator_envelope,
582 deploy.clone(),
583 ));
584 // The operator's trusted-extension allowlist — the only extensions a migration may
585 // enable (empty ⇒ none). See ExternalSqlConfig::migrate_trusted_extensions.
586 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
587 let trusted: std::collections::BTreeSet<String> = sql
588 .migrate_trusted_extensions
589 .clone()
590 .unwrap_or_default()
591 .into_iter()
592 .collect();
593 migration_substrate = Some(Arc::new(crate::managed_sql::DispatchMigrationRunner::new(
594 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
595 node_op.clone(),
596 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
597 trusted,
598 #[cfg(feature = "migrate")]
599 sql.databases.clone(),
600 ))
601 as Arc<dyn boatramp_core::sql::MigrationSubstrate>);
602 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
603 {
604 operator_sql = Some(node_op as Arc<dyn boatramp_core::sql::OperatorSql>);
605 }
606 }
607 None => {
608 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
609 {
610 operator_sql = None;
611 }
612 migration_substrate = None;
613 }
614 }
615 // When only `migrate` (no sqlx) is compiled, the operator-SQL cap does not exist.
616 #[cfg(all(
617 not(any(feature = "sql-postgres", feature = "sql-mysql")),
618 feature = "migrate"
619 ))]
620 let operator_sql: Option<Arc<dyn boatramp_core::sql::OperatorSql>> = None;
621 #[cfg(not(any(feature = "sql-postgres", feature = "sql-mysql", feature = "migrate")))]
622 let migration_substrate: Option<Arc<dyn boatramp_core::sql::MigrationSubstrate>> = None;
623 #[cfg(not(any(feature = "sql-postgres", feature = "sql-mysql", feature = "migrate")))]
624 let operator_sql: Option<Arc<dyn boatramp_core::sql::OperatorSql>> = None;
625
626 // Tenant-deprovision capability (drop a deleted tenant's managed DB/role/sealed
627 // credential on project/site delete). Wired only when a compute-backed managed
628 // database + a secrets envelope are both present — same gating as operator_sql,
629 // plus the envelope requirement (it must seal/unseal per-tenant credentials).
630 // The soft-delete grace window for a Shared-Postgres managed tenant
631 // (`handlers.bindings.sql.deprovision_grace_secs`, env-settable). Default 7 days;
632 // `0` disables the soft path (immediate hard drop). Threaded to the deprovisioner
633 // (which soft-deletes) and implicitly honored by the reaper (which only ever finds
634 // tombstones a >0 grace produced).
635 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
636 let deprovision_grace_secs = config
637 .handlers
638 .as_ref()
639 .and_then(|h| h.bindings.sql.as_ref())
640 .and_then(|sql| sql.deprovision_grace_secs)
641 .unwrap_or(crate::tenant_sql::DEFAULT_DEPROVISION_GRACE_SECS);
642 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
643 let tenant_deprovisioner: Option<Arc<dyn boatramp_core::sql::TenantDeprovisioner>> = config
644 .handlers
645 .as_ref()
646 .and_then(|h| h.bindings.sql.as_ref())
647 .filter(|sql| !sql.databases.is_empty())
648 .zip(deprovision_envelope)
649 .map(|(sql, envelope)| {
650 Arc::new(crate::tenant_sql::NodeTenantDeprovisioner::new(
651 deploy.clone(),
652 kv.clone(),
653 envelope,
654 sql.databases.clone(),
655 deprovision_grace_secs,
656 )) as Arc<_>
657 });
658 #[cfg(not(any(feature = "sql-postgres", feature = "sql-mysql")))]
659 let tenant_deprovisioner: Option<Arc<dyn boatramp_core::sql::TenantDeprovisioner>> = None;
660
661 // Provisioning drift-repair capability (owner-model retrofit / reconcile) — backs the
662 // `Project·Admin`-gated `/api/repair/{db}` + `/dry-run`. Same gating as operator_sql plus
663 // the envelope requirement (it re-seals the owner credential + connects as it). The
664 // envelope is threaded as `Some(_)` so a lean-but-managed node still gets a clear
665 // per-check error rather than a panic if none is configured.
666 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
667 let tenant_repair: Option<Arc<dyn boatramp_core::sql::TenantRepair>> = config
668 .handlers
669 .as_ref()
670 .and_then(|h| h.bindings.sql.as_ref())
671 .filter(|sql| !sql.databases.is_empty())
672 .map(|sql| {
673 Arc::new(crate::repair::NodeTenantRepair::new(
674 sql.databases.clone(),
675 deploy.clone(),
676 kv.clone(),
677 repair_envelope.clone(),
678 )) as Arc<_>
679 });
680 #[cfg(not(any(feature = "sql-postgres", feature = "sql-mysql")))]
681 let tenant_repair: Option<Arc<dyn boatramp_core::sql::TenantRepair>> = None;
682
683 // Declarative managed-database capability (#501 Stage B) — backs the `Project·Admin`-gated
684 // `PUT /api/projects/{proj}/databases/{name}` + `POST …/ensure`. It persists a manifest
685 // `databases:` entry to `project/{project}/database/{name}`, enforces daemon-config-wins at
686 // the merge point (against the node-static `sql.databases`), refuses an identity change,
687 // binds provisioning to the caller's project, and eagerly provisions via `provision_tenant`.
688 // Requires a sqlx engine (it provisions a Postgres/MySQL server) AND a `[secrets]` envelope
689 // (a managed DB seals its credential). UNLIKE repair/operator_sql it is NOT gated on
690 // `!sql.databases.is_empty()` — a node with NO node-static databases can still accept a
691 // project declaration (the whole point of the declarative front door). The node-static map
692 // (possibly empty) is threaded so the daemon-wins conflict check has both sources.
693 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
694 let managed_db_declare: Option<Arc<dyn boatramp_core::sql::ManagedDbDeclare>> =
695 declare_envelope.map(|envelope| {
696 let sql_cfg = config
697 .handlers
698 .as_ref()
699 .and_then(|h| h.bindings.sql.as_ref());
700 let static_dbs = sql_cfg.map(|sql| sql.databases.clone()).unwrap_or_default();
701 // Per-project declaration ceilings (#501 Stage B MEDIUM-1 — the disk-exhaustion
702 // guard): the operator's `max_declared_databases`/`max_declared_volume_mib`, or
703 // the built-in `apply_db_caps` defaults (16 DBs / 512 GiB) when unset.
704 let quota = crate::managed_db_declare::DeclareQuota::from_config(
705 sql_cfg.and_then(|sql| sql.max_declared_databases),
706 sql_cfg.and_then(|sql| sql.max_declared_volume_mib),
707 );
708 Arc::new(crate::managed_db_declare::NodeManagedDbDeclare::new(
709 static_dbs,
710 deploy.clone(),
711 kv.clone(),
712 envelope,
713 quota,
714 )) as Arc<_>
715 });
716 #[cfg(not(any(feature = "sql-postgres", feature = "sql-mysql")))]
717 let managed_db_declare: Option<Arc<dyn boatramp_core::sql::ManagedDbDeclare>> = None;
718
719 // Operator compute-exec capability (run a command inside a running workload) —
720 // backs `POST /api/compute/{name}/exec`, gated by the `allow_compute_exec`
721 // posture. Clone the backend registry before the reconcile loop consumes it.
722 let compute_exec: Option<Arc<dyn boatramp_core::compute::ComputeExec>> = Some(Arc::new(
723 crate::compute::NodeComputeExec::new(compute_backends.clone(), deploy.clone()),
724 ) as Arc<_>);
725
726 // Operator volume-reclamation capability (list + remove persistent volumes) —
727 // backs `GET /api/compute/volumes` + `DELETE /api/compute/volumes/{name}`.
728 // Same admin-scoped `/api/compute/*` gate; clone the registry before the
729 // reconcile loop consumes the original below.
730 let compute_volumes: Option<Arc<dyn boatramp_core::compute::ComputeVolumes>> = Some(Arc::new(
731 crate::compute::NodeComputeVolumes::new(compute_backends.clone(), deploy.clone()),
732 )
733 as Arc<_>);
734
735 // Operator reconcile-plane control capability (restart a replica) — backs
736 // `POST /api/compute/maintenance/restart` (admin-scoped). Clone the registry
737 // before the reconcile loop consumes the original below.
738 let compute_control: Option<Arc<dyn boatramp_core::compute::ComputeControl>> = Some(Arc::new(
739 crate::compute::NodeComputeControl::new(compute_backends.clone(), deploy.clone()),
740 )
741 as Arc<_>);
742
743 let compute_reconcile = boatramp_server::spawn_compute_reconcile(
744 deploy.clone(),
745 compute_backends,
746 vec![compute_node],
747 boatramp_core::compute::BackendPolicy::from_shared_kernel_allowed(allow_shared_kernel),
748 is_leader.clone(),
749 compute_reconcile_tick(),
750 COMPUTE_IDLE_TIMEOUT,
751 sql_resolver,
752 managed_db_resolver,
753 );
754
755 // Tenant tombstone reaper: leader-gated hard-drop of soft-deleted Shared-Postgres
756 // tenants past their grace window (safe deprovision — see `tenant_sql`). Wired only
757 // when a compute-backed managed database + a secrets envelope are both present
758 // (same gating as the deprovisioner); each tombstone carries its own server +
759 // superuser, so the reaper needs no per-binding config. A `0` grace never writes a
760 // tombstone, so the sweep is simply inert then.
761 #[cfg(any(feature = "sql-postgres", feature = "sql-mysql"))]
762 let tombstone_reaper: Option<tokio::task::JoinHandle<()>> = config
763 .handlers
764 .as_ref()
765 .and_then(|h| h.bindings.sql.as_ref())
766 .filter(|sql| !sql.databases.is_empty())
767 .zip(reaper_envelope)
768 .map(|(_sql, envelope)| {
769 crate::tenant_sql::spawn_tenant_tombstone_reaper(
770 deploy.clone(),
771 kv.clone(),
772 envelope,
773 is_leader.clone(),
774 crate::tenant_sql::TOMBSTONE_REAPER_TICK,
775 )
776 });
777 #[cfg(not(any(feature = "sql-postgres", feature = "sql-mysql")))]
778 let tombstone_reaper: Option<tokio::task::JoinHandle<()>> = None;
779
780 // Domain-verify auto-complete: periodically re-check every site's pending
781 // ownership challenges and attach any that now pass — a published token (e.g.
782 // via `domain add --provider`) converges without a manual `domain verify`.
783 let dv_reconcile = boatramp_server::spawn_domain_verify_reconcile(
784 deploy.clone(),
785 domain_verify_allow_private,
786 is_leader,
787 DOMAIN_VERIFY_RECONCILE_TICK,
788 );
789
790 // Wire the operator capabilities onto the options the router is built from.
791 let mut options = options;
792 options.operator_sql = operator_sql;
793 options.migration_substrate = migration_substrate;
794 options.tenant_repair = tenant_repair;
795 options.managed_db_declare = managed_db_declare;
796 options.tenant_deprovisioner = tenant_deprovisioner;
797 options.compute_exec = compute_exec;
798 options.compute_volumes = compute_volumes;
799 options.compute_control = compute_control;
800 // The internal secret store backs the admin secrets API (set/list/delete). Not
801 // handlers-gated — it must be reachable even on a lean node.
802 options.secret_store = secret_store;
803 // The per-tenant sealed secret store backs the control-plane CRUD (task #493). Like the secret
804 // store it is not handlers-gated, so the endpoints work on a lean node.
805 options.tenant_secret_store = tenant_secret_store;
806 // The email-profile store backs the admin API (`/api/email/profiles`); like the
807 // secret store it is not handlers-gated, so it works on a lean node.
808 options.email_profile_store = email_profile_store;
809
810 // The detached reconcile loops: the always-present compute + domain-verify ones,
811 // plus the optional tenant-tombstone reaper (only when a managed DB is configured).
812 let mut reconcile = vec![compute_reconcile, dv_reconcile];
813 if let Some(reaper) = tombstone_reaper {
814 reconcile.push(reaper);
815 }
816 if let Some(dns) = internal_dns {
817 reconcile.push(dns);
818 }
819
820 Ok(RunningNode {
821 deploy,
822 handlers,
823 auth,
824 options,
825 reconcile,
826 })
827}
828
829/// Build the `[secrets]` envelope (secrets-at-rest wrapping) from `boatramp.cfg`'s
830/// `[secrets]` section: `local` (a machine-local AES-256-GCM KEK) or `vault` (Vault
831/// Env var an operator points at a PEM file of extra CA(s) the guest's outbound `wasi:http` TLS
832/// client should trust on top of the webpki roots — honored only under the
833/// `allow_guest_egress_extra_ca` posture (a hermetic HTTPS test double lever).
834const GUEST_EGRESS_EXTRA_CA_ENV: &str = "BOATRAMP_GUEST_EGRESS_EXTRA_CA_FILE";
835
836/// Parse the operator's guest-egress extra-CA PEM ([`GUEST_EGRESS_EXTRA_CA_ENV`]) into rustls trust
837/// Collect the first-party (`own`-tier) JWKS URLs to pre-warm at startup: every site's gateway-level
838/// `[handlers.graphql.data].claims_from_token` that is single-issuer (`jwks_url` set) and NOT
839/// multi-issuer (`issuer_trust` unset) — foreign/multi-issuer issuers can't be pre-warmed (unknown
840/// until a token arrives), and `jwks_env` is an env var, not a fetch. Returns deduped
841/// `(url, issuer, audience)` tuples. Best-effort: a per-site read error is logged and skipped, never
842/// propagated (pre-warm must not fail node startup). Per-route `HandlerConfig`/consumer/session/
843/// function `token_claims` live in content-addressed deployment manifests not enumerated here; those
844/// warm lazily on first request (today's behavior).
845#[cfg(feature = "handlers")]
846async fn collect_own_jwks_prewarm(deploy: &DeployStore) -> Vec<(String, String, Option<String>)> {
847 use boatramp_core::project::ProjectRef;
848 let sites = match deploy.list_sites_all().await {
849 Ok(s) => s,
850 Err(e) => {
851 tracing::warn!(error = %e, "JWKS pre-warm: could not list sites; skipping pre-warm");
852 return Vec::new();
853 }
854 };
855 let mut seen = std::collections::HashSet::new();
856 let mut out = Vec::new();
857 for (project, site) in sites {
858 let sc = match deploy
859 .get_site_config(ProjectRef::new(&project), &site)
860 .await
861 {
862 Ok(Some(sc)) => sc,
863 Ok(None) => continue,
864 Err(e) => {
865 tracing::warn!(project = %project, site = %site, error = %e,
866 "JWKS pre-warm: could not load site config; skipping");
867 continue;
868 }
869 };
870 let Some(claims) = sc
871 .handlers
872 .as_ref()
873 .and_then(|h| h.graphql.as_ref())
874 .and_then(|g| g.data.as_ref())
875 .and_then(|d| d.claims_from_token.as_ref())
876 else {
877 continue;
878 };
879 // OWN tier only: operator-fixed single-issuer `jwks_url`, never multi-issuer discovery.
880 if claims.issuer_trust.is_some() {
881 continue;
882 }
883 if let Some(url) = claims.jwks_url.as_ref() {
884 let entry = (url.clone(), claims.issuer.clone(), claims.audience.clone());
885 if seen.insert(entry.clone()) {
886 out.push(entry);
887 }
888 }
889 }
890 out
891}
892
893/// anchors, gated by the `allow_guest_egress_extra_ca` posture (`allow`). No env set ⇒ empty (the
894/// default). `allow == false` (e.g. multi-tenant) with a file set ⇒ empty + a warning (the posture
895/// refuses it). Set + readable + ≥1 cert ⇒ those certs. Set-but-unreadable / no valid cert ⇒ a hard
896/// error (fail closed — a configured-but-broken CA must not silently degrade to no-trust).
897fn load_guest_egress_extra_roots(
898 allow: bool,
899) -> Result<Vec<rustls::pki_types::CertificateDer<'static>>> {
900 let path = match std::env::var(GUEST_EGRESS_EXTRA_CA_ENV) {
901 Ok(p) if !p.is_empty() => p,
902 _ => return Ok(Vec::new()),
903 };
904 if !allow {
905 tracing::warn!(
906 env = GUEST_EGRESS_EXTRA_CA_ENV,
907 "ignoring a guest-egress extra CA: the security posture forbids it \
908 (allow_guest_egress_extra_ca is off — e.g. under multi-tenant)"
909 );
910 return Ok(Vec::new());
911 }
912 let pem =
913 std::fs::read(&path).map_err(|e| Error::GuestEgressCa(format!("reading {path:?}: {e}")))?;
914 let certs = rustls_pemfile::certs(&mut std::io::BufReader::new(&pem[..]))
915 .collect::<std::result::Result<Vec<_>, _>>()
916 .map_err(|e| Error::GuestEgressCa(format!("parsing {path:?}: {e}")))?;
917 if certs.is_empty() {
918 return Err(Error::GuestEgressCa(format!(
919 "{path:?} contained no PEM certificate"
920 )));
921 }
922 tracing::info!(
923 env = GUEST_EGRESS_EXTRA_CA_ENV,
924 count = certs.len(),
925 path = %path,
926 "guest egress trusts operator extra CA(s) (dev-posture; webpki roots still apply)"
927 );
928 Ok(certs)
929}
930
931/// Transit). `None`/empty ⇒ no wrapping. The Vault token is read from the
932/// environment (`token_env`), never a file. This seals a managed SQL credential at
933/// rest; a managed database fails closed without it.
934fn build_secrets_envelope(
935 secrets: Option<&crate::config::SecretsConfig>,
936 data_dir: &Path,
937) -> Result<Option<Arc<dyn boatramp_core::envelope::KeyEnvelope>>> {
938 use boatramp_server::envelope::{EnvelopeSpec, build_envelope};
939 let Some(cfg) = secrets else {
940 return Ok(None);
941 };
942 let spec = match cfg.envelope.as_str() {
943 "" => EnvelopeSpec::None,
944 "local" => EnvelopeSpec::Local {
945 kek_file: cfg
946 .kek_file
947 .clone()
948 .unwrap_or_else(|| data_dir.join("secrets/kek")),
949 },
950 "vault" => {
951 let v = cfg.vault.as_ref().ok_or_else(|| {
952 Error::Envelope(
953 "secrets.envelope = \"vault\" needs a [secrets.vault] section".into(),
954 )
955 })?;
956 let token = std::env::var(&v.token_env).map_err(|_| {
957 Error::Envelope(format!("Vault token env `{}` is not set", v.token_env))
958 })?;
959 EnvelopeSpec::Vault {
960 addr: v.addr.clone(),
961 key: v.key.clone(),
962 token,
963 }
964 }
965 other => {
966 return Err(Error::Envelope(format!(
967 "unknown secrets.envelope {other:?} (want \"local\" or \"vault\")"
968 )));
969 }
970 };
971 build_envelope(spec).map_err(|e| Error::Envelope(e.to_string()))
972}
973
974#[cfg(all(test, feature = "fs"))]
975mod tests {
976 use super::*;
977 use boatramp_core::kv::MemoryKv;
978 use boatramp_core::security::SecurityProfile;
979
980 /// The dev/loopback ephemeral fleet-signer auto-provision is gated STRICTLY to a loopback bind
981 /// (or an in-process embedder with no bind): a public / wildcard / private-network bind must
982 /// NOT silently provision an ephemeral signing key (it would invalidate live capabilities on a
983 /// restart — such a node must supply a persistent key). This is the security-critical boundary.
984 #[cfg(feature = "handlers")]
985 #[test]
986 fn ephemeral_fleet_signer_auto_provisions_only_on_loopback_or_in_process() {
987 use std::net::SocketAddr;
988 let sa = |s: &str| s.parse::<SocketAddr>().unwrap();
989 // In-process (no listener) and loopback → auto-provision the dev fleet signer.
990 assert!(should_autoprovision_fleet_signer(None));
991 assert!(should_autoprovision_fleet_signer(Some(sa(
992 "127.0.0.1:8080"
993 ))));
994 assert!(should_autoprovision_fleet_signer(Some(sa("[::1]:8080"))));
995 // Off-host-reachable binds MUST NOT auto-provision an ephemeral key: a public IP, a
996 // private-network IP, and a wildcard bind (`0.0.0.0`/`::`, reachable off-host).
997 assert!(!should_autoprovision_fleet_signer(Some(sa(
998 "203.0.113.5:8080"
999 ))));
1000 assert!(!should_autoprovision_fleet_signer(Some(sa(
1001 "10.0.0.4:8080"
1002 ))));
1003 assert!(!should_autoprovision_fleet_signer(Some(sa("0.0.0.0:8080"))));
1004 assert!(!should_autoprovision_fleet_signer(Some(sa("[::]:8080"))));
1005 }
1006
1007 /// The headline in-process fidelity check (PLAN-node-library N2b.3): `assemble`
1008 /// over a temp `FsStorage` + `MemoryKv` produces a `RunningNode` whose deploy
1009 /// store is live (the reserved `default` project was materialized during
1010 /// assembly) and whose router — the exact one `boatramp serve` builds — answers
1011 /// `/healthz`. No listener is bound: the request is driven through the router
1012 /// via `tower::oneshot`, so the whole assembly runs in-process.
1013 #[tokio::test]
1014 async fn assemble_produces_a_serving_node_over_a_temp_store() {
1015 use axum::body::Body;
1016 use axum::http::{Request, StatusCode};
1017 use tower::ServiceExt;
1018
1019 let tmp = tempfile::tempdir().unwrap();
1020 let storage: Arc<dyn Storage> = Arc::new(boatramp_storage::FsStorage::new(tmp.path()));
1021 let kv: Arc<dyn KvStore> = Arc::new(MemoryKv::new());
1022 let config = ServerConfig::default();
1023 let options = boatramp_server::ServerOptions {
1024 // The strict `multi-tenant` posture, as an unconfigured `serve` resolves.
1025 posture: SecurityProfile::MultiTenant.preset(),
1026 ..Default::default()
1027 };
1028
1029 let node = assemble(NodeInput {
1030 config: &config,
1031 data_dir: tmp.path(),
1032 storage,
1033 kv,
1034 auth: boatramp_server::Auth::disabled(),
1035 options,
1036 serve_addr: None,
1037 watch_provider: None,
1038 provision_tier: boatramp_core::blob_notify::ProvisionTier::default(),
1039 messaging: None,
1040 is_leader: Arc::new(|| true),
1041 node_id: 0,
1042 worker_exe: None,
1043 })
1044 .await
1045 .expect("assemble a node over a temp store");
1046
1047 // The deploy store is live: `assemble` already materialized the reserved
1048 // `default` project, so a second ensure reports "already present" (`false`).
1049 assert!(
1050 !node
1051 .deploy
1052 .ensure_default_project()
1053 .await
1054 .expect("read the default project"),
1055 "assemble should have materialized the default project"
1056 );
1057
1058 // The assembled router (the same wiring `serve` binds) answers /healthz.
1059 let router =
1060 boatramp_server::router_with(node.deploy, node.auth, node.handlers, node.options);
1061 let response = router
1062 .oneshot(
1063 Request::builder()
1064 .uri("/healthz")
1065 .body(Body::empty())
1066 .unwrap(),
1067 )
1068 .await
1069 .expect("route /healthz");
1070 assert_eq!(response.status(), StatusCode::OK);
1071 }
1072}