acme_proxy/cli/mod.rs
1//! The command tree, and the startup path itself.
2//!
3//! **Nothing here prints or exits.** Every command body returns
4//! `Result<(), CliError>` and [`dispatch`] routes to it, so each arm is a plain
5//! function a test can call and assert on rather than an unreachable dead end.
6//! `src/main.rs` is where that `Result` becomes an exit status, and it is the
7//! only place in the project that calls `std::process::exit` — a library whose
8//! failure mode is ending the process is one nothing else can use.
9//!
10//! Startup is split on the socket boundary, which is what lets a test drive the
11//! whole path on an ephemeral port with its own shutdown future instead of a
12//! process signal:
13//!
14//! - `serve` binds `server.bind_address`, installs the `SIGHUP` handler, and
15//! hands the socket on.
16//! - [`serve_on`] validates the admin configuration and binds that socket too,
17//! when `[admin]` is enabled.
18//! - [`serve_on_with`] does everything else — profile resolution, deduplicated
19//! signer backends, per-profile filters and validators, TLS, the job registry
20//! (every signer's handlers, notification delivery and the four table sweeps),
21//! the runner draining it, and `axum::serve` with connect info attached.
22//!
23//! That assembly is [`build_generation`], and it is called again on every
24//! reload rather than only at startup — so the two cannot drift, and a
25//! subsystem added to one is added to the other by construction. What a reload
26//! may change, and what it refuses by name, is [`crate::reload`]'s to say;
27//! [`serve_on_with_reloads`] is where the two meet.
28//!
29//! The logic behind each admin subcommand lives in [`crate::admin`], not here;
30//! this module is the `clap` surface over it. [`logging`] turns `[logging]` into
31//! an installed subscriber, validating every value before installing anything.
32//!
33//! What a command *prints* is [`render`]'s, and how it is coloured is
34//! [`style`]'s. Those renderings sit here rather than in [`crate::admin`]
35//! because they have exactly one consumer — the terminal — where the JSON ones
36//! beside them are a wire format the web admin parses too. [`dispatch`]
37//! resolves one [`Palette`] and threads it down; `nonce` and `upstream` take
38//! none, printing only fixed text.
39
40use std::future::Future;
41use std::io::{BufRead, IsTerminal};
42use std::net::SocketAddr;
43use std::sync::Arc;
44use std::time::Duration;
45
46use clap::{Parser, Subcommand};
47use clap_complete::aot::Shell;
48use tokio::net::TcpListener;
49use tracing::{error, info, warn};
50
51pub mod account;
52pub mod audit;
53pub mod eab;
54pub mod filter;
55pub mod generate;
56mod logging;
57
58/// Installs the `[logging]` configuration. Re-exported because `main.rs` is
59/// what calls it — see [`dispatch`].
60pub use logging::init_logging;
61pub mod nonce;
62pub mod order;
63pub mod render;
64pub mod style;
65pub mod upstream;
66pub mod webadmin;
67pub mod window;
68
69pub use account::AccountCommand;
70pub use audit::AuditCommand;
71pub use eab::EabCommand;
72pub use nonce::NonceCommand;
73pub use order::OrderCommand;
74pub use upstream::UpstreamCommand;
75pub use webadmin::AdminCommand;
76
77use crate::cli::filter::FilterCommand;
78pub use crate::cli::style::{ColorChoice, Palette};
79use crate::config::Config;
80use crate::sqlite::db::Database;
81use crate::{Profile, build_app, tls};
82
83#[derive(Parser)]
84#[command(
85 name = "acme-proxy",
86 version = env!("CARGO_PKG_VERSION"),
87 about = "ACME server, plus admin commands for its database"
88)]
89pub struct Cli {
90 /// Skip interactive "Are you sure?" confirmation on destructive commands.
91 #[arg(short = 'y', long, global = true)]
92 pub yes: bool,
93
94 /// When to colour human-readable output. `--json` output never carries it.
95 #[arg(long, value_enum, default_value_t = ColorChoice::Auto, global = true)]
96 pub color: ColorChoice,
97
98 #[command(subcommand)]
99 pub command: Option<Command>,
100}
101
102#[derive(Subcommand)]
103pub enum Command {
104 /// Run the ACME HTTP(S) server. Default when no subcommand is given.
105 Serve,
106 /// Inspect and manage ACME accounts.
107 Account {
108 #[command(subcommand)]
109 command: AccountCommand,
110 },
111 /// Inspect and manage ACME orders.
112 Order {
113 #[command(subcommand)]
114 command: OrderCommand,
115 },
116 /// Read and prune the CA's audit trail.
117 Audit {
118 #[command(subcommand)]
119 command: AuditCommand,
120 },
121 /// Nonce table maintenance.
122 Nonce {
123 #[command(subcommand)]
124 command: NonceCommand,
125 },
126 /// Manage External Account Binding (EAB) credentials.
127 Eab {
128 #[command(subcommand)]
129 command: EabCommand,
130 },
131 /// Read and test the access policy of an endpoint.
132 Filter {
133 #[command(subcommand)]
134 command: FilterCommand,
135 },
136 /// Manage this server's own account at the upstream ACME server
137 /// (`signer.backend = "relay"`).
138 Upstream {
139 #[command(subcommand)]
140 command: UpstreamCommand,
141 },
142 /// Manage the web admin's operators and their sessions. This is how the
143 /// panel is bootstrapped: it has no sign-up page.
144 Admin {
145 #[command(subcommand)]
146 command: AdminCommand,
147 },
148 /// Print a shell completion script on stdout.
149 Completions {
150 /// The shell to generate for.
151 #[arg(value_enum)]
152 shell: Shell,
153 },
154 /// Print this binary's man page, in roff, on stdout.
155 Man,
156}
157
158/// Picks the profile a command acts on.
159///
160/// `--profile` is optional only when the configuration defines exactly one:
161/// most per-profile sections would otherwise be acted on ambiguously, and
162/// guessing is worse than asking. Shared by `upstream` and `filter`, which had
163/// grown one copy each.
164pub(crate) fn resolve_profile(
165 config: &Config,
166 wanted: Option<&str>,
167) -> Result<crate::config::ProfileConfig, CliError> {
168 let profiles = config
169 .resolve_profiles()
170 .map_err(|error| CliError(format!("configuration error: {error}")))?;
171
172 match wanted {
173 Some(name) => profiles
174 .into_iter()
175 .find(|profile| profile.name == name)
176 .ok_or_else(|| CliError(format!("no profile named `{name}` in this configuration"))),
177 None if profiles.len() == 1 => Ok(profiles.into_iter().next().expect("length checked")),
178 None => {
179 let names: Vec<&str> = profiles.iter().map(|p| p.name.as_str()).collect();
180 Err(CliError(format!(
181 "this configuration defines several profiles ({}); say which one with --profile",
182 names.join(", ")
183 )))
184 }
185 }
186}
187
188/// A command that could not complete, carrying the message to print.
189///
190/// Every failing branch below returns one of these instead of calling
191/// `std::process::exit` where it stands: `main.rs` is the single place that
192/// prints and exits, so each command body stays a plain function a test can
193/// call and assert on.
194#[derive(Debug, PartialEq, Eq, thiserror::Error)]
195#[error("{0}")]
196pub struct CliError(pub String);
197
198impl From<sqlx::Error> for CliError {
199 fn from(error: sqlx::Error) -> Self {
200 Self(format!("database error: {error}"))
201 }
202}
203
204/// Routes a parsed command to its handler.
205///
206/// The library's entry point. Everything above it — parsing argv, loading the
207/// configuration, installing the subscriber, opening the database, printing a
208/// failure and exiting — lives in `src/main.rs`, because those are the
209/// binary's job and not a library's: nothing that links this crate can use a
210/// function whose failure mode is `std::process::exit`.
211///
212/// Takes the `--color` *choice* rather than a resolved [`Palette`], and
213/// resolves it here: the answer depends on whether this process's stdout is a
214/// terminal and on `NO_COLOR`, and neither belongs in `main.rs`, which is
215/// excluded from the coverage floor precisely because nothing in it is
216/// reachable from a test.
217pub async fn dispatch(
218 command: Option<Command>,
219 yes: bool,
220 color: ColorChoice,
221 reader: &mut impl BufRead,
222 config: &Arc<Config>,
223 database: Arc<Database>,
224) -> Result<(), CliError> {
225 let palette = Palette::resolve(
226 color,
227 std::io::stdout().is_terminal(),
228 std::env::var("NO_COLOR").ok().as_deref(),
229 );
230 match command.unwrap_or(Command::Serve) {
231 Command::Serve => serve(config.clone(), database).await,
232 Command::Account { command } => {
233 account::run_account_command(command, yes, palette, reader, config, database).await
234 }
235 Command::Order { command } => {
236 order::run_order_command(command, yes, palette, reader, config, database).await
237 }
238 Command::Audit { command } => {
239 audit::run_audit_command(command, yes, palette, reader, database).await
240 }
241 Command::Nonce { command } => {
242 nonce::run_nonce_command(command, yes, reader, config, database).await
243 }
244 Command::Eab { command } => eab::run_eab_command(command, palette, database).await,
245 Command::Filter { command } => filter::run_filter_command(command, palette, config).await,
246 Command::Upstream { command } => {
247 upstream::run_upstream_command(command, reader, config).await
248 }
249 Command::Admin { command } => {
250 webadmin::run_admin_command(command, yes, palette, reader, config, database).await
251 }
252 // Reachable here, though `main.rs` answers both before it opens
253 // anything: an `unreachable!()` would be dead code under the coverage
254 // floor, and routing them keeps this a total function over `Command`.
255 command @ (Command::Completions { .. } | Command::Man) => {
256 generate::write(&command, &mut std::io::stdout().lock())
257 }
258 }
259}
260
261/// Binds the configured socket and runs the ACME HTTP(S) server until a
262/// shutdown signal arrives.
263pub async fn serve(config: Arc<Config>, database: Arc<Database>) -> Result<(), CliError> {
264 let listener = TcpListener::bind(&config.server.bind_address)
265 .await
266 .map_err(|error| {
267 error!(event = "server_socket_bind_failed", outcome = "failure", bind_address = %config.server.bind_address, error = %error);
268 CliError(format!(
269 "cannot bind {}: {error}",
270 config.server.bind_address
271 ))
272 })?;
273
274 // Installed here, before anything slow: `SIGHUP`'s default disposition is
275 // *terminate*, so until the handler exists a reload signal kills the
276 // process. `serve_on_with_reloads` does profile assembly and the relay's
277 // first upstream contact before it binds anything, which is exactly the
278 // window an operator's `systemctl reload` could land in.
279 let (reload_handle, reloads) = crate::reload::channel();
280 let _hangups = AbortOnDrop(tokio::spawn(watch_for_hangup(reload_handle)));
281
282 let admin_listener = bind_admin(&config).await.map_err(|error| {
283 error!(event = "server_fatal_error", outcome = "failure", error = %error);
284 CliError(error.to_string())
285 })?;
286 let metrics_listener = bind_metrics(&config).await.map_err(|error| {
287 error!(event = "server_fatal_error", outcome = "failure", error = %error);
288 CliError(error.to_string())
289 })?;
290
291 serve_on_with_reloads(
292 config,
293 database,
294 listener,
295 admin_listener,
296 metrics_listener,
297 shutdown_signal(),
298 reloads,
299 )
300 .await
301 .map_err(|error| {
302 error!(event = "server_fatal_error", outcome = "failure", error = %error);
303 CliError(error.to_string())
304 })
305}
306
307/// Turns every `SIGHUP` into a reload request, for the life of the process.
308///
309/// Unlike the shutdown signal, this one does not consume its stream: an operator
310/// reloads repeatedly, and a handler that fired once would leave the second
311/// `SIGHUP` back at its default disposition — killing the server.
312#[cfg(unix)]
313async fn watch_for_hangup(handle: crate::reload::ReloadHandle) {
314 let mut hangups = match tokio::signal::unix::signal(tokio::signal::unix::SignalKind::hangup()) {
315 Ok(stream) => stream,
316 Err(error) => {
317 error!(event = "server_signal_handler_failed", outcome = "failure", signal = "SIGHUP", error = %error);
318 return;
319 }
320 };
321 while hangups.recv().await.is_some() {
322 handle.trigger();
323 }
324}
325
326/// No `SIGHUP` off Unix, so there is nothing to watch for.
327#[cfg(not(unix))]
328async fn watch_for_hangup(_handle: crate::reload::ReloadHandle) {
329 std::future::pending::<()>().await;
330}
331
332/// Validates the admin configuration and binds its socket when it is enabled.
333///
334/// Shared by [`serve`] and [`serve_on`] rather than living in one of them: both
335/// need it, and the validation must happen **before anything binds**, so a
336/// misconfigured panel cannot take the ACME listener down with it halfway
337/// through startup.
338async fn bind_admin(config: &Arc<Config>) -> anyhow::Result<Option<TcpListener>> {
339 crate::webadmin::check_config(config).inspect_err(|error| {
340 error!(event = "admin_config_invalid", outcome = "failure", error = %error);
341 })?;
342
343 match config.admin.enabled {
344 false => Ok(None),
345 true => Ok(Some(
346 TcpListener::bind(&config.admin.bind_address)
347 .await
348 .inspect_err(|error| {
349 error!(event = "admin_socket_bind_failed",
350 outcome = "failure",
351 bind_address = %config.admin.bind_address,
352 error = %error);
353 })?,
354 )),
355 }
356}
357
358/// Refuses a `[metrics]` bind address that collides with another listener's.
359///
360/// Pure, so a reload runs the same check before rebinding anything — the twin of
361/// [`crate::webadmin::check_config`], and beside it in `apply_reload` for the
362/// same reason: a listener configuration that would not start must not be one a
363/// running server can be reloaded into.
364///
365/// Three listeners, so the check is pairwise. `webadmin` already refuses
366/// admin-versus-server; these are the two pairs it cannot see. Checked even when
367/// `[admin]` is off, since enabling the panel later must not be what surfaces a
368/// latent conflict.
369///
370/// # Errors
371///
372/// Names the other key when the two addresses are equal.
373pub fn check_metrics_config(config: &Config) -> anyhow::Result<()> {
374 if !config.metrics.enabled {
375 return Ok(());
376 }
377
378 let bind = &config.metrics.bind_address;
379 for (name, other) in [
380 ("server.bind_address", &config.server.bind_address),
381 ("admin.bind_address", &config.admin.bind_address),
382 ] {
383 if bind == other && (name != "admin.bind_address" || config.admin.enabled) {
384 error!(event = "metrics_config_invalid",
385 outcome = "failure",
386 bind_address = %bind);
387 anyhow::bail!(
388 "metrics.bind_address and {name} are both `{bind}`: the metrics endpoint is a \
389 separate listener and cannot share a socket (give it its own port)"
390 );
391 }
392 }
393 Ok(())
394}
395
396/// Binds the metrics socket when `[metrics]` is on, refusing a collision first.
397///
398/// The twin of [`bind_admin`], and the same shape for the same reason: a socket
399/// that cannot be bound must stop startup rather than leave the process running
400/// with one of its three listeners silently missing.
401///
402/// Deliberately **no loopback check**. `webadmin::check_config` refuses a
403/// non-loopback admin bind without TLS because that listener's cookie is always
404/// `Secure`, which a browser will not store over plain HTTP, so the failure
405/// would be invisible. Nothing here has a cookie: a metrics port reachable from
406/// a Prometheus host on another machine is the intended deployment, and the
407/// firewall is what bounds it.
408async fn bind_metrics(config: &Arc<Config>) -> anyhow::Result<Option<TcpListener>> {
409 check_metrics_config(config)?;
410 if !config.metrics.enabled {
411 return Ok(None);
412 }
413
414 let bind = &config.metrics.bind_address;
415 let listener = TcpListener::bind(bind).await.inspect_err(|error| {
416 error!(event = "metrics_socket_bind_failed",
417 outcome = "failure",
418 bind_address = %bind,
419 error = %error);
420 })?;
421 Ok(Some(listener))
422}
423
424/// Assembles and serves the application over an already-bound socket.
425///
426/// Split from [`serve`] on the socket boundary: a caller supplying its own
427/// listener and its own `shutdown` future can drive the whole startup path —
428/// profile assembly, TLS, backend resume, the nonce reaper, both `axum::serve`
429/// arms — without owning a fixed port or a process signal.
430///
431/// Binds the web admin socket itself when `[admin]` is enabled. The signature
432/// is unchanged, and `admin.enabled` is false by default, so every existing
433/// caller is untouched; a test that wants to drive *both* listeners supplies
434/// its own pair through [`serve_on_with`].
435pub async fn serve_on(
436 config: Arc<Config>,
437 database: Arc<Database>,
438 listener: TcpListener,
439 shutdown: impl Future<Output = ()> + Send + 'static,
440) -> anyhow::Result<()> {
441 let admin_listener = bind_admin(&config).await?;
442 let metrics_listener = bind_metrics(&config).await?;
443 serve_on_with(
444 config,
445 database,
446 listener,
447 admin_listener,
448 metrics_listener,
449 shutdown,
450 )
451 .await
452}
453
454/// [`serve_on`] with all three sockets supplied.
455///
456/// The full version, split on the same boundary and for the same reason: a
457/// caller handing in three ephemeral ports can drive the whole startup path —
458/// including that one shutdown signal stops all of them — without owning a
459/// fixed port or a process signal.
460pub async fn serve_on_with(
461 config: Arc<Config>,
462 database: Arc<Database>,
463 listener: TcpListener,
464 admin_listener: Option<TcpListener>,
465 metrics_listener: Option<TcpListener>,
466 shutdown: impl Future<Output = ()> + Send + 'static,
467) -> anyhow::Result<()> {
468 serve_on_with_reloads(
469 config,
470 database,
471 listener,
472 admin_listener,
473 metrics_listener,
474 shutdown,
475 crate::reload::Reloads::none(),
476 )
477 .await
478}
479
480/// [`serve_on_with`], serving configuration reloads as well as requests.
481///
482/// The variant `serve` uses, so a `SIGHUP` rebuilds both routers, the job
483/// registry, the notifier map and both TLS acceptors behind the sockets that are
484/// already bound. Every other caller goes through [`serve_on_with`] and gets a
485/// source that never fires, which costs one task that ends immediately.
486///
487/// See [`crate::reload`] for what a reload may change and what it refuses.
488pub async fn serve_on_with_reloads(
489 config: Arc<Config>,
490 database: Arc<Database>,
491 listener: TcpListener,
492 admin_listener: Option<TcpListener>,
493 metrics_listener: Option<TcpListener>,
494 shutdown: impl Future<Output = ()> + Send + 'static,
495 reloads: crate::reload::Reloads,
496) -> anyhow::Result<()> {
497 info!(
498 event = "server_startup",
499 outcome = "success",
500 bind_address = %config.server.bind_address,
501 base_url = %config.server.base_url,
502 tls = config.server.tls.enabled,
503 database_database_url = %config.database.url
504 );
505
506 // One `shutdown` future, several consumers: both listeners and the job
507 // runner. Created here rather than beside `axum::serve` below so a signal
508 // arriving *during* startup is not ignored — profile assembly and the
509 // relay's first upstream contact both happen before anything binds. The
510 // relay task is held under `AbortOnDrop` so an error path below does not
511 // leak a task parked on a signal that will never arrive.
512 let (shutdown_tx, shutdown_rx) = tokio::sync::watch::channel(false);
513 let _shutdown_relay = AbortOnDrop(tokio::spawn(async move {
514 shutdown.await;
515 let _ = shutdown_tx.send(true);
516 }));
517
518 // The enqueue side of the durable queue. Built before the profiles because
519 // a signer backend that defers issuance is handed one at construction, and
520 // process-wide for the reason `[audit]` is: one table, one runner, and a
521 // per-endpoint retry budget would make a job's pacing depend on which
522 // profile happened to queue it.
523 let job_queue = crate::jobs::JobQueue::new(database.clone(), &config.jobs);
524
525 let resolved = config.resolve_profiles().inspect_err(|error| {
526 error!(event = "profile_init_failed", outcome = "failure", error = %error);
527 })?;
528 // Everything that outlives a configuration generation — the signer backends
529 // above all, which are carried rather than rebuilt. See `crate::Assembly`.
530 let (assembly, parts) =
531 crate::Assembly::new(&resolved, database.clone(), job_queue.clone(), &config).inspect_err(
532 |error| {
533 error!(event = "profile_init_failed", outcome = "failure", error = %error);
534 },
535 )?;
536 let assembly = Arc::new(assembly);
537
538 let generation =
539 build_generation(&config, &resolved, &assembly, &parts, None).inspect_err(|error| {
540 error!(event = "profile_init_failed", outcome = "failure", error = %error);
541 })?;
542
543 // Every endpoint in the first generation came up, so every one is announced.
544 // A reload announces only the endpoints it *added* — see
545 // `supervise_reloads`, which shares this function for exactly that reason.
546 for profile in &generation.profiles {
547 announce_profile(profile).await;
548 }
549
550 let Generation {
551 profiles: _,
552 acme_app,
553 admin_app,
554 job_registry,
555 tls,
556 admin_tls,
557 logins,
558 } = generation;
559
560 // The one task that drains the queue. Held under `AbortOnDrop` for the same
561 // reason the reapers are — an un-cancelled loop holding an `Arc<Database>`
562 // per `serve_on` call — and, unlike the relay tasks this replaced, it does
563 // *not* need to outlive this function: the work is durable now, so a job cut
564 // short is re-claimed from its own row rather than lost. It takes the same
565 // shutdown signal as both listeners, so a stop is graceful rather than an
566 // abort: it releases its leases on the way out, and a restart therefore
567 // re-claims its own work immediately instead of waiting one out.
568 //
569 // Neither the registry it drains nor the `[jobs]` section it paces itself
570 // from is a value: both are cells a reload republishes, so a changed
571 // retention, a rebuilt notify map and a retuned lease or concurrency all
572 // reach the runner without restarting it. `jobs.max_attempts` is the third
573 // piece and does not come through here — it belongs to the enqueue side, so
574 // it is published onto `job_queue` itself.
575 let (registry_tx, registry_rx) = tokio::sync::watch::channel(Arc::new(job_registry));
576 let (jobs_tx, jobs_rx) = tokio::sync::watch::channel(Arc::new(config.jobs.clone()));
577 let _job_runner = AbortOnDrop(crate::jobs::spawn_runner_watching(
578 job_queue,
579 registry_rx,
580 jobs_rx,
581 shutdown_rx.clone(),
582 ));
583
584 info!(
585 event = "server_listening",
586 outcome = "success",
587 bind_address = %config.server.bind_address,
588 protocol = if tls.is_some() { "https" } else { "http" }
589 );
590
591 // One accept loop per role, each owning a socket a reload can replace and a
592 // TLS mode it can switch — see `crate::listener`. `axum::serve` below is
593 // handed one of these instead of a `TcpListener` and therefore outlives
594 // every rebind, which is what removes the listener from the list of things
595 // only a restart can change.
596 let admin_bound = bound_address(admin_listener.as_ref(), &config.admin.bind_address);
597 let metrics_bound = bound_address(metrics_listener.as_ref(), &config.metrics.bind_address);
598 let (acme_socket, acme_handle) = crate::listener::spawn("acme", Some(listener), tls);
599 let (admin_socket, admin_handle) = crate::listener::spawn("admin", admin_listener, admin_tls);
600 let (metrics_socket, metrics_handle) =
601 crate::listener::spawn("metrics", metrics_listener, None);
602
603 // Behind a swap cell rather than served directly, so a configuration reload
604 // can replace the whole router without the socket moving. The cell is what
605 // `axum::serve` holds; `acme_app` itself is only ever generation one.
606 let (acme_router_tx, acme_router_rx) = crate::reload::router_channel(acme_app);
607 let acme = serve_role(
608 crate::reload::swappable(acme_router_rx),
609 acme_socket,
610 shutdown_rx.clone(),
611 );
612
613 // Opened whether or not the panel is on, unlike the app inside it: with
614 // `admin.enabled` reloadable, a cell created only when the panel starts
615 // would be the one thing a reload turning it on could not reach. An empty
616 // `Router` answers `404` to everything, which is also what the panel being
617 // switched off later publishes here.
618 let (admin_router_tx, admin_router_rx) =
619 crate::reload::router_channel(admin_app.unwrap_or_default());
620 let admin = serve_role(
621 crate::reload::swappable(admin_router_rx),
622 admin_socket,
623 shutdown_rx.clone(),
624 );
625 if config.admin.enabled {
626 announce_admin_listener(&config, &database, &admin_bound).await;
627 }
628
629 // The third listener. Served directly rather than through a
630 // `reload::router_channel` like the other two, and the asymmetry is
631 // deliberate: this router has one route whose only state is the registry,
632 // and the registry is carried across generations rather than rebuilt (see
633 // `Assembly`), so a new generation could put nothing new in it.
634 if config.metrics.enabled {
635 announce_metrics_listener(&metrics_bound);
636 }
637 let metrics = serve_role(
638 crate::metrics_app(assembly.metrics.clone()),
639 metrics_socket,
640 shutdown_rx,
641 );
642
643 // The supervisor owns every cell sender from here on, which is what makes it
644 // the only writer: a generation is published by one task or by nobody.
645 // Aborted on drop, so an error return below does not leave it parked on a
646 // channel nothing will ever send to.
647 let _reload_supervisor = AbortOnDrop(tokio::spawn(supervise_reloads(
648 reloads,
649 config.clone(),
650 resolved,
651 assembly,
652 Cells {
653 acme_router: acme_router_tx,
654 admin_router: admin_router_tx,
655 job_registry: registry_tx,
656 jobs: jobs_tx,
657 acme: acme_handle,
658 admin: admin_handle,
659 metrics: metrics_handle,
660 },
661 logins,
662 )));
663
664 // Nothing is drained here any more. A notification in flight at shutdown is
665 // a `notify_deliver` row, not a spawned task: the runner released its lease
666 // on the way out and whoever starts next claims it. That is what replaced a
667 // best-effort five-second drain which still lost anything slower than it.
668 tokio::try_join!(acme, admin, metrics)?;
669 Ok(())
670}
671
672/// Serves `app` on one role's socket until the process shuts down.
673///
674/// One shape for all three roles, where there used to be a boxed future per
675/// listener per TLS arm: [`crate::listener::RoleSocket`] is the same type
676/// whether the role is speaking TLS, speaking cleartext or — a socket having
677/// been closed by a reload — not serving at all, so the four cases collapse into
678/// this one call. Its future lives for the process: a rebind replaces what is
679/// underneath it, never the `axum::serve` above.
680fn serve_role(
681 app: axum::Router,
682 socket: crate::listener::RoleSocket,
683 shutdown: tokio::sync::watch::Receiver<bool>,
684) -> impl Future<Output = std::io::Result<()>> + Send {
685 axum::serve(
686 socket,
687 app.into_make_service_with_connect_info::<SocketAddr>(),
688 )
689 .with_graceful_shutdown(on_shutdown(shutdown))
690 .into_future()
691}
692
693/// Everything one configuration generation contributes, built and validated
694/// before any of it is published.
695///
696/// The unit exists so startup and reload cannot drift: both go through
697/// [`build_generation`], so a subsystem added to one is added to the other by
698/// construction rather than by remembering.
699pub(crate) struct Generation {
700 /// Kept so startup can announce them; a reload drops them on the floor,
701 /// since `profile_mounted` is a lifecycle event and not a heartbeat.
702 profiles: Vec<Arc<Profile>>,
703 acme_app: axum::Router,
704 admin_app: Option<axum::Router>,
705 job_registry: crate::jobs::JobRegistry,
706 tls: Option<tls::TlsSettings>,
707 admin_tls: Option<tls::TlsSettings>,
708 /// The limiter this generation ended up with, for the next one to carry.
709 logins: Option<Arc<crate::webadmin::LoginLimiter>>,
710}
711
712/// Builds one generation: profiles, both routers, the job registry, the audit
713/// trail and both TLS acceptors.
714///
715/// Fallible throughout and side-effect-free on the *serving* state: nothing here
716/// touches a cell, so a failure leaves whatever is already running exactly as it
717/// was. That is what makes an atomic reload possible — everything is built
718/// first, and only a complete success publishes anything.
719///
720/// `dispatchers` is passed in rather than built here because a reload needs to
721/// hold it back: it is published to the long-lived [`crate::notify::Notifiers`]
722/// handle at swap time, *before* the routers, so a request served by the new
723/// generation cannot queue a delivery the job runner's map does not know.
724///
725/// Whether the panel is part of this generation is read from `admin.enabled`
726/// rather than from whether a socket exists. It used to be the latter, which
727/// was the same answer while the key was frozen and is the wrong one now that a
728/// reload can turn the panel on: the app, its TLS, its session sweep and its
729/// login limiter all have to appear in the generation *before* there is a
730/// listener to serve them.
731pub(crate) fn build_generation(
732 config: &Arc<Config>,
733 resolved: &[crate::config::ProfileConfig],
734 assembly: &crate::Assembly,
735 parts: &crate::GenerationParts,
736 previous_logins: Option<&crate::webadmin::LoginLimiter>,
737) -> anyhow::Result<Generation> {
738 let admin_enabled = config.admin.enabled;
739 let database = assembly.database.clone();
740 let profiles = Profile::build_all_with(config, resolved, parts)?;
741
742 let tls = tls::from_config(&config.server)
743 .inspect_err(|error| {
744 error!(event = "tls_init_failed", outcome = "failure", error = %error);
745 })?
746 .map(|acceptor| {
747 tls::TlsSettings::new(
748 acceptor,
749 Duration::from_millis(config.server.tls.handshake_timeout_ms),
750 )
751 });
752
753 let admin_tls = match admin_enabled {
754 false => None,
755 true => tls::admin_from_config(&config.admin)
756 .inspect_err(|error| {
757 error!(event = "admin_tls_init_failed", outcome = "failure", error = %error);
758 })?
759 .map(|acceptor| {
760 tls::TlsSettings::new(
761 acceptor,
762 Duration::from_millis(config.admin.tls.handshake_timeout_ms),
763 )
764 }),
765 };
766
767 // Every subsystem with background work, in one registry. Deduplicated by
768 // `Arc` identity: two profiles can share one signer backend, and its handler
769 // must be registered once — the registry refuses a second handler for one
770 // kind outright, since two would each claim about half the rows. The
771 // identity is kept as a `usize` rather than the pointer itself, so this
772 // function's caller stays `Send`; it is spawned.
773 let mut job_registry = crate::jobs::JobRegistry::new();
774 let mut registered: Vec<usize> = Vec::new();
775 let mut pruners: Vec<Arc<dyn crate::signer::CrlPruner>> = Vec::new();
776 for profile in &profiles {
777 let identity = Arc::as_ptr(&profile.signer).cast::<()>() as usize;
778 if registered.contains(&identity) {
779 continue;
780 }
781 registered.push(identity);
782 for handler in profile.signer.jobs() {
783 job_registry.register(handler).inspect_err(|error| {
784 error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
785 })?;
786 }
787 // Collected rather than registered, and that is the whole reason
788 // `crl_pruner` is not simply another entry in `jobs()`: the registry
789 // refuses two handlers for one kind, so two profiles over *different*
790 // local CAs — which the dedup above deliberately does not collapse —
791 // would be a startup error. One handler over every ledger instead.
792 pruners.extend(profile.signer.crl_pruner());
793 }
794 // The periodic CRL prune, over whichever CAs keep a ledger of their own.
795 // Registered only when there is one, the way the audit sweep is registered
796 // only for a non-zero retention.
797 if !pruners.is_empty() {
798 job_registry
799 .register(Arc::new(crate::signer::local_ca::sweep::CrlSweepJob::new(
800 pruners,
801 )))
802 .inspect_err(|error| {
803 error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
804 })?;
805 }
806 // Notification delivery. **Not** deduplicated by `Arc` identity like the
807 // signer handlers above: there is one handler for every profile, holding the
808 // whole `profile name -> dispatcher` map, and a job row names its own
809 // profile. Registered unconditionally — a profile with no `[notify]`
810 // backends queues nothing, so the handler simply never claims a row, and
811 // making the registration conditional would mean a row queued before a
812 // configuration change had nobody to run it.
813 //
814 // It takes the *handle*, not this generation's map: the handler is
815 // registered per generation but must read whichever map is current, and a
816 // row queued by a reloaded router names a slot id only the new one has.
817 job_registry
818 .register(Arc::new(crate::notify::NotifyJob::new(
819 assembly.notifiers.clone(),
820 )))
821 .inspect_err(|error| {
822 error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
823 })?;
824
825 // The expiry digest, registered only when some profile asked for one
826 // (`notify.expiry.lead_days`), the way `CrlSweepJob` is registered only
827 // when there is a ledger to prune. It takes the `Notifiers` handle rather
828 // than the profiles' own dispatchers for `NotifyJob`'s reason above, and
829 // the queue because its per-profile rows are something it maintains on
830 // every pass rather than only at `recover`.
831 if let Some(digest) = crate::notify::expiry::ExpiryDigestJob::from_profiles(
832 resolved,
833 assembly.notifiers.clone(),
834 database.clone(),
835 assembly.jobs.clone(),
836 ) {
837 job_registry
838 .register(Arc::new(digest))
839 .inspect_err(|error| {
840 error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
841 })?;
842 }
843
844 // The periodic table sweeps. Each is one self-rescheduling row rather than
845 // its own interval loop, so a sweep that dies is reclaimed by lease expiry
846 // and its schedule survives a restart. Their `recover` is also the startup
847 // sweep — it queues at `run_at = now`, so the runner performs the first pass
848 // on its way into the loop and there is nothing to run separately here.
849 let ttl = Duration::from_secs(config.nonce.ttl_seconds);
850 let mut sweeps = vec![crate::jobs::SweepJob::nonces(database.clone(), ttl)];
851 // `0` keeps everything for ever on both of these, and is a handler not
852 // registered rather than a sweep with a cutoff at the epoch.
853 if config.audit.retention_days > 0 {
854 sweeps.push(crate::jobs::SweepJob::audit(
855 database.clone(),
856 config.audit.retention_days,
857 ));
858 }
859 if config.jobs.retention_days > 0 {
860 sweeps.push(crate::jobs::SweepJob::jobs(
861 database.clone(),
862 config.jobs.retention_days,
863 ));
864 }
865 // One handler covering every mounted profile, since the registry refuses a
866 // second handler for one kind. A profile keeping everything (`0`) is left
867 // out of the list rather than swept with a cutoff at the epoch.
868 let order_retention: Vec<(String, u64)> = resolved
869 .iter()
870 .filter(|profile| profile.sections.order.retention_days > 0)
871 .map(|profile| (profile.name.clone(), profile.sections.order.retention_days))
872 .collect();
873 if !order_retention.is_empty() {
874 sweeps.push(crate::jobs::SweepJob::orders(
875 database.clone(),
876 order_retention,
877 ));
878 }
879 if admin_enabled {
880 sweeps.push(crate::jobs::SweepJob::admin_sessions(
881 database.clone(),
882 Duration::from_secs(config.admin.session_idle_timeout_seconds),
883 config.admin.session_ttl_seconds,
884 ));
885 }
886 for sweep in sweeps {
887 job_registry
888 .register(Arc::new(sweep))
889 .inspect_err(|error| {
890 error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
891 })?;
892 }
893
894 // The CA's audit trail: one per process, shared by every profile's router
895 // and by the web admin listener, because `[audit]` is process-wide. Built
896 // here rather than in `Profile::build_all` for exactly that reason — it is
897 // not a per-endpoint subsystem.
898 let auditor = Arc::new(
899 crate::audit::Auditor::from_config(
900 &config.audit,
901 &config.dns,
902 database.clone(),
903 // The registry the certificate counters land in. Carried by
904 // `Assembly`, so it is the same one across every generation and the
905 // same one `metrics_app` serves from.
906 assembly.metrics.clone(),
907 )
908 .inspect_err(|error| {
909 error!(event = "audit_init_failed", outcome = "failure", error = %error);
910 })?,
911 );
912
913 // Built **before** `build_app`, which consumes `profiles`. The admin state
914 // needs the same profiles (revoking an order resolves that order's own
915 // signer), and `build_admin_app` takes a slice precisely so the ordering
916 // is a signature constraint rather than a borrow error to rediscover.
917 let (admin_app, logins) = match admin_enabled {
918 false => (None, None),
919 true => {
920 let (router, logins) = crate::webadmin::build_admin_app_with_logins(
921 database.clone(),
922 config.clone(),
923 &profiles,
924 auditor.clone(),
925 previous_logins,
926 );
927 (Some(router), Some(logins))
928 }
929 };
930 let acme_app = build_app(
931 database,
932 config.clone(),
933 profiles.clone(),
934 auditor,
935 assembly.metrics.clone(),
936 );
937
938 Ok(Generation {
939 profiles,
940 acme_app,
941 admin_app,
942 job_registry,
943 tls,
944 admin_tls,
945 logins,
946 })
947}
948
949/// The cells one generation is published into.
950///
951/// Held by the supervisor and by nothing else. Every field is a `watch::Sender`,
952/// and `send_replace` is synchronous — so publishing a generation is a run of
953/// sends with no `.await` between them, which no other task can interleave with.
954/// That is what makes a reload atomic without a lock.
955struct Cells {
956 acme_router: tokio::sync::watch::Sender<axum::routing::RouterIntoService<axum::body::Body>>,
957 admin_router: tokio::sync::watch::Sender<axum::routing::RouterIntoService<axum::body::Body>>,
958 job_registry: tokio::sync::watch::Sender<Arc<crate::jobs::JobRegistry>>,
959 /// The runner's own pacing. Separate from the registry above because the two
960 /// reach it by different routes: the registry carries what a *handler*
961 /// captured, this carries what the *loop* re-reads each pass.
962 jobs: tokio::sync::watch::Sender<Arc<crate::config::JobsConfig>>,
963 /// The three sockets. Each carries its role's TLS mode as well, since both
964 /// are read by the same accept loop and both are published the same
965 /// synchronous way — see [`crate::listener::ListenerHandle`].
966 acme: crate::listener::ListenerHandle,
967 admin: crate::listener::ListenerHandle,
968 metrics: crate::listener::ListenerHandle,
969}
970
971/// Serves reload requests for the life of the process.
972///
973/// One task, so reloads are serialised: two overlapping rebuilds could publish
974/// their cells interleaved, and the second-newest generation would win some of
975/// them. Ends when the last [`crate::reload::ReloadHandle`] is dropped, which is
976/// what makes [`crate::reload::Reloads::none`] cost nothing.
977async fn supervise_reloads(
978 mut reloads: crate::reload::Reloads,
979 mut config: Arc<Config>,
980 mut resolved: Vec<crate::config::ProfileConfig>,
981 assembly: Arc<crate::Assembly>,
982 cells: Cells,
983 mut logins: Option<Arc<crate::webadmin::LoginLimiter>>,
984) {
985 let mut generation: u64 = 1;
986
987 while let Some(request) = reloads.recv().await {
988 let started = std::time::Instant::now();
989 info!(
990 event = "server_config_reload_requested",
991 outcome = "progress",
992 generation = generation,
993 );
994
995 // The build phase runs on a blocking thread, and that is not a
996 // precaution: `RelaySigner::from_config` contacts the upstream the first
997 // time it is built for an account with no `kid` sidecar yet, on a scoped
998 // OS thread it then *joins*. Mounting a relay profile by `SIGHUP` would
999 // otherwise park a runtime worker for as long as
1000 // `signer.relay.poll_timeout_secs` allows — five minutes by default —
1001 // with every connection that worker was polling parked behind it. The
1002 // publish phase stays on this task, where its lack of an await point is
1003 // what makes a generation unobservable half-applied.
1004 let outcome = {
1005 let config = config.clone();
1006 let resolved = resolved.clone();
1007 let assembly = assembly.clone();
1008 let logins = logins.clone();
1009 tokio::task::spawn_blocking(move || {
1010 prepare_reload(&config, &resolved, &assembly, logins.as_deref())
1011 })
1012 .await
1013 .unwrap_or_else(|error| {
1014 Err(crate::reload::ReloadError::Build(format!(
1015 "the reload build task did not finish: {error}"
1016 )))
1017 })
1018 }
1019 .map(|prepared| {
1020 publish_reload(
1021 prepared,
1022 &config,
1023 &assembly,
1024 &cells,
1025 generation + 1,
1026 started,
1027 )
1028 });
1029
1030 match outcome {
1031 Ok(reloaded) => {
1032 let report = reloaded.report;
1033 config = reloaded.config;
1034 resolved = reloaded.resolved;
1035 logins = reloaded.logins;
1036 generation = report.generation;
1037 info!(
1038 event = "server_config_reloaded",
1039 outcome = "success",
1040 generation = report.generation,
1041 profiles = ?report.profiles,
1042 job_kinds = ?report.job_kinds,
1043 tls_reloaded = report.tls_reloaded,
1044 admin_tls_reloaded = report.admin_tls_reloaded,
1045 listeners_rebound = ?report.listeners_rebound,
1046 logging_reloaded = report.logging_reloaded,
1047 duration_ms = crate::millis(report.duration),
1048 );
1049 // After the reload's own line, and under the new configuration,
1050 // since that is what these describe. Each is the same
1051 // announcement startup makes for a listener that has just come
1052 // up — including the panel's two warnings, which is why this is
1053 // here rather than inside the synchronous publishing run.
1054 for (role, address) in reloaded.opened {
1055 match role {
1056 Role::Acme => info!(
1057 event = "server_listening",
1058 outcome = "success",
1059 bind_address = %address,
1060 protocol = if config.server.tls.enabled { "https" } else { "http" }
1061 ),
1062 Role::Admin => {
1063 announce_admin_listener(&config, &assembly.database, &address).await;
1064 }
1065 Role::Metrics => announce_metrics_listener(&address),
1066 }
1067 }
1068 // An endpoint this reload mounted really did come up, so it gets
1069 // the same announcement and the same notification startup makes
1070 // for one. An endpoint that was *already* mounted stays silent:
1071 // `profile_mounted` is a lifecycle event and not a heartbeat,
1072 // and re-firing it per `SIGHUP` would make the notify surface
1073 // noisiest in exactly the config-managed deployments that would
1074 // least want it. Here rather than in the publishing run because
1075 // dispatching reaches the database.
1076 for profile in reloaded.mounted {
1077 announce_profile(&profile).await;
1078 }
1079 if let Some(respond) = request.respond {
1080 let _ = respond.send(Ok(report));
1081 }
1082 }
1083 Err(error) => {
1084 // Two names, because they are two different things for whoever
1085 // is reading: a refusal is a configuration an operator must
1086 // change, a failure is one the server could not build.
1087 match &error {
1088 crate::reload::ReloadError::Frozen { .. } => warn!(
1089 event = "server_config_reload_refused",
1090 outcome = "failure",
1091 generation = generation,
1092 reason = error.kind(),
1093 error = %error,
1094 ),
1095 _ => error!(
1096 event = "server_config_reload_failed",
1097 outcome = "failure",
1098 generation = generation,
1099 reason = error.kind(),
1100 error = %error,
1101 ),
1102 }
1103 if let Some(respond) = request.respond {
1104 let _ = respond.send(Err(error));
1105 }
1106 }
1107 }
1108 }
1109}
1110
1111/// One of the three sockets this process may hold.
1112///
1113/// An enum rather than the `&'static str` the log field wants, so the reload
1114/// path's per-role handling is exhaustive: a fourth listener would be a compile
1115/// error at every point that has to decide something about one, which is exactly
1116/// how the third arrived with `bind_metrics` and `check_metrics_config` in
1117/// place and nothing else remembering it existed.
1118#[derive(Clone, Copy, PartialEq, Eq)]
1119enum Role {
1120 Acme,
1121 Admin,
1122 Metrics,
1123}
1124
1125impl Role {
1126 /// The `listener` field every log line about this socket carries.
1127 fn label(self) -> &'static str {
1128 match self {
1129 Self::Acme => "acme",
1130 Self::Admin => "admin",
1131 Self::Metrics => "metrics",
1132 }
1133 }
1134
1135 /// The key an operator edits to move this socket, for a refusal to name.
1136 fn bind_key(self) -> &'static str {
1137 match self {
1138 Self::Acme => "server.bind_address",
1139 Self::Admin => "admin.bind_address",
1140 Self::Metrics => "metrics.bind_address",
1141 }
1142 }
1143}
1144
1145/// What a reload does to one role's socket.
1146///
1147/// Built while a failure can still refuse the whole reload, applied once nothing
1148/// can fail — the same build-then-publish split every other part of a generation
1149/// makes, applied to the one resource that cannot simply be constructed twice.
1150enum SocketPlan {
1151 /// The role's address and enablement are both unchanged. Note this is
1152 /// decided from the *configuration*, never from the address actually bound:
1153 /// a caller supplying its own socket (every test that binds `127.0.0.1:0`)
1154 /// is entitled to one that does not match the file, and rebinding it out
1155 /// from under them would be this feature breaking its own callers.
1156 Keep,
1157 /// Serve this newly bound socket: the role was switched on, or its address
1158 /// moved. Connections already established are untouched — hyper owns those,
1159 /// and only the socket beneath them changes.
1160 Serve(TcpListener),
1161 /// Release the socket: the role was switched off.
1162 Close,
1163}
1164
1165/// The three roles' socket plans, and what to say about them afterwards.
1166struct SocketPlans {
1167 acme: SocketPlan,
1168 admin: SocketPlan,
1169 metrics: SocketPlan,
1170 /// The resolved address of each freshly bound socket, so the announcement
1171 /// after the swap names where the listener actually landed.
1172 bound: Vec<(Role, String)>,
1173}
1174
1175impl SocketPlans {
1176 /// The roles whose socket this reload moved, for [`ReloadReport`] — which is
1177 /// what a test waits on and what an operator greps.
1178 ///
1179 /// [`ReloadReport`]: crate::reload::ReloadReport
1180 fn rebound(&self) -> Vec<&'static str> {
1181 self.bound.iter().map(|(role, _)| role.label()).collect()
1182 }
1183
1184 /// Hands each socket to its accept loop. Synchronous and infallible, so it
1185 /// sits inside the publishing run beside the routers.
1186 fn publish(self, cells: &Cells) {
1187 for (role, plan, handle) in [
1188 (Role::Acme, self.acme, &cells.acme),
1189 (Role::Admin, self.admin, &cells.admin),
1190 (Role::Metrics, self.metrics, &cells.metrics),
1191 ] {
1192 match plan {
1193 SocketPlan::Keep => {}
1194 SocketPlan::Serve(listener) => handle.serve(listener),
1195 SocketPlan::Close => {
1196 handle.close();
1197 info!(
1198 event = "server_listener_stopped",
1199 outcome = "success",
1200 listener = role.label(),
1201 "switched off by a configuration reload: the socket is released \
1202 and nothing new is accepted on it"
1203 );
1204 }
1205 }
1206 }
1207 }
1208}
1209
1210/// Decides, and performs, every bind this reload needs.
1211///
1212/// The ordering rule the whole reload path rests on, applied to sockets: bind
1213/// first, so a bad address refuses the reload rather than having already
1214/// dropped the live one. Two addresses that differ as strings but collide in
1215/// the kernel — `[::]:3000` against `0.0.0.0:3000` — make that bind fail with
1216/// `EADDRINUSE`, which is the safe direction: the running socket is still
1217/// serving and the refusal names the key.
1218///
1219/// A `tls.enabled` flip does not appear here at all. The mode is read per
1220/// connection (see [`crate::listener`]), so turning TLS on or off keeps the
1221/// socket exactly where it is — which is what makes the one case a bind-first
1222/// scheme could not serve, an unchanged address, not a case.
1223fn plan_sockets(
1224 applied: &Config,
1225 proposed: &Config,
1226) -> Result<SocketPlans, crate::reload::ReloadError> {
1227 let mut bound = Vec::new();
1228 let mut plan = |role: Role,
1229 was: Option<&str>,
1230 now: Option<&str>|
1231 -> Result<SocketPlan, crate::reload::ReloadError> {
1232 match (was, now) {
1233 (None, None) => Ok(SocketPlan::Keep),
1234 (Some(_), None) => Ok(SocketPlan::Close),
1235 (Some(was), Some(now)) if was == now => Ok(SocketPlan::Keep),
1236 (_, Some(now)) => {
1237 let listener = crate::listener::bind_blocking(now).map_err(|error| {
1238 error!(event = "server_socket_bind_failed",
1239 outcome = "failure",
1240 listener = role.label(),
1241 bind_address = %now,
1242 error = %error);
1243 crate::reload::ReloadError::Build(format!(
1244 "`{}` is `{now}`, which cannot be bound: {error}",
1245 role.bind_key()
1246 ))
1247 })?;
1248 bound.push((role, bound_address(Some(&listener), now)));
1249 Ok(SocketPlan::Serve(listener))
1250 }
1251 }
1252 };
1253
1254 // The ACME listener is never switched off — there is no `server.enabled`,
1255 // and a CA serving no ACME would be a process with nothing to do.
1256 let acme = plan(
1257 Role::Acme,
1258 Some(&applied.server.bind_address),
1259 Some(&proposed.server.bind_address),
1260 )?;
1261 let admin = plan(
1262 Role::Admin,
1263 applied
1264 .admin
1265 .enabled
1266 .then_some(applied.admin.bind_address.as_str()),
1267 proposed
1268 .admin
1269 .enabled
1270 .then_some(proposed.admin.bind_address.as_str()),
1271 )?;
1272 let metrics = plan(
1273 Role::Metrics,
1274 applied
1275 .metrics
1276 .enabled
1277 .then_some(applied.metrics.bind_address.as_str()),
1278 proposed
1279 .metrics
1280 .enabled
1281 .then_some(proposed.metrics.bind_address.as_str()),
1282 )?;
1283
1284 Ok(SocketPlans {
1285 acme,
1286 admin,
1287 metrics,
1288 bound,
1289 })
1290}
1291
1292/// What one successful reload hands back to the supervisor: the report to log,
1293/// and the pieces of state the *next* reload compares against.
1294///
1295/// A struct rather than the tuple this used to return — which needed
1296/// `#[allow(clippy::type_complexity)]` and left the caller destructuring four
1297/// same-shaped values positionally, where swapping two would still compile.
1298/// The same move `ProfileParts` made, for the same reason.
1299struct Reloaded {
1300 report: crate::reload::ReloadReport,
1301 config: Arc<Config>,
1302 resolved: Vec<crate::config::ProfileConfig>,
1303 logins: Option<Arc<crate::webadmin::LoginLimiter>>,
1304 /// Each socket this reload bound, with the address it landed on. Announced
1305 /// by the supervisor rather than here, because saying a listener is up
1306 /// reaches the database (the panel's "nobody can sign in yet" warning) and
1307 /// [`publish_reload`] has no await point to spend on it — deliberately, that
1308 /// being what keeps its publishing run uninterruptible.
1309 opened: Vec<(Role, String)>,
1310 /// The endpoints this reload **added**, for the same reason and announced in
1311 /// the same place: `profile_mounted` is dispatched to the `[notify]`
1312 /// backends, which queues a job row.
1313 mounted: Vec<Arc<Profile>>,
1314}
1315
1316/// Everything a reload built and validated, waiting to be published.
1317///
1318/// The build/publish split is the whole shape of a reload, and making it two
1319/// values rather than two halves of one function buys the thing the split was
1320/// always claiming: [`prepare_reload`] can run wherever it likes — it runs on a
1321/// blocking thread, since building a `relay` backend can contact its upstream —
1322/// while [`publish_reload`] stays on the supervisor task, where having no await
1323/// point is what makes a generation unobservable half-applied.
1324struct Prepared {
1325 config: Arc<Config>,
1326 resolved: Vec<crate::config::ProfileConfig>,
1327 parts: crate::GenerationParts,
1328 generation: Generation,
1329 sockets: SocketPlans,
1330 logging: logging::PreparedLogging,
1331 /// Whether `RUST_LOG` is what the filter came from, so the publish phase can
1332 /// say when an edited `logging.filter` changed nothing.
1333 logging_filter_from_env: bool,
1334 /// The endpoints in this generation that the previous one did not mount.
1335 mounted: Vec<Arc<Profile>>,
1336 /// The endpoints the previous generation mounted and this one does not.
1337 unmounted: Vec<String>,
1338}
1339
1340/// The build half of one reload: everything that can fail, and everything that
1341/// can block.
1342///
1343/// Nothing here touches a cell, so a failure anywhere leaves the running
1344/// generation exactly as it was — which is the property the whole "atomic,
1345/// refuse by name" decision exists for. Every socket this reload needs is bound
1346/// here too, where a port already in use is still a refusal rather than a
1347/// listener already dropped.
1348///
1349/// Run on a blocking thread by [`supervise_reloads`], because building a `relay`
1350/// backend for the first time contacts its upstream synchronously.
1351fn prepare_reload(
1352 config: &Arc<Config>,
1353 resolved: &[crate::config::ProfileConfig],
1354 assembly: &crate::Assembly,
1355 logins: Option<&crate::webadmin::LoginLimiter>,
1356) -> Result<Prepared, crate::reload::ReloadError> {
1357 use crate::reload::{Applied, ReloadError, check_frozen};
1358
1359 // Re-read from scratch: `Config::load` consults the file *and* the
1360 // `ACME_PROXY_*` environment, so a reload sees whatever the process would
1361 // see if it restarted right now.
1362 let next = Arc::new(Config::load().map_err(|error| ReloadError::Load(error.to_string()))?);
1363 let next_resolved = next
1364 .resolve_profiles()
1365 .map_err(|error| ReloadError::Load(error.to_string()))?;
1366
1367 check_frozen(
1368 &Applied {
1369 config,
1370 profiles: resolved,
1371 },
1372 &Applied {
1373 config: &next,
1374 profiles: &next_resolved,
1375 },
1376 )?;
1377
1378 // Built here rather than published here: a bad `logging.target` must refuse
1379 // the whole reload with the message startup would have printed, not leave a
1380 // half-swapped generation behind. The same build-then-publish split
1381 // `Assembly::build_parts` makes.
1382 let logging = logging::prepare_logging(&next.logging).map_err(ReloadError::Build)?;
1383 let logging_filter_from_env = logging.filter_from_env;
1384
1385 // The same validation startup runs before either socket binds, so a panel
1386 // that would refuse to start refuses to be reloaded into. It also compiles
1387 // every `admin.template_dir` override, which is what keeps a broken one a
1388 // failed reload rather than a 500 in a browser.
1389 crate::webadmin::check_config(&next).map_err(|error| ReloadError::Build(error.to_string()))?;
1390 // Its twin for the third listener: `webadmin::check_config` sees the
1391 // admin-versus-server pair, this one sees the two it cannot.
1392 check_metrics_config(&next).map_err(|error| ReloadError::Build(error.to_string()))?;
1393
1394 // Every socket this reload needs is bound **here**, where a failure is still
1395 // a refusal: a port already taken, an address that does not resolve, a
1396 // privileged port after a `setcap` was lost. Past the publish phase nothing
1397 // can fail, so the running listeners are never dropped for a configuration
1398 // that then turns out not to work.
1399 let sockets = plan_sockets(config, &next)?;
1400
1401 // The egress clients, the notification dispatchers and the signer backends.
1402 // The last is where a newly mounted endpoint gets a backend, a removed one's
1403 // is left out, and an edited `[signer]` is rebuilt over the live instance's
1404 // in-memory state — see `signer::build_backends`. It is also the one step
1405 // that can make a network call, hence this whole function's blocking thread.
1406 let parts = assembly
1407 .build_parts(&next_resolved, &next)
1408 .map_err(|error| ReloadError::Build(error.to_string()))?;
1409 let generation = build_generation(&next, &next_resolved, assembly, &parts, logins)
1410 .map_err(|error| ReloadError::Build(error.to_string()))?;
1411
1412 // Compared by name against what is running, not against what is written
1413 // down: `resolve_profiles` has already dropped every `enabled = false`
1414 // entry, so this is the set of endpoints actually served.
1415 let running: std::collections::HashSet<&str> = resolved
1416 .iter()
1417 .map(|profile| profile.name.as_str())
1418 .collect();
1419 let mounted = generation
1420 .profiles
1421 .iter()
1422 .filter(|profile| !running.contains(profile.name.as_str()))
1423 .cloned()
1424 .collect();
1425 let next_names: std::collections::HashSet<&str> = next_resolved
1426 .iter()
1427 .map(|profile| profile.name.as_str())
1428 .collect();
1429 let unmounted = resolved
1430 .iter()
1431 .map(|profile| profile.name.clone())
1432 .filter(|name| !next_names.contains(name.as_str()))
1433 .collect();
1434
1435 Ok(Prepared {
1436 config: next,
1437 resolved: next_resolved,
1438 parts,
1439 generation,
1440 sockets,
1441 logging,
1442 logging_filter_from_env,
1443 mounted,
1444 unmounted,
1445 })
1446}
1447
1448/// The publish half: infallible, synchronous, and uninterruptible.
1449///
1450/// There is no `.await` in here, and that is load-bearing rather than
1451/// incidental. `watch::Sender::send_replace` and
1452/// `mpsc::UnboundedSender::send` are both synchronous, so a run of them with no
1453/// await point between cannot be interleaved — no task can observe a generation
1454/// half-applied, and no lock is needed to say so.
1455///
1456/// The order matters in three places. `[logging]` goes **first**, because an
1457/// operator who raised the level did it to see what happens next, starting with
1458/// this reload's own completion line. The notifier map, the signer set and the
1459/// job registry go **before** the routers: a request served by the new
1460/// generation queues a `notify_deliver` row naming a slot id from the new
1461/// configuration, and a `NotifyJob` still holding the old map would retire it —
1462/// permanently, since an unknown backend id is a `Failed`, not a `Retry`. And
1463/// the TLS mode goes before the socket, so a freshly bound listener's very first
1464/// connection is already accepted under this generation's settings.
1465fn publish_reload(
1466 prepared: Prepared,
1467 applied: &Arc<Config>,
1468 assembly: &crate::Assembly,
1469 cells: &Cells,
1470 generation: u64,
1471 started: std::time::Instant,
1472) -> Reloaded {
1473 use crate::reload::ReloadReport;
1474
1475 let Prepared {
1476 config: next,
1477 resolved: next_resolved,
1478 parts,
1479 generation: built,
1480 sockets,
1481 logging,
1482 logging_filter_from_env,
1483 mounted,
1484 unmounted,
1485 } = prepared;
1486
1487 let logging_reloaded = logging::publish_logging(logging);
1488
1489 let report = ReloadReport {
1490 generation,
1491 profiles: built
1492 .profiles
1493 .iter()
1494 .map(|profile| profile.name.clone())
1495 .collect(),
1496 job_kinds: built.job_registry.kinds(),
1497 tls_reloaded: built.tls.is_some(),
1498 admin_tls_reloaded: built.admin_tls.is_some(),
1499 listeners_rebound: sockets.rebound(),
1500 logging_reloaded,
1501 duration: started.elapsed(),
1502 };
1503 let next_logins = built.logins.clone();
1504
1505 assembly.publish_notifiers(parts.dispatchers);
1506 // The set the *next* reload compares against, and the point at which the
1507 // backends this one dropped are finally released — after their replacements
1508 // were built and adopted their state, never before.
1509 assembly.publish_signers(parts.signers);
1510 cells
1511 .job_registry
1512 .send_replace(Arc::new(built.job_registry));
1513
1514 // `[jobs]` in its two halves, both synchronous and neither able to fail —
1515 // which is what lets them sit in this run rather than needing a build phase
1516 // of their own. The runner re-derives its pacing from the cell on its next
1517 // pass; `max_attempts` goes to the queue instead, because it is the enqueue
1518 // side that reads it, and it sets the budget for work queued from here on
1519 // rather than for the rows already waiting.
1520 cells.jobs.send_replace(Arc::new(next.jobs.clone()));
1521 assembly.jobs.set_max_attempts(next.jobs.max_attempts);
1522
1523 cells.acme.set_tls(built.tls);
1524 cells.admin.set_tls(built.admin_tls);
1525 let opened = sockets.bound.clone();
1526 sockets.publish(cells);
1527
1528 cells
1529 .acme_router
1530 .send_replace(built.acme_app.into_service::<axum::body::Body>());
1531 // An empty router when the panel is off, which is what a request arriving
1532 // on a connection established a moment before it was switched off now gets:
1533 // closing the socket stops the next client, and this stops that one.
1534 cells.admin_router.send_replace(
1535 built
1536 .admin_app
1537 .unwrap_or_default()
1538 .into_service::<axum::body::Body>(),
1539 );
1540
1541 // Said here rather than by the supervisor because it needs nothing but a
1542 // name, unlike the mounting half, which dispatches a notification.
1543 for profile in unmounted {
1544 warn!(
1545 event = "profile_unmounted",
1546 outcome = "advisory",
1547 profile = %profile,
1548 "the endpoint is no longer served: its accounts and orders stay in the \
1549 database and come back if it is mounted again, but any issuance still in \
1550 flight for it has no handler left to finish it"
1551 );
1552 }
1553
1554 // `RUST_LOG` outranks `logging.filter` on a reload exactly as it does at
1555 // startup — the two disagreeing would be worse — but that makes an edited
1556 // `logging.filter` a silent no-op, which is the one outcome an operator
1557 // would read as "my reload did not land". Said only when both halves hold:
1558 // the environment won, *and* the file's filter actually moved.
1559 if logging_filter_from_env && applied.logging.filter != next.logging.filter {
1560 warn!(
1561 event = "server_logging_filter_overridden",
1562 outcome = "advisory",
1563 configured = %next.logging.filter,
1564 );
1565 }
1566
1567 Reloaded {
1568 report,
1569 config: next,
1570 resolved: next_resolved,
1571 logins: next_logins,
1572 opened,
1573 mounted,
1574 }
1575}
1576
1577/// The one announcement an endpoint that has just come up makes: a log line and
1578/// a `[notify]` lifecycle event.
1579///
1580/// Shared by startup and by a reload that mounted a new endpoint, so the two
1581/// cannot drift — before the profile set could reload there was only one caller
1582/// and the sharing was not needed.
1583async fn announce_profile(profile: &Arc<Profile>) {
1584 info!(
1585 event = "profile_mounted",
1586 outcome = "success",
1587 profile = %profile.name,
1588 directory = %profile.directory_url(),
1589 challenge_bypass = profile.challenges.is_bypassed(),
1590 eab_enabled = profile.eab.enabled
1591 );
1592 profile
1593 .notify
1594 .dispatch(crate::notify::NotifyEvent::ProfileMounted(
1595 crate::notify::ProfileMountedData {
1596 profile: profile.name.clone(),
1597 },
1598 ))
1599 .await;
1600}
1601
1602/// A future that completes when the shutdown relay fires.
1603async fn on_shutdown(mut receiver: tokio::sync::watch::Receiver<bool>) {
1604 // An error means the sender was dropped, which only happens when the relay
1605 // task itself is gone — treat it as "shut down" rather than parking
1606 // forever.
1607 let _ = receiver.wait_for(|ready| *ready).await;
1608}
1609
1610/// Says the panel is up, and warns about the two states that make it useless.
1611///
1612/// Run whenever the admin listener **opens** — at startup, and again on a
1613/// reload that turns `admin.enabled` on. Separate from the serving path since
1614/// that no longer starts or stops per role: the socket is what comes and goes,
1615/// and this is what an operator needs told when it does.
1616async fn announce_admin_listener(config: &Arc<Config>, database: &Arc<Database>, bound: &str) {
1617 // A listener nobody holds an account for is a running service with no way
1618 // in; say so once, naming the command that fixes it.
1619 if crate::sqlite::admin_user::AdminUser::list_all(database)
1620 .await
1621 .is_ok_and(|users| users.is_empty())
1622 {
1623 warn!(
1624 event = "admin_no_users",
1625 outcome = "advisory",
1626 "the web admin is enabled but has no operators: create one with \
1627 `acme-proxy admin user create <username>`"
1628 );
1629 }
1630
1631 // Repeated on every start while it holds, the `challenge_validation_bypassed`
1632 // treatment: these operators can still sign in, they are simply made to
1633 // enrol before their session becomes usable, and that stays worth seeing
1634 // for exactly as long as it is true.
1635 if config.admin.require_mfa
1636 && let Ok(count) = crate::admin::mfa::operators_without_a_factor(database.clone()).await
1637 && count > 0
1638 {
1639 warn!(
1640 event = "admin_mfa_enrolment_pending",
1641 outcome = "advisory",
1642 count = count,
1643 "admin.require_mfa is on and some operators have no second factor: \
1644 their next sign-in will require enrolment before the session is usable"
1645 );
1646 }
1647
1648 // Swept once now, then on an interval: sessions outlive a restart, so a
1649 // startup-only sweep would leak every one an operator never signed out of.
1650 let idle = Duration::from_secs(config.admin.session_idle_timeout_seconds);
1651 if let Err(error) = crate::sqlite::admin_session::AdminSession::cleanup(idle, database).await {
1652 error!(event = "admin_session_cleanup_failed", outcome = "failure", error = %error);
1653 }
1654
1655 // The **resolved** address, so a `:0` bind is discoverable.
1656 info!(
1657 event = "admin_listening",
1658 outcome = "success",
1659 bind_address = %bound,
1660 protocol = if config.admin.tls.enabled { "https" } else { "http" },
1661 base_url = %config.admin.base_url
1662 );
1663}
1664
1665/// The metrics listener's own one-line announcement, and its standing warning.
1666fn announce_metrics_listener(bound: &str) {
1667 info!(
1668 event = "metrics_listening",
1669 outcome = "success",
1670 bind_address = %bound,
1671 "unauthenticated by design: the port is the boundary, so firewall it"
1672 );
1673}
1674
1675/// The address a socket ended up on, falling back to what was configured.
1676///
1677/// The two differ for `:0` and for a caller that supplied its own listener; the
1678/// resolved one is what an operator needs, and the configured one is all there
1679/// is to say when the socket cannot answer.
1680fn bound_address(listener: Option<&TcpListener>, configured: &str) -> String {
1681 listener
1682 .and_then(|listener| listener.local_addr().ok())
1683 .map_or_else(|| configured.to_string(), |address| address.to_string())
1684}
1685
1686async fn shutdown_signal() {
1687 let ctrl_c = async {
1688 tokio::signal::ctrl_c()
1689 .await
1690 .expect("failed to install Ctrl+C handler");
1691 };
1692
1693 #[cfg(unix)]
1694 let terminate = async {
1695 tokio::signal::unix::signal(tokio::signal::unix::SignalKind::terminate())
1696 .expect("failed to install signal handler")
1697 .recv()
1698 .await;
1699 };
1700
1701 #[cfg(not(unix))]
1702 let terminate = std::future::pending::<()>();
1703
1704 tokio::select! {
1705 _ = ctrl_c => {},
1706 _ = terminate => {},
1707 }
1708}
1709
1710/// Aborts a background task when it goes out of scope.
1711struct AbortOnDrop(tokio::task::JoinHandle<()>);
1712
1713impl Drop for AbortOnDrop {
1714 fn drop(&mut self) {
1715 self.0.abort();
1716 }
1717}
1718
1719#[cfg(test)]
1720mod tests {
1721 use super::*;
1722
1723 /// `--version` exists and reports the crate version. The bug report
1724 /// template tells people to run it, and clap generates the flag only
1725 /// because `#[command(version = …)]` says so — drop that and the first
1726 /// instruction on the form starts erroring out.
1727 #[test]
1728 fn version_flag_reports_the_crate_version() {
1729 let Err(error) = Cli::try_parse_from(["acme-proxy", "--version"]) else {
1730 panic!("--version parsed as a command rather than printing a version");
1731 };
1732 assert_eq!(error.kind(), clap::error::ErrorKind::DisplayVersion);
1733 assert!(error.to_string().contains(env!("CARGO_PKG_VERSION")));
1734 }
1735
1736 #[test]
1737 fn parse_cli_subcommands() {
1738 let cli = Cli::try_parse_from(["acme-proxy"]).unwrap();
1739 assert!(cli.command.is_none());
1740
1741 let cli = Cli::try_parse_from(["acme-proxy", "serve"]).unwrap();
1742 assert!(matches!(cli.command, Some(Command::Serve)));
1743
1744 let cli = Cli::try_parse_from(["acme-proxy", "account", "list", "--json"]).unwrap();
1745 assert!(matches!(
1746 cli.command,
1747 Some(Command::Account {
1748 command: AccountCommand::List {
1749 json: true,
1750 profile: None,
1751 limit: window::DEFAULT_LIMIT,
1752 offset: 0
1753 }
1754 })
1755 ));
1756
1757 let cli = Cli::try_parse_from(["acme-proxy", "account", "show", "acct-1"]).unwrap();
1758 assert!(matches!(
1759 cli.command,
1760 Some(Command::Account {
1761 command: AccountCommand::Show { id, json: false }
1762 }) if id == "acct-1"
1763 ));
1764
1765 let cli = Cli::try_parse_from([
1766 "acme-proxy",
1767 "account",
1768 "update-contact",
1769 "acct-1",
1770 "--contact",
1771 "mailto:test@example.com",
1772 ])
1773 .unwrap();
1774 assert!(matches!(
1775 cli.command,
1776 Some(Command::Account {
1777 command: AccountCommand::UpdateContact { id, contact }
1778 }) if id == "acct-1" && contact == vec!["mailto:test@example.com"]
1779 ));
1780
1781 let cli = Cli::try_parse_from(["acme-proxy", "account", "deactivate", "acct-1"]).unwrap();
1782 assert!(matches!(
1783 cli.command,
1784 Some(Command::Account {
1785 command: AccountCommand::Deactivate { id }
1786 }) if id == "acct-1"
1787 ));
1788
1789 let cli = Cli::try_parse_from(["acme-proxy", "-y", "account", "delete", "acct-1"]).unwrap();
1790 assert!(cli.yes);
1791 assert!(matches!(
1792 cli.command,
1793 Some(Command::Account {
1794 command: AccountCommand::Delete { id }
1795 }) if id == "acct-1"
1796 ));
1797
1798 let cli = Cli::try_parse_from([
1799 "acme-proxy",
1800 "order",
1801 "list",
1802 "--account-id",
1803 "acct-1",
1804 "--status",
1805 "pending",
1806 "--json",
1807 ])
1808 .unwrap();
1809 assert!(matches!(
1810 cli.command,
1811 Some(Command::Order {
1812 command: OrderCommand::List {
1813 profile: None,
1814 account_id: Some(a),
1815 status: Some(s),
1816 expiring_in: None,
1817 hide_superseded: false,
1818 limit: window::DEFAULT_LIMIT,
1819 offset: 0,
1820 json: true
1821 }
1822 }) if a == "acct-1" && s == "pending"
1823 ));
1824
1825 let cli = Cli::try_parse_from([
1826 "acme-proxy",
1827 "order",
1828 "list",
1829 "--expiring-in",
1830 "30",
1831 "--hide-superseded",
1832 ])
1833 .unwrap();
1834 assert!(matches!(
1835 cli.command,
1836 Some(Command::Order {
1837 command: OrderCommand::List {
1838 expiring_in: Some(30),
1839 hide_superseded: true,
1840 status: None,
1841 account_id: None,
1842 profile: None,
1843 limit: window::DEFAULT_LIMIT,
1844 offset: 0,
1845 json: false
1846 }
1847 })
1848 ));
1849
1850 let cli = Cli::try_parse_from(["acme-proxy", "order", "show", "ord-1"]).unwrap();
1851 assert!(matches!(
1852 cli.command,
1853 Some(Command::Order {
1854 command: OrderCommand::Show { id, json: false }
1855 }) if id == "ord-1"
1856 ));
1857
1858 let cli = Cli::try_parse_from(["acme-proxy", "order", "delete", "ord-1"]).unwrap();
1859 assert!(matches!(
1860 cli.command,
1861 Some(Command::Order {
1862 command: OrderCommand::Delete { id }
1863 }) if id == "ord-1"
1864 ));
1865
1866 let cli = Cli::try_parse_from(["acme-proxy", "order", "revoke", "ord-1", "--reason", "1"])
1867 .unwrap();
1868 assert!(matches!(
1869 cli.command,
1870 Some(Command::Order {
1871 command: OrderCommand::Revoke { id, reason: Some(1) }
1872 }) if id == "ord-1"
1873 ));
1874
1875 let cli =
1876 Cli::try_parse_from(["acme-proxy", "nonce", "cleanup", "--ttl-seconds", "60"]).unwrap();
1877 assert!(matches!(
1878 cli.command,
1879 Some(Command::Nonce {
1880 command: NonceCommand::Cleanup {
1881 ttl_seconds: Some(60)
1882 }
1883 })
1884 ));
1885
1886 let cli = Cli::try_parse_from([
1887 "acme-proxy",
1888 "eab",
1889 "create",
1890 "--label",
1891 "test-key",
1892 "--json",
1893 ])
1894 .unwrap();
1895 assert!(matches!(
1896 cli.command,
1897 Some(Command::Eab {
1898 command: EabCommand::Create { label: Some(l), profile: None, json: true }
1899 }) if l == "test-key"
1900 ));
1901
1902 let cli = Cli::try_parse_from(["acme-proxy", "eab", "list"]).unwrap();
1903 assert!(matches!(
1904 cli.command,
1905 Some(Command::Eab {
1906 command: EabCommand::List { json: false }
1907 })
1908 ));
1909
1910 let cli = Cli::try_parse_from(["acme-proxy", "eab", "show", "kid-1", "--json"]).unwrap();
1911 assert!(matches!(
1912 cli.command,
1913 Some(Command::Eab {
1914 command: EabCommand::Show { kid, json: true }
1915 }) if kid == "kid-1"
1916 ));
1917
1918 let cli = Cli::try_parse_from(["acme-proxy", "upstream", "register", "--eab-kid", "kid-1"])
1919 .unwrap();
1920 assert!(matches!(
1921 cli.command,
1922 Some(Command::Upstream {
1923 command: UpstreamCommand::Register { eab_kid: Some(kid), eab_hmac_key_file: None, profile: None }
1924 }) if kid == "kid-1"
1925 ));
1926
1927 // Registering against an upstream that needs no credential.
1928 let cli = Cli::try_parse_from(["acme-proxy", "upstream", "register"]).unwrap();
1929 assert!(matches!(
1930 cli.command,
1931 Some(Command::Upstream {
1932 command: UpstreamCommand::Register {
1933 eab_kid: None,
1934 eab_hmac_key_file: None,
1935 profile: None,
1936 }
1937 })
1938 ));
1939
1940 // The secret itself has no flag: it is stdin- or file-only, never argv.
1941 assert!(
1942 Cli::try_parse_from(["acme-proxy", "upstream", "register", "--eab-hmac-key", "s"])
1943 .is_err(),
1944 "an EAB secret must not be accepted on the command line"
1945 );
1946
1947 let cli = Cli::try_parse_from(["acme-proxy", "upstream", "show", "--json"]).unwrap();
1948 assert!(matches!(
1949 cli.command,
1950 Some(Command::Upstream {
1951 command: UpstreamCommand::Show {
1952 json: true,
1953 profile: None
1954 }
1955 })
1956 ));
1957
1958 let cli = Cli::try_parse_from(["acme-proxy", "eab", "revoke", "kid-1"]).unwrap();
1959 assert!(matches!(
1960 cli.command,
1961 Some(Command::Eab {
1962 command: EabCommand::Revoke { kid }
1963 }) if kid == "kid-1"
1964 ));
1965
1966 let cli = Cli::try_parse_from(["acme-proxy", "admin", "user", "create", "alice"]).unwrap();
1967 assert!(matches!(
1968 cli.command,
1969 Some(Command::Admin {
1970 command: AdminCommand::User {
1971 command: crate::cli::webadmin::AdminUserCommand::Create {
1972 username,
1973 password_file: None
1974 }
1975 }
1976 }) if username == "alice"
1977 ));
1978
1979 let cli = Cli::try_parse_from([
1980 "acme-proxy",
1981 "admin",
1982 "user",
1983 "passwd",
1984 "alice",
1985 "--password-file",
1986 "/run/secrets/pw",
1987 ])
1988 .unwrap();
1989 assert!(matches!(
1990 cli.command,
1991 Some(Command::Admin {
1992 command: AdminCommand::User {
1993 command: crate::cli::webadmin::AdminUserCommand::Passwd {
1994 username,
1995 password_file: Some(path)
1996 }
1997 }
1998 }) if username == "alice" && path == std::path::Path::new("/run/secrets/pw")
1999 ));
2000
2001 // The password itself has no flag, for the same reason the EAB secret
2002 // has none: argv is visible in `ps` and lands in shell history.
2003 for command in ["create", "passwd"] {
2004 assert!(
2005 Cli::try_parse_from([
2006 "acme-proxy",
2007 "admin",
2008 "user",
2009 command,
2010 "alice",
2011 "--password",
2012 "hunter2",
2013 ])
2014 .is_err(),
2015 "`admin user {command}` must not accept a password on the command line"
2016 );
2017 }
2018
2019 // `--color` is global like `--yes`, so it may sit anywhere on the line,
2020 // and an unknown value is refused by clap rather than falling back to
2021 // `auto` — the same rule `--status`/`--event` follow, for the same
2022 // reason: a silently ignored value looks exactly like a working one.
2023 let cli = Cli::try_parse_from(["acme-proxy", "account", "list", "--color", "never"])
2024 .expect("--color is global and accepts `never`");
2025 assert_eq!(cli.color, ColorChoice::Never);
2026
2027 let cli = Cli::try_parse_from(["acme-proxy", "--color", "always", "account", "list"])
2028 .expect("--color is global, so it may precede the subcommand");
2029 assert_eq!(cli.color, ColorChoice::Always);
2030
2031 assert_eq!(
2032 Cli::try_parse_from(["acme-proxy", "account", "list"])
2033 .unwrap()
2034 .color,
2035 ColorChoice::Auto,
2036 "unset means auto"
2037 );
2038
2039 assert!(
2040 Cli::try_parse_from(["acme-proxy", "account", "list", "--color", "sometimes"]).is_err(),
2041 "an unknown --color value must be refused, not ignored"
2042 );
2043
2044 let cli = Cli::try_parse_from(["acme-proxy", "admin", "user", "totp", "status", "alice"])
2045 .unwrap();
2046 assert!(matches!(
2047 cli.command,
2048 Some(Command::Admin {
2049 command: AdminCommand::User {
2050 command: crate::cli::webadmin::AdminUserCommand::Totp {
2051 command: crate::cli::webadmin::AdminUserTotpCommand::Status {
2052 username,
2053 json: false
2054 }
2055 }
2056 }
2057 }) if username == "alice"
2058 ));
2059
2060 let cli = Cli::try_parse_from([
2061 "acme-proxy",
2062 "admin",
2063 "user",
2064 "totp",
2065 "recovery-codes",
2066 "alice",
2067 ])
2068 .unwrap();
2069 assert!(matches!(
2070 cli.command,
2071 Some(Command::Admin {
2072 command: AdminCommand::User {
2073 command: crate::cli::webadmin::AdminUserCommand::Totp {
2074 command: crate::cli::webadmin::AdminUserTotpCommand::RecoveryCodes {
2075 username
2076 }
2077 }
2078 }
2079 }) if username == "alice"
2080 ));
2081
2082 // There is no `enrol` from a terminal, deliberately: it would put the
2083 // base32 secret in scrollback and shell history. See the doc comment on
2084 // `AdminUserTotpCommand`.
2085 assert!(
2086 Cli::try_parse_from(["acme-proxy", "admin", "user", "totp", "enrol", "alice"]).is_err()
2087 );
2088
2089 let cli =
2090 Cli::try_parse_from(["acme-proxy", "-y", "admin", "user", "delete", "alice"]).unwrap();
2091 assert!(cli.yes);
2092 assert!(matches!(
2093 cli.command,
2094 Some(Command::Admin {
2095 command: AdminCommand::User {
2096 command: crate::cli::webadmin::AdminUserCommand::Delete { username }
2097 }
2098 }) if username == "alice"
2099 ));
2100
2101 let cli =
2102 Cli::try_parse_from(["acme-proxy", "admin", "session", "list", "--json"]).unwrap();
2103 assert!(matches!(
2104 cli.command,
2105 Some(Command::Admin {
2106 command: AdminCommand::Session {
2107 command: crate::cli::webadmin::AdminSessionCommand::List {
2108 username: None,
2109 json: true
2110 }
2111 }
2112 })
2113 ));
2114
2115 // `--user` and `--all` answer the same question two ways; clap refuses
2116 // both rather than letting one silently win.
2117 assert!(
2118 Cli::try_parse_from([
2119 "acme-proxy",
2120 "admin",
2121 "session",
2122 "revoke",
2123 "--user",
2124 "alice",
2125 "--all",
2126 ])
2127 .is_err(),
2128 "--user and --all are mutually exclusive"
2129 );
2130 }
2131
2132 #[test]
2133 fn a_database_error_renders_as_a_cli_error() {
2134 let error = CliError::from(sqlx::Error::PoolClosed);
2135 assert!(error.to_string().starts_with("database error: "), "{error}");
2136 }
2137
2138 /// Every arm reaches its command handler. `Serve` is deliberately absent —
2139 /// it owns a socket, and [`serve_on`] is what the tests below drive.
2140 #[tokio::test]
2141 async fn dispatch_routes_each_command() {
2142 let database = Arc::new(Database::connect_in_memory().await.unwrap());
2143 let config = Arc::new(Config::default());
2144 let mut reader: &[u8] = &[];
2145
2146 let commands = vec![
2147 Command::Account {
2148 command: AccountCommand::List {
2149 profile: None,
2150 limit: window::DEFAULT_LIMIT,
2151 offset: 0,
2152 json: false,
2153 },
2154 },
2155 Command::Order {
2156 command: OrderCommand::List {
2157 profile: None,
2158 account_id: None,
2159 status: None,
2160 expiring_in: None,
2161 hide_superseded: false,
2162 limit: window::DEFAULT_LIMIT,
2163 offset: 0,
2164 json: false,
2165 },
2166 },
2167 Command::Nonce {
2168 command: NonceCommand::Cleanup {
2169 ttl_seconds: Some(1),
2170 },
2171 },
2172 Command::Eab {
2173 command: EabCommand::List { json: false },
2174 },
2175 Command::Man,
2176 Command::Completions {
2177 shell: clap_complete::aot::Shell::Bash,
2178 },
2179 // `Upstream` is deliberately absent: it acts on a *profile's*
2180 // `[signer.relay]`, and this config has none, so it now
2181 // reports that rather than silently reading the global base
2182 // section nothing serves from. Covered in `cli::upstream`'s own
2183 // tests, which supply a configuration with profiles.
2184 ];
2185 for command in commands {
2186 dispatch(
2187 Some(command),
2188 true,
2189 ColorChoice::Never,
2190 &mut reader,
2191 &config,
2192 database.clone(),
2193 )
2194 .await
2195 .expect("every command must succeed against an empty database");
2196 }
2197 }
2198
2199 /// A failing command's message reaches [`dispatch`]'s caller rather than
2200 /// exiting the process where it was raised.
2201 #[tokio::test]
2202 async fn dispatch_propagates_a_command_failure() {
2203 let database = Arc::new(Database::connect_in_memory().await.unwrap());
2204 let config = Arc::new(Config::default());
2205 let mut reader: &[u8] = &[];
2206
2207 let error = dispatch(
2208 Some(Command::Account {
2209 command: AccountCommand::Show {
2210 id: "acct-nope".to_string(),
2211 json: false,
2212 },
2213 }),
2214 true,
2215 ColorChoice::Never,
2216 &mut reader,
2217 &config,
2218 database,
2219 )
2220 .await
2221 .expect_err("an unknown account must fail");
2222 assert_eq!(error, CliError("no such account: acct-nope".to_string()));
2223 }
2224
2225 mod serving {
2226 use super::*;
2227 use tokio::io::{AsyncReadExt, AsyncWriteExt};
2228 use tokio::net::TcpStream;
2229
2230 /// A single-profile configuration whose CA and TLS material all live
2231 /// under `dir`, so a test never touches the repository.
2232 fn config_in(dir: impl AsRef<std::path::Path>, tls: bool) -> Config {
2233 let dir = dir.as_ref();
2234 let _lock = crate::config::ENV_LOCK
2235 .lock()
2236 .unwrap_or_else(std::sync::PoisonError::into_inner);
2237 let ca = dir.join("ca");
2238 let body = format!(
2239 r#"
2240 [server]
2241 bind_address = "127.0.0.1:0"
2242 base_url = "http://localhost:3000"
2243
2244 [server.tls]
2245 enabled = {tls}
2246 cert_path = "{dir}/server.pem"
2247 key_path = "{dir}/server.key"
2248
2249 [profiles.default]
2250 signer.local_ca.cert_path = "{ca}.pem"
2251 signer.local_ca.key_path = "{ca}.key"
2252 signer.local_ca.crl_path = "{ca}.crl"
2253 "#,
2254 dir = dir.display(),
2255 ca = ca.display(),
2256 );
2257 std::fs::write(dir.join("config.toml"), body).unwrap();
2258 // SAFETY: the lock above makes this the only thread touching the
2259 // environment, and the variable is removed before returning.
2260 unsafe {
2261 std::env::set_var("ACME_PROXY_CONFIG", dir.join("config").to_str().unwrap());
2262 }
2263 let config = Config::load().expect("the configuration must load");
2264 unsafe {
2265 std::env::remove_var("ACME_PROXY_CONFIG");
2266 }
2267 config
2268 }
2269
2270 fn temp_dir() -> crate::testutil::TempDir {
2271 crate::testutil::TempDir::new("serve")
2272 }
2273
2274 /// Boots `serve_on` on an ephemeral loopback port and returns it with
2275 /// the handle and the trigger that shuts it back down.
2276 async fn boot(
2277 config: Config,
2278 ) -> (
2279 SocketAddr,
2280 tokio::sync::oneshot::Sender<()>,
2281 tokio::task::JoinHandle<anyhow::Result<()>>,
2282 ) {
2283 let database = Arc::new(Database::connect_in_memory().await.unwrap());
2284 let listener = TcpListener::bind("127.0.0.1:0").await.unwrap();
2285 let addr = listener.local_addr().unwrap();
2286 let (tx, rx) = tokio::sync::oneshot::channel::<()>();
2287 let handle = tokio::spawn(serve_on(Arc::new(config), database, listener, async {
2288 let _ = rx.await;
2289 }));
2290 (addr, tx, handle)
2291 }
2292
2293 /// The cleartext path: real socket, real router, real graceful
2294 /// shutdown. `/health` is a root route, so this also proves the app
2295 /// `serve_on` assembles is the one `build_app` produces.
2296 #[tokio::test]
2297 async fn a_cleartext_server_answers_then_shuts_down() {
2298 let dir = temp_dir();
2299 let (addr, shutdown, handle) = boot(config_in(&dir, false)).await;
2300
2301 let mut stream = TcpStream::connect(addr).await.unwrap();
2302 stream
2303 .write_all(b"GET /health HTTP/1.1\r\nHost: localhost\r\nConnection: close\r\n\r\n")
2304 .await
2305 .unwrap();
2306 let mut response = String::new();
2307 stream.read_to_string(&mut response).await.unwrap();
2308 assert!(response.starts_with("HTTP/1.1 200 OK"), "{response}");
2309
2310 shutdown.send(()).unwrap();
2311 handle
2312 .await
2313 .unwrap()
2314 .expect("a clean shutdown is not an error");
2315 }
2316
2317 /// The same path with `server.tls.enabled`, which swaps the listener
2318 /// for a `TlsListener` — a different `axum::serve` arm, and the only
2319 /// place `TlsListener::spawn` is wired up in production.
2320 #[tokio::test]
2321 async fn a_tls_server_answers_over_a_real_handshake() {
2322 use tokio_rustls::TlsConnector;
2323 use tokio_rustls::rustls::pki_types::ServerName;
2324
2325 let dir = temp_dir();
2326 let (addr, shutdown, handle) = boot(config_in(&dir, true)).await;
2327
2328 // The generated certificate is self-signed, so the client must not
2329 // try to verify it — the point here is the listener, not the trust
2330 // chain.
2331 let client =
2332 crate::challenge::tls_alpn_01::accept_any_client_config(&[b"http/1.1"]).unwrap();
2333 let stream = TcpStream::connect(addr).await.unwrap();
2334 let mut tls = TlsConnector::from(client)
2335 .connect(ServerName::try_from("localhost").unwrap(), stream)
2336 .await
2337 .unwrap();
2338 tls.write_all(b"GET /health HTTP/1.1\r\nHost: localhost\r\nConnection: close\r\n\r\n")
2339 .await
2340 .unwrap();
2341 let mut response = String::new();
2342 tls.read_to_string(&mut response).await.unwrap();
2343 assert!(response.starts_with("HTTP/1.1 200 OK"), "{response}");
2344
2345 shutdown.send(()).unwrap();
2346 handle
2347 .await
2348 .unwrap()
2349 .expect("a clean shutdown is not an error");
2350 }
2351
2352 /// The three-listener path: one shutdown signal, three sockets, and
2353 /// each one serving only what belongs to it.
2354 ///
2355 /// This is the regression test for the `watch` split — before it, the
2356 /// `shutdown` future was consumed once and only one listener stopped —
2357 /// and now also for the metrics listener being genuinely *separate*:
2358 /// `/metrics` answering on the ACME port would put issuance volume and
2359 /// every profile name on the public socket, which is exactly what
2360 /// giving it its own port is for.
2361 #[tokio::test]
2362 async fn all_three_listeners_serve_and_one_signal_stops_them() {
2363 let dir = temp_dir();
2364 let mut config = config_in(&dir, false);
2365 config.admin.enabled = true;
2366 config.metrics.enabled = true;
2367
2368 let database = Arc::new(Database::connect_in_memory().await.unwrap());
2369 let acme_listener = TcpListener::bind("127.0.0.1:0").await.unwrap();
2370 let admin_listener = TcpListener::bind("127.0.0.1:0").await.unwrap();
2371 let metrics_listener = TcpListener::bind("127.0.0.1:0").await.unwrap();
2372 let acme_addr = acme_listener.local_addr().unwrap();
2373 let admin_addr = admin_listener.local_addr().unwrap();
2374 let metrics_addr = metrics_listener.local_addr().unwrap();
2375 assert_ne!(acme_addr, admin_addr);
2376 assert_ne!(acme_addr, metrics_addr);
2377 assert_ne!(admin_addr, metrics_addr);
2378
2379 let (tx, rx) = tokio::sync::oneshot::channel::<()>();
2380 let handle = tokio::spawn(serve_on_with(
2381 Arc::new(config),
2382 database,
2383 acme_listener,
2384 Some(admin_listener),
2385 Some(metrics_listener),
2386 async {
2387 let _ = rx.await;
2388 },
2389 ));
2390
2391 // `/health` is on both of the two that have it.
2392 for addr in [acme_addr, admin_addr] {
2393 let response = get(addr, "/health").await;
2394 assert!(
2395 response.starts_with("HTTP/1.1 200 OK"),
2396 "{addr}: {response}"
2397 );
2398 }
2399
2400 // The admin API answers on the admin socket (401, since this
2401 // request carries no session) and is absent from the ACME one.
2402 let admin = get(admin_addr, "/api/accounts").await;
2403 assert!(admin.starts_with("HTTP/1.1 401"), "{admin}");
2404 let acme = get(acme_addr, "/api/accounts").await;
2405 assert!(acme.starts_with("HTTP/1.1 404"), "{acme}");
2406
2407 // And the converse: ACME is on the ACME socket only.
2408 let directory = get(acme_addr, "/profile/default/directory").await;
2409 assert!(directory.starts_with("HTTP/1.1 200 OK"), "{directory}");
2410 let no_directory = get(admin_addr, "/profile/default/directory").await;
2411 assert!(no_directory.starts_with("HTTP/1.1 404"), "{no_directory}");
2412
2413 // The exposition is on its own socket...
2414 let metrics = get(metrics_addr, "/metrics").await;
2415 assert!(metrics.starts_with("HTTP/1.1 200 OK"), "{metrics}");
2416 assert!(metrics.contains("acme_proxy_requests_total"), "{metrics}");
2417 // ...and on **neither** of the other two. The whole point of the
2418 // third listener is that firewalling this port is what controls who
2419 // can read it.
2420 for addr in [acme_addr, admin_addr] {
2421 let leaked = get(addr, "/metrics").await;
2422 assert!(leaked.starts_with("HTTP/1.1 404"), "{addr}: {leaked}");
2423 }
2424
2425 // One signal, all three stop.
2426 tx.send(()).unwrap();
2427 tokio::time::timeout(Duration::from_secs(10), handle)
2428 .await
2429 .expect("every listener must stop on one signal")
2430 .unwrap()
2431 .expect("a clean shutdown is not an error");
2432 }
2433
2434 /// Three listeners means the collision check is pairwise, and
2435 /// `webadmin::check_config` only covers admin-versus-server.
2436 #[tokio::test]
2437 async fn a_metrics_bind_colliding_with_another_listener_is_refused() {
2438 let dir = temp_dir();
2439
2440 for (other, set) in [
2441 (
2442 "server.bind_address",
2443 Box::new(|c: &mut Config| {
2444 c.metrics.bind_address = c.server.bind_address.clone()
2445 }) as Box<dyn Fn(&mut Config)>,
2446 ),
2447 (
2448 "admin.bind_address",
2449 Box::new(|c: &mut Config| {
2450 c.admin.enabled = true;
2451 c.metrics.bind_address = c.admin.bind_address.clone();
2452 }),
2453 ),
2454 ] {
2455 let mut config = config_in(&dir, false);
2456 config.metrics.enabled = true;
2457 set(&mut config);
2458
2459 let error = bind_metrics(&Arc::new(config))
2460 .await
2461 .expect_err("a shared socket must not start");
2462 let message = error.to_string();
2463 assert!(message.contains(other), "{message}");
2464 }
2465 }
2466
2467 /// The admin bind is only a conflict when the panel is actually going
2468 /// to bind it. Both default to loopback ports, so refusing on the
2469 /// *value* alone would reject a configuration that works.
2470 #[tokio::test]
2471 async fn a_metrics_bind_matching_a_disabled_admin_is_allowed() {
2472 let dir = temp_dir();
2473 let mut config = config_in(&dir, false);
2474 config.metrics.enabled = true;
2475 config.admin.enabled = false;
2476 config.metrics.bind_address = config.admin.bind_address.clone();
2477
2478 let listener = bind_metrics(&Arc::new(config))
2479 .await
2480 .expect("a disabled panel holds no socket");
2481 assert!(listener.is_some());
2482 }
2483
2484 /// Off by default, and off means no socket at all rather than one
2485 /// answering 404.
2486 #[tokio::test]
2487 async fn metrics_disabled_binds_nothing() {
2488 let dir = temp_dir();
2489 let config = config_in(&dir, false);
2490 assert!(!config.metrics.enabled);
2491
2492 assert!(bind_metrics(&Arc::new(config)).await.unwrap().is_none());
2493 }
2494
2495 /// The admin listener's **own** TLS arm.
2496 ///
2497 /// `[server.tls]` and `[admin.tls]` are separate settings with separate
2498 /// certificate paths, on purpose — the two listeners answer to
2499 /// different names — and they go through separate `axum::serve` arms in
2500 /// `serve_admin`. `a_tls_server_answers_over_a_real_handshake` covers
2501 /// the ACME one; this one was the only `TlsListener::spawn` call site
2502 /// in the crate with no test at all, which for the listener that
2503 /// carries an operator's session cookie is the wrong one to miss.
2504 #[tokio::test]
2505 async fn the_admin_listener_answers_over_its_own_tls() {
2506 use tokio_rustls::TlsConnector;
2507 use tokio_rustls::rustls::pki_types::ServerName;
2508
2509 let dir = temp_dir();
2510 let mut config = config_in(&dir, false);
2511 config.admin.enabled = true;
2512 // A distinct certificate from the ACME listener's, which is the
2513 // whole reason these are two settings.
2514 config.admin.tls.enabled = true;
2515 config.admin.tls.cert_path = dir.as_ref().join("admin.pem").display().to_string();
2516 config.admin.tls.key_path = dir.as_ref().join("admin.key").display().to_string();
2517 // `check_config` refuses a non-loopback bind without TLS; with TLS
2518 // on it is allowed, and this exercises that branch too.
2519 config.admin.base_url = "https://localhost:3001".to_string();
2520
2521 let database = Arc::new(Database::connect_in_memory().await.unwrap());
2522 let acme_listener = TcpListener::bind("127.0.0.1:0").await.unwrap();
2523 let admin_listener = TcpListener::bind("127.0.0.1:0").await.unwrap();
2524 let acme_addr = acme_listener.local_addr().unwrap();
2525 let admin_addr = admin_listener.local_addr().unwrap();
2526
2527 let (tx, rx) = tokio::sync::oneshot::channel::<()>();
2528 let handle = tokio::spawn(serve_on_with(
2529 Arc::new(config),
2530 database,
2531 acme_listener,
2532 Some(admin_listener),
2533 // This test is about the admin listener's own TLS arm; the
2534 // metrics listener has none.
2535 None,
2536 async {
2537 let _ = rx.await;
2538 },
2539 ));
2540
2541 // The ACME socket is still cleartext — the two settings really are
2542 // independent, which a shared switch would hide.
2543 let acme = get(acme_addr, "/health").await;
2544 assert!(acme.starts_with("HTTP/1.1 200 OK"), "{acme}");
2545
2546 // The admin socket needs a handshake. Self-signed, so the client
2547 // verifies nothing: the listener is the subject, not the chain.
2548 let client =
2549 crate::challenge::tls_alpn_01::accept_any_client_config(&[b"http/1.1"]).unwrap();
2550 let stream = TcpStream::connect(admin_addr).await.unwrap();
2551 let mut tls = TlsConnector::from(client)
2552 .connect(ServerName::try_from("localhost").unwrap(), stream)
2553 .await
2554 .expect("the admin listener must complete a handshake");
2555 tls.write_all(
2556 b"GET /api/accounts HTTP/1.1\r\nHost: localhost\r\nConnection: close\r\n\r\n",
2557 )
2558 .await
2559 .unwrap();
2560 let mut response = String::new();
2561 tls.read_to_string(&mut response).await.unwrap();
2562 assert!(
2563 response.starts_with("HTTP/1.1 401"),
2564 "the admin API answers over TLS, unauthenticated: {response}"
2565 );
2566
2567 // The certificate really was written to the admin paths, not the
2568 // server's — a shared path would make one listener overwrite the
2569 // other's key on every start.
2570 assert!(dir.as_ref().join("admin.pem").exists());
2571 assert!(dir.as_ref().join("admin.key").exists());
2572 assert!(!dir.as_ref().join("server.pem").exists());
2573
2574 tx.send(()).unwrap();
2575 tokio::time::timeout(Duration::from_secs(10), handle)
2576 .await
2577 .expect("both listeners must stop")
2578 .unwrap()
2579 .expect("a clean shutdown is not an error");
2580 }
2581
2582 /// With `[admin]` off — the default — nothing is bound but the ACME
2583 /// socket, and the panel's routes do not exist anywhere.
2584 #[tokio::test]
2585 async fn the_admin_listener_is_absent_by_default() {
2586 let dir = temp_dir();
2587 let config = config_in(&dir, false);
2588 assert!(!config.admin.enabled, "the default must stay off");
2589 let (addr, shutdown, handle) = boot(config).await;
2590
2591 let response = get(addr, "/api/accounts").await;
2592 assert!(response.starts_with("HTTP/1.1 404"), "{response}");
2593
2594 shutdown.send(()).unwrap();
2595 handle.await.unwrap().unwrap();
2596 }
2597
2598 /// A `[admin]` section that cannot work stops the whole process before
2599 /// either socket is bound — it must not leave the ACME listener up and
2600 /// the panel silently missing.
2601 #[tokio::test]
2602 async fn an_invalid_admin_section_refuses_to_serve() {
2603 let dir = temp_dir();
2604 let mut config = config_in(&dir, false);
2605 config.admin.enabled = true;
2606 // Non-loopback with TLS off: the `Secure` cookie would never be
2607 // stored, so this is a hard startup error.
2608 config.admin.bind_address = "0.0.0.0:0".to_string();
2609
2610 let database = Arc::new(Database::connect_in_memory().await.unwrap());
2611 let listener = TcpListener::bind("127.0.0.1:0").await.unwrap();
2612 let error = serve_on(Arc::new(config), database, listener, std::future::ready(()))
2613 .await
2614 .expect_err("a panel that cannot work must not start");
2615 assert!(error.to_string().contains("is not loopback"), "{error}");
2616 }
2617
2618 /// Sends one request and returns the raw response.
2619 async fn get(addr: SocketAddr, path: &str) -> String {
2620 let mut stream = TcpStream::connect(addr).await.unwrap();
2621 stream
2622 .write_all(
2623 format!("GET {path} HTTP/1.1\r\nHost: localhost\r\nConnection: close\r\n\r\n")
2624 .as_bytes(),
2625 )
2626 .await
2627 .unwrap();
2628 let mut response = String::new();
2629 stream.read_to_string(&mut response).await.unwrap();
2630 response
2631 }
2632
2633 /// A configuration mounting nothing fails before the socket is ever
2634 /// served, rather than starting a server that answers 404 everywhere.
2635 #[tokio::test]
2636 async fn a_configuration_with_no_profile_refuses_to_serve() {
2637 let database = Arc::new(Database::connect_in_memory().await.unwrap());
2638 let listener = TcpListener::bind("127.0.0.1:0").await.unwrap();
2639 let error = serve_on(
2640 Arc::new(Config::default()),
2641 database,
2642 listener,
2643 std::future::ready(()),
2644 )
2645 .await
2646 .expect_err("a server with no endpoint must not start");
2647 assert!(error.to_string().contains("profile"), "{error}");
2648 }
2649
2650 /// `Serve` is a `dispatch` arm like any other: its failure travels back
2651 /// as a value instead of taking the process down where it happened.
2652 #[tokio::test]
2653 async fn dispatch_serve_reports_a_startup_failure() {
2654 let database = Arc::new(Database::connect_in_memory().await.unwrap());
2655 // Binds fine, but mounts no endpoint — so it fails inside
2656 // `serve_on` rather than at the socket.
2657 let mut config = Config::default();
2658 config.server.bind_address = "127.0.0.1:0".to_string();
2659 let mut reader: &[u8] = &[];
2660
2661 let error = dispatch(
2662 Some(Command::Serve),
2663 true,
2664 ColorChoice::Never,
2665 &mut reader,
2666 &Arc::new(config),
2667 database,
2668 )
2669 .await
2670 .expect_err("a server with no endpoint must not start");
2671 assert!(error.to_string().contains("profile"), "{error}");
2672 }
2673
2674 /// Unreadable TLS material stops startup instead of silently falling
2675 /// back to cleartext on a port operators believe is HTTPS.
2676 #[tokio::test]
2677 async fn unusable_tls_material_stops_startup() {
2678 let dir = temp_dir();
2679 let mut config = config_in(&dir, true);
2680 std::fs::write(dir.join("server.pem"), "not a certificate").unwrap();
2681 std::fs::write(dir.join("server.key"), "not a key").unwrap();
2682 config.server.tls.cert_path = dir.join("server.pem").display().to_string();
2683 config.server.tls.key_path = dir.join("server.key").display().to_string();
2684
2685 let database = Arc::new(Database::connect_in_memory().await.unwrap());
2686 let listener = TcpListener::bind("127.0.0.1:0").await.unwrap();
2687 let error = serve_on(Arc::new(config), database, listener, std::future::ready(()))
2688 .await
2689 .expect_err("unreadable TLS material must not start a server");
2690 assert!(!error.to_string().is_empty());
2691 }
2692
2693 /// `serve` binds `server.bind_address` itself, so an unusable one is
2694 /// reported rather than panicking.
2695 #[tokio::test]
2696 async fn an_unbindable_address_is_reported() {
2697 let database = Arc::new(Database::connect_in_memory().await.unwrap());
2698 let mut config = Config::default();
2699 // A port on an address this process does not hold.
2700 config.server.bind_address = "192.0.2.1:1".to_string();
2701 let error = serve(Arc::new(config), database)
2702 .await
2703 .expect_err("binding an unroutable address must fail");
2704 assert!(error.to_string().contains("192.0.2.1:1"), "{error}");
2705 }
2706 }
2707}