acme_proxy_server/generation.rs
1//! One configuration generation: built and validated in full, then published
2//! in one uninterruptible run. Startup builds the first; a reload builds and
3//! publishes every later one.
4
5use std::sync::Arc;
6use std::time::Duration;
7
8use tracing::{error, info, warn};
9
10use super::logging;
11use acme_proxy_core::config::Config;
12use acme_proxy_net::tls;
13
14use super::sockets::{Role, SocketPlans, check_metrics_config, plan_sockets};
15use super::supervisor::Cells;
16use super::{Assembly, GenerationParts};
17use acme_proxy_protocol::profile::Profile;
18use acme_proxy_protocol::router::build_app;
19
20/// Everything one configuration generation contributes, built and validated
21/// before any of it is published.
22///
23/// The unit exists so startup and reload cannot drift: both go through
24/// [`build_generation`], so a subsystem added to one is added to the other by
25/// construction rather than by remembering.
26pub(crate) struct Generation {
27 /// Kept so startup can announce them; a reload drops them on the floor,
28 /// since `profile_mounted` is a lifecycle event and not a heartbeat.
29 pub(super) profiles: Vec<Arc<Profile>>,
30 pub(super) acme_app: axum::Router,
31 pub(super) admin_app: Option<axum::Router>,
32 pub(super) job_registry: acme_proxy_jobs::jobs::JobRegistry,
33 pub(super) tls: Option<tls::TlsSettings>,
34 pub(super) admin_tls: Option<tls::TlsSettings>,
35 /// The limiter this generation ended up with, for the next one to carry.
36 pub(super) logins: Option<Arc<acme_proxy_admin::webadmin::LoginLimiter>>,
37}
38
39/// Builds one generation: profiles, both routers, the job registry, the audit
40/// trail and both TLS acceptors.
41///
42/// Fallible throughout and side-effect-free on the *serving* state: nothing here
43/// touches a cell, so a failure leaves whatever is already running exactly as it
44/// was. That is what makes an atomic reload possible — everything is built
45/// first, and only a complete success publishes anything.
46///
47/// `dispatchers` is passed in rather than built here because a reload needs to
48/// hold it back: it is published to the long-lived [`acme_proxy_jobs::notify::Notifiers`]
49/// handle at swap time, *before* the routers, so a request served by the new
50/// generation cannot queue a delivery the job runner's map does not know.
51///
52/// Whether the panel is part of this generation is read from `admin.enabled`
53/// rather than from whether a socket exists. It used to be the latter, which
54/// was the same answer while the key was frozen and is the wrong one now that a
55/// reload can turn the panel on: the app, its TLS, its session sweep and its
56/// login limiter all have to appear in the generation *before* there is a
57/// listener to serve them.
58pub(crate) fn build_generation(
59 roles: crate::RoleSet,
60 config: &Arc<Config>,
61 resolved: &[acme_proxy_core::config::ProfileConfig],
62 assembly: &Assembly,
63 parts: &GenerationParts,
64 previous_logins: Option<&acme_proxy_admin::webadmin::LoginLimiter>,
65) -> anyhow::Result<Generation> {
66 // A process that does not run the `admin` role binds no admin socket
67 // (`sockets::bind_admin`), so building its router, its TLS and its login
68 // limiter would be work for a listener that will never exist — and it would
69 // make `[admin]` settings this process never serves able to fail its
70 // reloads.
71 let admin_enabled = config.admin.enabled && roles.has(crate::ProcessRole::Admin);
72 let database = assembly.database.clone();
73 let profiles = crate::profile::build_all_with(config, resolved, parts)?;
74
75 let tls = tls::from_config(&config.server)
76 .inspect_err(|error| {
77 error!(event = "tls_init_failed", outcome = "failure", error = %error);
78 })?
79 .map(|acceptor| {
80 tls::TlsSettings::new(
81 acceptor,
82 Duration::from_millis(config.server.tls.handshake_timeout_ms),
83 )
84 });
85
86 let admin_tls = match admin_enabled {
87 false => None,
88 true => tls::admin_from_config(&config.admin)
89 .inspect_err(|error| {
90 error!(event = "admin_tls_init_failed", outcome = "failure", error = %error);
91 })?
92 .map(|acceptor| {
93 tls::TlsSettings::new(
94 acceptor,
95 Duration::from_millis(config.admin.tls.handshake_timeout_ms),
96 )
97 }),
98 };
99
100 let job_registry =
101 job_registry_for(config, resolved, assembly, parts, &profiles, admin_enabled)?;
102
103 // The CA's audit trail: one per process, shared by every profile's router
104 // and by the web admin listener, because `[audit]` is process-wide. Built
105 // here rather than in `server::profile::build_all` for exactly that reason — it is
106 // not a per-endpoint subsystem.
107 let auditor = Arc::new(
108 acme_proxy_jobs::auditor::Auditor::from_config(
109 &config.audit,
110 &config.dns,
111 database.clone(),
112 // The registry the certificate counters land in. Carried by
113 // `Assembly`, so it is the same one across every generation and the
114 // same one `metrics_app` serves from.
115 assembly.metrics.clone(),
116 )
117 .inspect_err(|error| {
118 error!(event = "audit_init_failed", outcome = "failure", error = %error);
119 })?,
120 );
121
122 // Built **before** `build_app`, which consumes `profiles`. The admin state
123 // needs the same profiles (revoking an order resolves that order's own
124 // signer), and `build_admin_app` takes a slice precisely so the ordering
125 // is a signature constraint rather than a borrow error to rediscover.
126 let (admin_app, logins) = match admin_enabled {
127 false => (None, None),
128 true => {
129 // Built again, although `check_config` already did: that call is
130 // the refusal, this one is the policy the router keeps.
131 let policy =
132 acme_proxy_admin::webadmin::filter::build(config).inspect_err(|error| {
133 error!(event = "admin_filter_init_failed", outcome = "failure", error = %error);
134 })?;
135 let (router, logins) = acme_proxy_admin::webadmin::build_admin_app_with_logins(
136 database.clone(),
137 config.clone(),
138 &profiles,
139 auditor.clone(),
140 assembly.notifiers.clone(),
141 assembly.jobs.clone(),
142 policy,
143 previous_logins,
144 );
145 (Some(router), Some(logins))
146 }
147 };
148 let acme_app = build_app(
149 database,
150 config.clone(),
151 profiles.clone(),
152 auditor,
153 assembly.metrics.clone(),
154 assembly.jobs.clone(),
155 );
156
157 Ok(Generation {
158 profiles,
159 acme_app,
160 admin_app,
161 job_registry,
162 tls,
163 admin_tls,
164 logins,
165 })
166}
167
168/// Every subsystem with background work, registered for one generation.
169///
170/// A sweep or handler that is never registered never runs, and nothing else
171/// notices: the table it should prune just grows. Which ones a configuration
172/// registers is pinned by `tests::a_generation_registers_the_sweeps_its_configuration_asks_for`.
173fn job_registry_for(
174 config: &Config,
175 resolved: &[acme_proxy_core::config::ProfileConfig],
176 assembly: &Assembly,
177 parts: &GenerationParts,
178 profiles: &[Arc<Profile>],
179 admin_enabled: bool,
180) -> anyhow::Result<acme_proxy_jobs::jobs::JobRegistry> {
181 let database = assembly.database.clone();
182 // Every subsystem with background work, in one registry. **Nothing here
183 // registers a handler per backend**: the registry refuses a second handler
184 // for one kind outright, since two would each claim about half the rows, and
185 // two profiles over different `[signer]` sections are two backends that
186 // `build_backends` deliberately does not collapse. So each backend hands
187 // over *state* and one handler is built over all of it.
188 let mut job_registry = acme_proxy_jobs::jobs::JobRegistry::new();
189 // The CRLs are collected once per distinct backend, since two profiles
190 // sharing one CA share one CRL and refreshing it twice a day would be
191 // pointless work. The identity is kept as a `usize` rather than the
192 // pointer itself, so this function's caller stays `Send`; it is spawned.
193 let mut registered: Vec<usize> = Vec::new();
194 let mut refreshers: Vec<Arc<dyn acme_proxy_signer::CrlRefresher>> = Vec::new();
195 // The relay backends are collected per *profile*, because that is the key a
196 // job row is dispatched on — and because taking the profile list from the
197 // backend would take a stale one: a backend whose configuration did not
198 // move is reused verbatim across a reload, so a profile newly mounted onto
199 // it is not in any list it remembers.
200 //
201 // Both come from this generation's **backends**, which only a process
202 // running the `worker` role builds: everywhere else the set is empty, so
203 // none of these handlers has a backend to reach, and none of them runs
204 // there anyway.
205 let mut relays: Vec<(String, acme_proxy_signer::relay::RelayState)> = Vec::new();
206 let mut backends = parts.signers.by_profile();
207 backends.sort_by(|a, b| a.0.cmp(&b.0));
208 for (profile, backend) in &backends {
209 relays.extend(backend.relay_state().map(|state| (profile.clone(), state)));
210 let identity = Arc::as_ptr(backend).cast::<()>() as usize;
211 if registered.contains(&identity) {
212 continue;
213 }
214 registered.push(identity);
215 refreshers.extend(backend.crl_refresher());
216 }
217 // The daily CRL refresh, over whichever CAs keep a CRL of their own.
218 // Registered only when there is one, the way the audit sweep is registered
219 // only for a non-zero retention.
220 // Beside it, the handler that signs revocations the CLI recorded without
221 // the key.
222 if !refreshers.is_empty() {
223 job_registry
224 .register(Arc::new(
225 acme_proxy_signer::local_ca::sweep::CrlRegenerateJob::new(refreshers.clone()),
226 ))
227 .inspect_err(|error| {
228 error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
229 })?;
230 job_registry
231 .register(Arc::new(
232 acme_proxy_signer::local_ca::sweep::CrlSweepJob::new(refreshers),
233 ))
234 .inspect_err(|error| {
235 error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
236 })?;
237 }
238 // Relayed issuance, over every relay profile at once. Registered only when
239 // some profile relays, for `CrlSweepJob`'s reason: a deployment with none
240 // has no row of this kind to claim.
241 if !relays.is_empty() {
242 job_registry
243 .register(Arc::new(acme_proxy_signer::relay::flow::RelayJob::new(
244 database.clone(),
245 relays,
246 )))
247 .inspect_err(|error| {
248 error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
249 })?;
250 }
251 // Revocations the host CLI queued for a backend it does not build (`relay`,
252 // `custom`). Registered unconditionally, `NotifyJob`'s reason below: a row
253 // queued before a configuration change must still find a handler, and one
254 // naming a profile this generation does not mount is retried rather than
255 // lost.
256 job_registry
257 .register(Arc::new(
258 acme_proxy_protocol::acme::revoke::SignerRevokeJob::new(
259 database.clone(),
260 Arc::new(
261 acme_proxy_jobs::auditor::Auditor::offline(database.clone())
262 .with_metrics(assembly.metrics.clone()),
263 ),
264 backends.clone(),
265 assembly.notifiers.clone(),
266 ),
267 ))
268 .inspect_err(|error| {
269 error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
270 })?;
271 // Issuance. `finalize` claims the order and queues the signing, so the
272 // process answering ACME holds no backend. Registered unconditionally for
273 // `SignerRevokeJob`'s reason above, over this generation's backends — none
274 // at all in a process without the worker role, which never claims a row.
275 job_registry
276 .register(Arc::new(
277 acme_proxy_protocol::acme::issue::SignerIssueJob::new(
278 database.clone(),
279 Arc::new(
280 acme_proxy_jobs::auditor::Auditor::offline(database.clone())
281 .with_metrics(assembly.metrics.clone()),
282 ),
283 backends.clone(),
284 assembly.notifiers.clone(),
285 ),
286 ))
287 .inspect_err(|error| {
288 error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
289 })?;
290 // Challenge validation. `POST /chall/{id}` claims the challenge and queues
291 // the outbound check, so the probe of a client-chosen host no longer holds
292 // an admission permit for the length of `challenge.timeout_ms`. Registered
293 // unconditionally for `SignerRevokeJob`'s reason above — a row queued
294 // before a configuration change must still find a handler — and holding the
295 // profiles rather than one profile, since the registry refuses a second
296 // handler for one kind.
297 job_registry
298 .register(Arc::new(
299 acme_proxy_protocol::acme::validate::ChallengeValidateJob::new(
300 database.clone(),
301 Arc::new(
302 acme_proxy_jobs::auditor::Auditor::offline(database.clone())
303 .with_metrics(assembly.metrics.clone()),
304 ),
305 profiles
306 .iter()
307 .map(|profile| (profile.name.clone(), profile.clone()))
308 .collect(),
309 ),
310 ))
311 .inspect_err(|error| {
312 error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
313 })?;
314
315 // Notification delivery. The same shape as the two above and the reason it
316 // is: one handler for every profile, holding the whole
317 // `profile name -> dispatcher` map, with a job row naming its own profile.
318 // Registered unconditionally — a profile with no `[notify]`
319 // backends queues nothing, so the handler simply never claims a row, and
320 // making the registration conditional would mean a row queued before a
321 // configuration change had nobody to run it.
322 //
323 // It takes the *handle*, not this generation's map: the handler is
324 // registered per generation but must read whichever map is current, and a
325 // row queued by a reloaded router names a slot id only the new one has.
326 job_registry
327 .register(Arc::new(acme_proxy_jobs::notify::NotifyJob::new(
328 assembly.notifiers.clone(),
329 )))
330 .inspect_err(|error| {
331 error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
332 })?;
333
334 // The expiry digest, registered only when some profile asked for one
335 // (`notify.expiry.lead_days`), the way `CrlSweepJob` is registered only
336 // when there is a ledger to prune. It takes the `Notifiers` handle rather
337 // than the profiles' own dispatchers for `NotifyJob`'s reason above, and
338 // the queue because its per-profile rows are something it maintains on
339 // every pass rather than only at `recover`.
340 if let Some(digest) = acme_proxy_jobs::notify::expiry::ExpiryDigestJob::from_profiles(
341 resolved,
342 assembly.notifiers.clone(),
343 database.clone(),
344 assembly.jobs.clone(),
345 ) {
346 job_registry
347 .register(Arc::new(digest))
348 .inspect_err(|error| {
349 error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
350 })?;
351 }
352
353 // The periodic table sweeps. Each is one self-rescheduling row rather than
354 // its own interval loop, so a sweep that dies is reclaimed by lease expiry
355 // and its schedule survives a restart. Their `recover` is also the startup
356 // sweep — it queues at `run_at = now`, so the runner performs the first pass
357 // on its way into the loop and there is nothing to run separately here.
358 let ttl = Duration::from_secs(config.nonce.ttl_seconds);
359 let mut sweeps = vec![acme_proxy_jobs::jobs::SweepJob::nonces(
360 database.clone(),
361 ttl,
362 )];
363 // `0` keeps everything for ever on both of these, and is a handler not
364 // registered rather than a sweep with a cutoff at the epoch.
365 if config.audit.retention_days > 0 {
366 sweeps.push(acme_proxy_jobs::jobs::SweepJob::audit(
367 database.clone(),
368 config.audit.retention_days,
369 ));
370 }
371 if config.jobs.retention_days > 0 {
372 sweeps.push(acme_proxy_jobs::jobs::SweepJob::jobs(
373 database.clone(),
374 config.jobs.retention_days,
375 ));
376 }
377 // One handler covering every mounted profile, since the registry refuses a
378 // second handler for one kind. A profile keeping everything (`0`) is left
379 // out of the list rather than swept with a cutoff at the epoch.
380 let order_retention: Vec<(String, u64)> = resolved
381 .iter()
382 .filter(|profile| profile.sections.order.retention_days > 0)
383 .map(|profile| (profile.name.clone(), profile.sections.order.retention_days))
384 .collect();
385 if !order_retention.is_empty() {
386 sweeps.push(acme_proxy_jobs::jobs::SweepJob::orders(
387 database.clone(),
388 order_retention,
389 ));
390 }
391 // Only where some backend publishes http-01 tokens: the table stays empty
392 // otherwise, and the CRL refresh's rule applies.
393 if profiles
394 .iter()
395 .any(|profile| profile.signer_info.http01_tokens().is_some())
396 {
397 sweeps.push(acme_proxy_jobs::jobs::SweepJob::http01_tokens(
398 database.clone(),
399 ));
400 }
401 if admin_enabled {
402 sweeps.push(acme_proxy_jobs::jobs::SweepJob::admin_sessions(
403 database.clone(),
404 Duration::from_secs(config.admin.session_idle_timeout_seconds),
405 config.admin.session_ttl_seconds,
406 ));
407 }
408 for sweep in sweeps {
409 job_registry
410 .register(Arc::new(sweep))
411 .inspect_err(|error| {
412 error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
413 })?;
414 }
415 Ok(job_registry)
416}
417
418/// What one successful reload hands back to the supervisor: the report to log,
419/// and the pieces of state the *next* reload compares against.
420///
421/// A struct rather than the tuple this used to return — which needed
422/// `#[allow(clippy::type_complexity)]` and left the caller destructuring four
423/// same-shaped values positionally, where swapping two would still compile.
424/// The same move `ProfileParts` made, for the same reason.
425pub(super) struct Reloaded {
426 pub(super) report: crate::reload::ReloadReport,
427 pub(super) config: Arc<Config>,
428 pub(super) resolved: Vec<acme_proxy_core::config::ProfileConfig>,
429 pub(super) logins: Option<Arc<acme_proxy_admin::webadmin::LoginLimiter>>,
430 /// Each socket this reload bound, with the address it landed on. Announced
431 /// by the supervisor rather than here, because saying a listener is up
432 /// reaches the database (the panel's "nobody can sign in yet" warning) and
433 /// [`publish_reload`] has no await point to spend on it — deliberately, that
434 /// being what keeps its publishing run uninterruptible.
435 pub(super) opened: Vec<(Role, String)>,
436 /// The endpoints this reload **added**, for the same reason and announced in
437 /// the same place: `profile_mounted` is dispatched to the `[notify]`
438 /// backends, which queues a job row.
439 pub(super) mounted: Vec<Arc<Profile>>,
440}
441
442/// Everything a reload built and validated, waiting to be published.
443///
444/// The build/publish split is the whole shape of a reload, and making it two
445/// values rather than two halves of one function buys the thing the split was
446/// always claiming: [`prepare_reload`] can run wherever it likes — it runs on a
447/// blocking thread, since building a `relay` backend can contact its upstream —
448/// while [`publish_reload`] stays on the supervisor task, where having no await
449/// point is what makes a generation unobservable half-applied.
450pub(super) struct Prepared {
451 config: Arc<Config>,
452 resolved: Vec<acme_proxy_core::config::ProfileConfig>,
453 parts: GenerationParts,
454 generation: Generation,
455 sockets: SocketPlans,
456 logging: logging::PreparedLogging,
457 /// Whether `RUST_LOG` is what the filter came from, so the publish phase can
458 /// say when an edited `logging.filter` changed nothing, and which of the
459 /// two outranking layers is why.
460 logging_filter_source: logging::FilterSource,
461 /// The endpoints in this generation that the previous one did not mount.
462 mounted: Vec<Arc<Profile>>,
463 /// The endpoints the previous generation mounted and this one does not.
464 unmounted: Vec<String>,
465}
466
467impl Prepared {
468 /// The backends this reload built or carried, for the one step the
469 /// supervisor takes between building and publishing.
470 pub(super) fn signers(&self) -> &acme_proxy_signer::SignerSet {
471 &self.parts.signers
472 }
473}
474
475/// The build half of one reload: everything that can fail, and everything that
476/// can block.
477///
478/// Nothing here touches a cell, so a failure anywhere leaves the running
479/// generation exactly as it was — which is the property the whole "atomic,
480/// refuse by name" decision exists for. Every socket this reload needs is bound
481/// here too, where a port already in use is still a refusal rather than a
482/// listener already dropped.
483///
484/// Run on a blocking thread by
485/// [`supervise_reloads`](super::supervisor::supervise_reloads), because
486/// building a `relay` backend for the first time contacts its upstream
487/// synchronously.
488pub(super) fn prepare_reload(
489 roles: crate::RoleSet,
490 config: &Arc<Config>,
491 resolved: &[acme_proxy_core::config::ProfileConfig],
492 assembly: &Assembly,
493 logins: Option<&acme_proxy_admin::webadmin::LoginLimiter>,
494) -> Result<Prepared, crate::reload::ReloadError> {
495 use crate::reload::{Applied, ReloadError, check_frozen};
496
497 // Re-read from scratch: `Config::load` consults the file *and* the
498 // `ACME_PROXY_*` environment, so a reload sees whatever the process would
499 // see if it restarted right now.
500 let next = Arc::new(Config::load().map_err(|error| ReloadError::Load(error.to_string()))?);
501 let next_resolved = next
502 .resolve_profiles()
503 .map_err(|error| ReloadError::Load(error.to_string()))?;
504
505 check_frozen(
506 &Applied {
507 config,
508 profiles: resolved,
509 },
510 &Applied {
511 config: &next,
512 profiles: &next_resolved,
513 },
514 )?;
515
516 // Built here rather than published here: a bad `logging.target` must refuse
517 // the whole reload with the message startup would have printed, not leave a
518 // half-swapped generation behind. The same build-then-publish split
519 // `Assembly::build_parts` makes.
520 // `flag_override()` is how a `--log-level` typed at startup survives a
521 // `SIGHUP`: the stack is rebuilt from the file, so without re-reading it
522 // the reload would silently demote the server to `logging.filter`.
523 let logging = logging::prepare_logging(&next.logging, logging::flag_override())
524 .map_err(ReloadError::Build)?;
525 let logging_filter_source = logging.filter_source;
526
527 // The same validation startup runs before either socket binds, so a panel
528 // that would refuse to start refuses to be reloaded into. It also compiles
529 // every `admin.template_dir` override, which is what keeps a broken one a
530 // failed reload rather than a 500 in a browser.
531 // Only where this process serves the panel, exactly as startup checks it
532 // (`sockets::bind_admin`). An acme-only or worker-only process that can
533 // start with an `[admin]` section it never serves must not then refuse
534 // every `SIGHUP` over it.
535 if roles.has(crate::ProcessRole::Admin) {
536 acme_proxy_admin::webadmin::check_config(&next)
537 .map_err(|error| ReloadError::Build(error.to_string()))?;
538 }
539 // Its twin for the third listener: `webadmin::check_config` sees the
540 // admin-versus-server pair, this one sees the two it cannot.
541 check_metrics_config(&next).map_err(|error| ReloadError::Build(error.to_string()))?;
542
543 // Every socket this reload needs is bound **here**, where a failure is still
544 // a refusal: a port already taken, an address that does not resolve, a
545 // privileged port after a `setcap` was lost. Past the publish phase nothing
546 // can fail, so the running listeners are never dropped for a configuration
547 // that then turns out not to work.
548 let sockets = plan_sockets(roles, config, &next)?;
549
550 // The egress clients, the notification dispatchers and the signer backends.
551 // The last is where a newly mounted endpoint gets a backend, a removed one's
552 // is left out, and an edited `[signer]` is rebuilt over the live instance's
553 // in-memory state — see `signer::build_backends`. It is also the one step
554 // that can make a network call, hence this whole function's blocking thread.
555 let parts = assembly
556 .build_parts(&next_resolved, &next)
557 .map_err(|error| ReloadError::Build(error.to_string()))?;
558 let generation = build_generation(roles, &next, &next_resolved, assembly, &parts, logins)
559 .map_err(|error| ReloadError::Build(error.to_string()))?;
560
561 // Compared by name against what is running, not against what is written
562 // down: `resolve_profiles` has already dropped every `enabled = false`
563 // entry, so this is the set of endpoints actually served.
564 let running: std::collections::HashSet<&str> = resolved
565 .iter()
566 .map(|profile| profile.name.as_str())
567 .collect();
568 let mounted = generation
569 .profiles
570 .iter()
571 .filter(|profile| !running.contains(profile.name.as_str()))
572 .cloned()
573 .collect();
574 let next_names: std::collections::HashSet<&str> = next_resolved
575 .iter()
576 .map(|profile| profile.name.as_str())
577 .collect();
578 let unmounted = resolved
579 .iter()
580 .map(|profile| profile.name.clone())
581 .filter(|name| !next_names.contains(name.as_str()))
582 .collect();
583
584 Ok(Prepared {
585 config: next,
586 resolved: next_resolved,
587 parts,
588 generation,
589 sockets,
590 logging,
591 logging_filter_source,
592 mounted,
593 unmounted,
594 })
595}
596
597/// The publish half: infallible, synchronous, and uninterruptible.
598///
599/// There is no `.await` in here, and that is load-bearing rather than
600/// incidental. `watch::Sender::send_replace` and
601/// `mpsc::UnboundedSender::send` are both synchronous, so a run of them with no
602/// await point between cannot be interleaved — no task can observe a generation
603/// half-applied, and no lock is needed to say so.
604///
605/// The order matters in three places. `[logging]` goes **first**, because an
606/// operator who raised the level did it to see what happens next, starting with
607/// this reload's own completion line. The notifier map, the signer set and the
608/// job registry go **before** the routers: a request served by the new
609/// generation queues a `notify_deliver` row naming a slot id from the new
610/// configuration, and a `NotifyJob` still holding the old map would retire it —
611/// permanently, since an unknown backend id is a `Failed`, not a `Retry`. And
612/// the TLS mode goes before the socket, so a freshly bound listener's very first
613/// connection is already accepted under this generation's settings.
614pub(super) fn publish_reload(
615 prepared: Prepared,
616 applied: &Arc<Config>,
617 assembly: &Assembly,
618 cells: &Cells,
619 generation: u64,
620 started: std::time::Instant,
621) -> Reloaded {
622 use crate::reload::ReloadReport;
623
624 let Prepared {
625 config: next,
626 resolved: next_resolved,
627 parts,
628 generation: built,
629 sockets,
630 logging,
631 logging_filter_source,
632 mounted,
633 unmounted,
634 } = prepared;
635
636 let logging_reloaded = logging::publish_logging(logging);
637
638 let report = ReloadReport {
639 generation,
640 profiles: built
641 .profiles
642 .iter()
643 .map(|profile| profile.name.clone())
644 .collect(),
645 job_kinds: built.job_registry.kinds(),
646 tls_reloaded: built.tls.is_some(),
647 admin_tls_reloaded: built.admin_tls.is_some(),
648 listeners_rebound: sockets.rebound(),
649 logging_reloaded,
650 duration: started.elapsed(),
651 };
652 let next_logins = built.logins.clone();
653
654 assembly.publish_notifiers(parts.dispatchers);
655 // The set the *next* reload compares against, and the point at which the
656 // backends this one dropped are finally released — after their replacements
657 // were built and adopted their state, never before.
658 assembly.publish_signers(parts.signers, parts.infos);
659 cells
660 .job_registry
661 .send_replace(Arc::new(built.job_registry));
662
663 // `[jobs]` in its two halves, both synchronous and neither able to fail —
664 // which is what lets them sit in this run rather than needing a build phase
665 // of their own. The runner re-derives its pacing from the cell on its next
666 // pass; `max_attempts` goes to the queue instead, because it is the enqueue
667 // side that reads it, and it sets the budget for work queued from here on
668 // rather than for the rows already waiting.
669 cells.jobs.send_replace(Arc::new(next.jobs.clone()));
670 assembly.jobs.set_max_attempts(next.jobs.max_attempts);
671
672 cells.acme.set_tls(built.tls);
673 cells.admin.set_tls(built.admin_tls);
674 let opened = sockets.bound.clone();
675 sockets.publish(cells);
676
677 cells
678 .acme_router
679 .send_replace(built.acme_app.into_service::<axum::body::Body>());
680 // An empty router when the panel is off, which is what a request arriving
681 // on a connection established a moment before it was switched off now gets:
682 // closing the socket stops the next client, and this stops that one.
683 cells.admin_router.send_replace(
684 built
685 .admin_app
686 .unwrap_or_default()
687 .into_service::<axum::body::Body>(),
688 );
689
690 // Said here rather than by the supervisor because it needs nothing but a
691 // name, unlike the mounting half, which dispatches a notification.
692 for profile in unmounted {
693 warn!(
694 event = "profile_unmounted",
695 outcome = "advisory",
696 profile = %profile,
697 "the endpoint is no longer served: its accounts and orders stay in the \
698 database and come back if it is mounted again, but any issuance still in \
699 flight for it has no handler left to finish it"
700 );
701 }
702
703 // `--log-level` and `RUST_LOG` both outrank `logging.filter` on a reload
704 // exactly as they do at startup — the two disagreeing would be worse — but
705 // that makes an edited `logging.filter` a silent no-op, which is the one
706 // outcome an operator would read as "my reload did not land". Said only
707 // when both halves hold: something outranked the file, *and* the file's
708 // filter actually moved. `source` names which, since the two are unset in
709 // different places.
710 if logging_filter_source.outranks_config() && applied.logging.filter != next.logging.filter {
711 warn!(
712 event = "server_logging_filter_overridden",
713 outcome = "advisory",
714 source = logging_filter_source.as_str(),
715 configured = %next.logging.filter,
716 );
717 }
718
719 Reloaded {
720 report,
721 config: next,
722 resolved: next_resolved,
723 logins: next_logins,
724 opened,
725 mounted,
726 }
727}
728
729/// The one announcement an endpoint that has just come up makes: a log line and
730/// a `[notify]` lifecycle event.
731///
732/// Shared by startup and by a reload that mounted a new endpoint, so the two
733/// cannot drift — before the profile set could reload there was only one caller
734/// and the sharing was not needed.
735pub(super) async fn announce_profile(profile: &Arc<Profile>) {
736 info!(
737 event = "profile_mounted",
738 outcome = "success",
739 profile = %profile.name,
740 directory = %profile.directory_url(),
741 challenge_bypass = profile.challenges.is_bypassed(),
742 eab_enabled = profile.eab.enabled
743 );
744 profile
745 .notify
746 .dispatch(acme_proxy_jobs::notify::NotifyEvent::ProfileMounted(
747 acme_proxy_jobs::notify::ProfileMountedData {
748 profile: profile.name.clone(),
749 },
750 ))
751 .await;
752}