Skip to main content

acme_proxy_server/
generation.rs

1//! One configuration generation: built and validated in full, then published
2//! in one uninterruptible run. Startup builds the first; a reload builds and
3//! publishes every later one.
4
5use std::sync::Arc;
6use std::time::Duration;
7
8use tracing::{error, info, warn};
9
10use super::logging;
11use acme_proxy_core::config::Config;
12use acme_proxy_net::tls;
13
14use super::sockets::{Role, SocketPlans, check_metrics_config, plan_sockets};
15use super::supervisor::Cells;
16use super::{Assembly, GenerationParts};
17use acme_proxy_protocol::profile::Profile;
18use acme_proxy_protocol::router::build_app;
19
20/// Everything one configuration generation contributes, built and validated
21/// before any of it is published.
22///
23/// The unit exists so startup and reload cannot drift: both go through
24/// [`build_generation`], so a subsystem added to one is added to the other by
25/// construction rather than by remembering.
26pub(crate) struct Generation {
27    /// Kept so startup can announce them; a reload drops them on the floor,
28    /// since `profile_mounted` is a lifecycle event and not a heartbeat.
29    pub(super) profiles: Vec<Arc<Profile>>,
30    pub(super) acme_app: axum::Router,
31    pub(super) admin_app: Option<axum::Router>,
32    pub(super) job_registry: acme_proxy_jobs::jobs::JobRegistry,
33    pub(super) tls: Option<tls::TlsSettings>,
34    pub(super) admin_tls: Option<tls::TlsSettings>,
35    /// The limiter this generation ended up with, for the next one to carry.
36    pub(super) logins: Option<Arc<acme_proxy_admin::webadmin::LoginLimiter>>,
37}
38
39/// Builds one generation: profiles, both routers, the job registry, the audit
40/// trail and both TLS acceptors.
41///
42/// Fallible throughout and side-effect-free on the *serving* state: nothing here
43/// touches a cell, so a failure leaves whatever is already running exactly as it
44/// was. That is what makes an atomic reload possible — everything is built
45/// first, and only a complete success publishes anything.
46///
47/// `dispatchers` is passed in rather than built here because a reload needs to
48/// hold it back: it is published to the long-lived [`acme_proxy_jobs::notify::Notifiers`]
49/// handle at swap time, *before* the routers, so a request served by the new
50/// generation cannot queue a delivery the job runner's map does not know.
51///
52/// Whether the panel is part of this generation is read from `admin.enabled`
53/// rather than from whether a socket exists. It used to be the latter, which
54/// was the same answer while the key was frozen and is the wrong one now that a
55/// reload can turn the panel on: the app, its TLS, its session sweep and its
56/// login limiter all have to appear in the generation *before* there is a
57/// listener to serve them.
58pub(crate) fn build_generation(
59    roles: crate::RoleSet,
60    config: &Arc<Config>,
61    resolved: &[acme_proxy_core::config::ProfileConfig],
62    assembly: &Assembly,
63    parts: &GenerationParts,
64    previous_logins: Option<&acme_proxy_admin::webadmin::LoginLimiter>,
65) -> anyhow::Result<Generation> {
66    // A process that does not run the `admin` role binds no admin socket
67    // (`sockets::bind_admin`), so building its router, its TLS and its login
68    // limiter would be work for a listener that will never exist — and it would
69    // make `[admin]` settings this process never serves able to fail its
70    // reloads.
71    let admin_enabled = config.admin.enabled && roles.has(crate::ProcessRole::Admin);
72    let database = assembly.database.clone();
73    let profiles = crate::profile::build_all_with(config, resolved, parts)?;
74
75    let tls = tls::from_config(&config.server)
76        .inspect_err(|error| {
77            error!(event = "tls_init_failed", outcome = "failure", error = %error);
78        })?
79        .map(|acceptor| {
80            tls::TlsSettings::new(
81                acceptor,
82                Duration::from_millis(config.server.tls.handshake_timeout_ms),
83            )
84        });
85
86    let admin_tls = match admin_enabled {
87        false => None,
88        true => tls::admin_from_config(&config.admin)
89            .inspect_err(|error| {
90                error!(event = "admin_tls_init_failed", outcome = "failure", error = %error);
91            })?
92            .map(|acceptor| {
93                tls::TlsSettings::new(
94                    acceptor,
95                    Duration::from_millis(config.admin.tls.handshake_timeout_ms),
96                )
97            }),
98    };
99
100    let job_registry =
101        job_registry_for(config, resolved, assembly, parts, &profiles, admin_enabled)?;
102
103    // The CA's audit trail: one per process, shared by every profile's router
104    // and by the web admin listener, because `[audit]` is process-wide. Built
105    // here rather than in `server::profile::build_all` for exactly that reason — it is
106    // not a per-endpoint subsystem.
107    let auditor = Arc::new(
108        acme_proxy_jobs::auditor::Auditor::from_config(
109            &config.audit,
110            &config.dns,
111            database.clone(),
112            // The registry the certificate counters land in. Carried by
113            // `Assembly`, so it is the same one across every generation and the
114            // same one `metrics_app` serves from.
115            assembly.metrics.clone(),
116        )
117        .inspect_err(|error| {
118            error!(event = "audit_init_failed", outcome = "failure", error = %error);
119        })?,
120    );
121
122    // Built **before** `build_app`, which consumes `profiles`. The admin state
123    // needs the same profiles (revoking an order resolves that order's own
124    // signer), and `build_admin_app` takes a slice precisely so the ordering
125    // is a signature constraint rather than a borrow error to rediscover.
126    let (admin_app, logins) = match admin_enabled {
127        false => (None, None),
128        true => {
129            // Built again, although `check_config` already did: that call is
130            // the refusal, this one is the policy the router keeps.
131            let policy =
132                acme_proxy_admin::webadmin::filter::build(config).inspect_err(|error| {
133                    error!(event = "admin_filter_init_failed", outcome = "failure", error = %error);
134                })?;
135            let (router, logins) = acme_proxy_admin::webadmin::build_admin_app_with_logins(
136                database.clone(),
137                config.clone(),
138                &profiles,
139                auditor.clone(),
140                assembly.notifiers.clone(),
141                assembly.jobs.clone(),
142                policy,
143                previous_logins,
144            );
145            (Some(router), Some(logins))
146        }
147    };
148    let acme_app = build_app(
149        database,
150        config.clone(),
151        profiles.clone(),
152        auditor,
153        assembly.metrics.clone(),
154        assembly.jobs.clone(),
155    );
156
157    Ok(Generation {
158        profiles,
159        acme_app,
160        admin_app,
161        job_registry,
162        tls,
163        admin_tls,
164        logins,
165    })
166}
167
168/// Every subsystem with background work, registered for one generation.
169///
170/// A sweep or handler that is never registered never runs, and nothing else
171/// notices: the table it should prune just grows. Which ones a configuration
172/// registers is pinned by `tests::a_generation_registers_the_sweeps_its_configuration_asks_for`.
173fn job_registry_for(
174    config: &Config,
175    resolved: &[acme_proxy_core::config::ProfileConfig],
176    assembly: &Assembly,
177    parts: &GenerationParts,
178    profiles: &[Arc<Profile>],
179    admin_enabled: bool,
180) -> anyhow::Result<acme_proxy_jobs::jobs::JobRegistry> {
181    let database = assembly.database.clone();
182    // Every subsystem with background work, in one registry. **Nothing here
183    // registers a handler per backend**: the registry refuses a second handler
184    // for one kind outright, since two would each claim about half the rows, and
185    // two profiles over different `[signer]` sections are two backends that
186    // `build_backends` deliberately does not collapse. So each backend hands
187    // over *state* and one handler is built over all of it.
188    let mut job_registry = acme_proxy_jobs::jobs::JobRegistry::new();
189    // The CRLs are collected once per distinct backend, since two profiles
190    // sharing one CA share one CRL and refreshing it twice a day would be
191    // pointless work. The identity is kept as a `usize` rather than the
192    // pointer itself, so this function's caller stays `Send`; it is spawned.
193    let mut registered: Vec<usize> = Vec::new();
194    let mut refreshers: Vec<Arc<dyn acme_proxy_signer::CrlRefresher>> = Vec::new();
195    // The relay backends are collected per *profile*, because that is the key a
196    // job row is dispatched on — and because taking the profile list from the
197    // backend would take a stale one: a backend whose configuration did not
198    // move is reused verbatim across a reload, so a profile newly mounted onto
199    // it is not in any list it remembers.
200    //
201    // Both come from this generation's **backends**, which only a process
202    // running the `worker` role builds: everywhere else the set is empty, so
203    // none of these handlers has a backend to reach, and none of them runs
204    // there anyway.
205    let mut relays: Vec<(String, acme_proxy_signer::relay::RelayState)> = Vec::new();
206    let mut backends = parts.signers.by_profile();
207    backends.sort_by(|a, b| a.0.cmp(&b.0));
208    for (profile, backend) in &backends {
209        relays.extend(backend.relay_state().map(|state| (profile.clone(), state)));
210        let identity = Arc::as_ptr(backend).cast::<()>() as usize;
211        if registered.contains(&identity) {
212            continue;
213        }
214        registered.push(identity);
215        refreshers.extend(backend.crl_refresher());
216    }
217    // The daily CRL refresh, over whichever CAs keep a CRL of their own.
218    // Registered only when there is one, the way the audit sweep is registered
219    // only for a non-zero retention.
220    // Beside it, the handler that signs revocations the CLI recorded without
221    // the key.
222    if !refreshers.is_empty() {
223        job_registry
224            .register(Arc::new(
225                acme_proxy_signer::local_ca::sweep::CrlRegenerateJob::new(refreshers.clone()),
226            ))
227            .inspect_err(|error| {
228                error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
229            })?;
230        job_registry
231            .register(Arc::new(
232                acme_proxy_signer::local_ca::sweep::CrlSweepJob::new(refreshers),
233            ))
234            .inspect_err(|error| {
235                error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
236            })?;
237    }
238    // Relayed issuance, over every relay profile at once. Registered only when
239    // some profile relays, for `CrlSweepJob`'s reason: a deployment with none
240    // has no row of this kind to claim.
241    if !relays.is_empty() {
242        job_registry
243            .register(Arc::new(acme_proxy_signer::relay::flow::RelayJob::new(
244                database.clone(),
245                relays,
246            )))
247            .inspect_err(|error| {
248                error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
249            })?;
250    }
251    // Revocations the host CLI queued for a backend it does not build (`relay`,
252    // `custom`). Registered unconditionally, `NotifyJob`'s reason below: a row
253    // queued before a configuration change must still find a handler, and one
254    // naming a profile this generation does not mount is retried rather than
255    // lost.
256    job_registry
257        .register(Arc::new(
258            acme_proxy_protocol::acme::revoke::SignerRevokeJob::new(
259                database.clone(),
260                Arc::new(
261                    acme_proxy_jobs::auditor::Auditor::offline(database.clone())
262                        .with_metrics(assembly.metrics.clone()),
263                ),
264                backends.clone(),
265                assembly.notifiers.clone(),
266            ),
267        ))
268        .inspect_err(|error| {
269            error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
270        })?;
271    // Issuance. `finalize` claims the order and queues the signing, so the
272    // process answering ACME holds no backend. Registered unconditionally for
273    // `SignerRevokeJob`'s reason above, over this generation's backends — none
274    // at all in a process without the worker role, which never claims a row.
275    job_registry
276        .register(Arc::new(
277            acme_proxy_protocol::acme::issue::SignerIssueJob::new(
278                database.clone(),
279                Arc::new(
280                    acme_proxy_jobs::auditor::Auditor::offline(database.clone())
281                        .with_metrics(assembly.metrics.clone()),
282                ),
283                backends.clone(),
284                assembly.notifiers.clone(),
285            ),
286        ))
287        .inspect_err(|error| {
288            error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
289        })?;
290    // Challenge validation. `POST /chall/{id}` claims the challenge and queues
291    // the outbound check, so the probe of a client-chosen host no longer holds
292    // an admission permit for the length of `challenge.timeout_ms`. Registered
293    // unconditionally for `SignerRevokeJob`'s reason above — a row queued
294    // before a configuration change must still find a handler — and holding the
295    // profiles rather than one profile, since the registry refuses a second
296    // handler for one kind.
297    job_registry
298        .register(Arc::new(
299            acme_proxy_protocol::acme::validate::ChallengeValidateJob::new(
300                database.clone(),
301                Arc::new(
302                    acme_proxy_jobs::auditor::Auditor::offline(database.clone())
303                        .with_metrics(assembly.metrics.clone()),
304                ),
305                profiles
306                    .iter()
307                    .map(|profile| (profile.name.clone(), profile.clone()))
308                    .collect(),
309            ),
310        ))
311        .inspect_err(|error| {
312            error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
313        })?;
314
315    // Notification delivery. The same shape as the two above and the reason it
316    // is: one handler for every profile, holding the whole
317    // `profile name -> dispatcher` map, with a job row naming its own profile.
318    // Registered unconditionally — a profile with no `[notify]`
319    // backends queues nothing, so the handler simply never claims a row, and
320    // making the registration conditional would mean a row queued before a
321    // configuration change had nobody to run it.
322    //
323    // It takes the *handle*, not this generation's map: the handler is
324    // registered per generation but must read whichever map is current, and a
325    // row queued by a reloaded router names a slot id only the new one has.
326    job_registry
327        .register(Arc::new(acme_proxy_jobs::notify::NotifyJob::new(
328            assembly.notifiers.clone(),
329        )))
330        .inspect_err(|error| {
331            error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
332        })?;
333
334    // The expiry digest, registered only when some profile asked for one
335    // (`notify.expiry.lead_days`), the way `CrlSweepJob` is registered only
336    // when there is a ledger to prune. It takes the `Notifiers` handle rather
337    // than the profiles' own dispatchers for `NotifyJob`'s reason above, and
338    // the queue because its per-profile rows are something it maintains on
339    // every pass rather than only at `recover`.
340    if let Some(digest) = acme_proxy_jobs::notify::expiry::ExpiryDigestJob::from_profiles(
341        resolved,
342        assembly.notifiers.clone(),
343        database.clone(),
344        assembly.jobs.clone(),
345    ) {
346        job_registry
347            .register(Arc::new(digest))
348            .inspect_err(|error| {
349                error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
350            })?;
351    }
352
353    // The periodic table sweeps. Each is one self-rescheduling row rather than
354    // its own interval loop, so a sweep that dies is reclaimed by lease expiry
355    // and its schedule survives a restart. Their `recover` is also the startup
356    // sweep — it queues at `run_at = now`, so the runner performs the first pass
357    // on its way into the loop and there is nothing to run separately here.
358    let ttl = Duration::from_secs(config.nonce.ttl_seconds);
359    let mut sweeps = vec![acme_proxy_jobs::jobs::SweepJob::nonces(
360        database.clone(),
361        ttl,
362    )];
363    // `0` keeps everything for ever on both of these, and is a handler not
364    // registered rather than a sweep with a cutoff at the epoch.
365    if config.audit.retention_days > 0 {
366        sweeps.push(acme_proxy_jobs::jobs::SweepJob::audit(
367            database.clone(),
368            config.audit.retention_days,
369        ));
370    }
371    if config.jobs.retention_days > 0 {
372        sweeps.push(acme_proxy_jobs::jobs::SweepJob::jobs(
373            database.clone(),
374            config.jobs.retention_days,
375        ));
376    }
377    // One handler covering every mounted profile, since the registry refuses a
378    // second handler for one kind. A profile keeping everything (`0`) is left
379    // out of the list rather than swept with a cutoff at the epoch.
380    let order_retention: Vec<(String, u64)> = resolved
381        .iter()
382        .filter(|profile| profile.sections.order.retention_days > 0)
383        .map(|profile| (profile.name.clone(), profile.sections.order.retention_days))
384        .collect();
385    if !order_retention.is_empty() {
386        sweeps.push(acme_proxy_jobs::jobs::SweepJob::orders(
387            database.clone(),
388            order_retention,
389        ));
390    }
391    // Only where some backend publishes http-01 tokens: the table stays empty
392    // otherwise, and the CRL refresh's rule applies.
393    if profiles
394        .iter()
395        .any(|profile| profile.signer_info.http01_tokens().is_some())
396    {
397        sweeps.push(acme_proxy_jobs::jobs::SweepJob::http01_tokens(
398            database.clone(),
399        ));
400    }
401    if admin_enabled {
402        sweeps.push(acme_proxy_jobs::jobs::SweepJob::admin_sessions(
403            database.clone(),
404            Duration::from_secs(config.admin.session_idle_timeout_seconds),
405            config.admin.session_ttl_seconds,
406        ));
407    }
408    for sweep in sweeps {
409        job_registry
410            .register(Arc::new(sweep))
411            .inspect_err(|error| {
412                error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
413            })?;
414    }
415    Ok(job_registry)
416}
417
418/// What one successful reload hands back to the supervisor: the report to log,
419/// and the pieces of state the *next* reload compares against.
420///
421/// A struct rather than the tuple this used to return — which needed
422/// `#[allow(clippy::type_complexity)]` and left the caller destructuring four
423/// same-shaped values positionally, where swapping two would still compile.
424/// The same move `ProfileParts` made, for the same reason.
425pub(super) struct Reloaded {
426    pub(super) report: crate::reload::ReloadReport,
427    pub(super) config: Arc<Config>,
428    pub(super) resolved: Vec<acme_proxy_core::config::ProfileConfig>,
429    pub(super) logins: Option<Arc<acme_proxy_admin::webadmin::LoginLimiter>>,
430    /// Each socket this reload bound, with the address it landed on. Announced
431    /// by the supervisor rather than here, because saying a listener is up
432    /// reaches the database (the panel's "nobody can sign in yet" warning) and
433    /// [`publish_reload`] has no await point to spend on it — deliberately, that
434    /// being what keeps its publishing run uninterruptible.
435    pub(super) opened: Vec<(Role, String)>,
436    /// The endpoints this reload **added**, for the same reason and announced in
437    /// the same place: `profile_mounted` is dispatched to the `[notify]`
438    /// backends, which queues a job row.
439    pub(super) mounted: Vec<Arc<Profile>>,
440}
441
442/// Everything a reload built and validated, waiting to be published.
443///
444/// The build/publish split is the whole shape of a reload, and making it two
445/// values rather than two halves of one function buys the thing the split was
446/// always claiming: [`prepare_reload`] can run wherever it likes — it runs on a
447/// blocking thread, since building a `relay` backend can contact its upstream —
448/// while [`publish_reload`] stays on the supervisor task, where having no await
449/// point is what makes a generation unobservable half-applied.
450pub(super) struct Prepared {
451    config: Arc<Config>,
452    resolved: Vec<acme_proxy_core::config::ProfileConfig>,
453    parts: GenerationParts,
454    generation: Generation,
455    sockets: SocketPlans,
456    logging: logging::PreparedLogging,
457    /// Whether `RUST_LOG` is what the filter came from, so the publish phase can
458    /// say when an edited `logging.filter` changed nothing, and which of the
459    /// two outranking layers is why.
460    logging_filter_source: logging::FilterSource,
461    /// The endpoints in this generation that the previous one did not mount.
462    mounted: Vec<Arc<Profile>>,
463    /// The endpoints the previous generation mounted and this one does not.
464    unmounted: Vec<String>,
465}
466
467impl Prepared {
468    /// The backends this reload built or carried, for the one step the
469    /// supervisor takes between building and publishing.
470    pub(super) fn signers(&self) -> &acme_proxy_signer::SignerSet {
471        &self.parts.signers
472    }
473}
474
475/// The build half of one reload: everything that can fail, and everything that
476/// can block.
477///
478/// Nothing here touches a cell, so a failure anywhere leaves the running
479/// generation exactly as it was — which is the property the whole "atomic,
480/// refuse by name" decision exists for. Every socket this reload needs is bound
481/// here too, where a port already in use is still a refusal rather than a
482/// listener already dropped.
483///
484/// Run on a blocking thread by
485/// [`supervise_reloads`](super::supervisor::supervise_reloads), because
486/// building a `relay` backend for the first time contacts its upstream
487/// synchronously.
488pub(super) fn prepare_reload(
489    roles: crate::RoleSet,
490    config: &Arc<Config>,
491    resolved: &[acme_proxy_core::config::ProfileConfig],
492    assembly: &Assembly,
493    logins: Option<&acme_proxy_admin::webadmin::LoginLimiter>,
494) -> Result<Prepared, crate::reload::ReloadError> {
495    use crate::reload::{Applied, ReloadError, check_frozen};
496
497    // Re-read from scratch: `Config::load` consults the file *and* the
498    // `ACME_PROXY_*` environment, so a reload sees whatever the process would
499    // see if it restarted right now.
500    let next = Arc::new(Config::load().map_err(|error| ReloadError::Load(error.to_string()))?);
501    let next_resolved = next
502        .resolve_profiles()
503        .map_err(|error| ReloadError::Load(error.to_string()))?;
504
505    check_frozen(
506        &Applied {
507            config,
508            profiles: resolved,
509        },
510        &Applied {
511            config: &next,
512            profiles: &next_resolved,
513        },
514    )?;
515
516    // Built here rather than published here: a bad `logging.target` must refuse
517    // the whole reload with the message startup would have printed, not leave a
518    // half-swapped generation behind. The same build-then-publish split
519    // `Assembly::build_parts` makes.
520    // `flag_override()` is how a `--log-level` typed at startup survives a
521    // `SIGHUP`: the stack is rebuilt from the file, so without re-reading it
522    // the reload would silently demote the server to `logging.filter`.
523    let logging = logging::prepare_logging(&next.logging, logging::flag_override())
524        .map_err(ReloadError::Build)?;
525    let logging_filter_source = logging.filter_source;
526
527    // The same validation startup runs before either socket binds, so a panel
528    // that would refuse to start refuses to be reloaded into. It also compiles
529    // every `admin.template_dir` override, which is what keeps a broken one a
530    // failed reload rather than a 500 in a browser.
531    // Only where this process serves the panel, exactly as startup checks it
532    // (`sockets::bind_admin`). An acme-only or worker-only process that can
533    // start with an `[admin]` section it never serves must not then refuse
534    // every `SIGHUP` over it.
535    if roles.has(crate::ProcessRole::Admin) {
536        acme_proxy_admin::webadmin::check_config(&next)
537            .map_err(|error| ReloadError::Build(error.to_string()))?;
538    }
539    // Its twin for the third listener: `webadmin::check_config` sees the
540    // admin-versus-server pair, this one sees the two it cannot.
541    check_metrics_config(&next).map_err(|error| ReloadError::Build(error.to_string()))?;
542
543    // Every socket this reload needs is bound **here**, where a failure is still
544    // a refusal: a port already taken, an address that does not resolve, a
545    // privileged port after a `setcap` was lost. Past the publish phase nothing
546    // can fail, so the running listeners are never dropped for a configuration
547    // that then turns out not to work.
548    let sockets = plan_sockets(roles, config, &next)?;
549
550    // The egress clients, the notification dispatchers and the signer backends.
551    // The last is where a newly mounted endpoint gets a backend, a removed one's
552    // is left out, and an edited `[signer]` is rebuilt over the live instance's
553    // in-memory state — see `signer::build_backends`. It is also the one step
554    // that can make a network call, hence this whole function's blocking thread.
555    let parts = assembly
556        .build_parts(&next_resolved, &next)
557        .map_err(|error| ReloadError::Build(error.to_string()))?;
558    let generation = build_generation(roles, &next, &next_resolved, assembly, &parts, logins)
559        .map_err(|error| ReloadError::Build(error.to_string()))?;
560
561    // Compared by name against what is running, not against what is written
562    // down: `resolve_profiles` has already dropped every `enabled = false`
563    // entry, so this is the set of endpoints actually served.
564    let running: std::collections::HashSet<&str> = resolved
565        .iter()
566        .map(|profile| profile.name.as_str())
567        .collect();
568    let mounted = generation
569        .profiles
570        .iter()
571        .filter(|profile| !running.contains(profile.name.as_str()))
572        .cloned()
573        .collect();
574    let next_names: std::collections::HashSet<&str> = next_resolved
575        .iter()
576        .map(|profile| profile.name.as_str())
577        .collect();
578    let unmounted = resolved
579        .iter()
580        .map(|profile| profile.name.clone())
581        .filter(|name| !next_names.contains(name.as_str()))
582        .collect();
583
584    Ok(Prepared {
585        config: next,
586        resolved: next_resolved,
587        parts,
588        generation,
589        sockets,
590        logging,
591        logging_filter_source,
592        mounted,
593        unmounted,
594    })
595}
596
597/// The publish half: infallible, synchronous, and uninterruptible.
598///
599/// There is no `.await` in here, and that is load-bearing rather than
600/// incidental. `watch::Sender::send_replace` and
601/// `mpsc::UnboundedSender::send` are both synchronous, so a run of them with no
602/// await point between cannot be interleaved — no task can observe a generation
603/// half-applied, and no lock is needed to say so.
604///
605/// The order matters in three places. `[logging]` goes **first**, because an
606/// operator who raised the level did it to see what happens next, starting with
607/// this reload's own completion line. The notifier map, the signer set and the
608/// job registry go **before** the routers: a request served by the new
609/// generation queues a `notify_deliver` row naming a slot id from the new
610/// configuration, and a `NotifyJob` still holding the old map would retire it —
611/// permanently, since an unknown backend id is a `Failed`, not a `Retry`. And
612/// the TLS mode goes before the socket, so a freshly bound listener's very first
613/// connection is already accepted under this generation's settings.
614pub(super) fn publish_reload(
615    prepared: Prepared,
616    applied: &Arc<Config>,
617    assembly: &Assembly,
618    cells: &Cells,
619    generation: u64,
620    started: std::time::Instant,
621) -> Reloaded {
622    use crate::reload::ReloadReport;
623
624    let Prepared {
625        config: next,
626        resolved: next_resolved,
627        parts,
628        generation: built,
629        sockets,
630        logging,
631        logging_filter_source,
632        mounted,
633        unmounted,
634    } = prepared;
635
636    let logging_reloaded = logging::publish_logging(logging);
637
638    let report = ReloadReport {
639        generation,
640        profiles: built
641            .profiles
642            .iter()
643            .map(|profile| profile.name.clone())
644            .collect(),
645        job_kinds: built.job_registry.kinds(),
646        tls_reloaded: built.tls.is_some(),
647        admin_tls_reloaded: built.admin_tls.is_some(),
648        listeners_rebound: sockets.rebound(),
649        logging_reloaded,
650        duration: started.elapsed(),
651    };
652    let next_logins = built.logins.clone();
653
654    assembly.publish_notifiers(parts.dispatchers);
655    // The set the *next* reload compares against, and the point at which the
656    // backends this one dropped are finally released — after their replacements
657    // were built and adopted their state, never before.
658    assembly.publish_signers(parts.signers, parts.infos);
659    cells
660        .job_registry
661        .send_replace(Arc::new(built.job_registry));
662
663    // `[jobs]` in its two halves, both synchronous and neither able to fail —
664    // which is what lets them sit in this run rather than needing a build phase
665    // of their own. The runner re-derives its pacing from the cell on its next
666    // pass; `max_attempts` goes to the queue instead, because it is the enqueue
667    // side that reads it, and it sets the budget for work queued from here on
668    // rather than for the rows already waiting.
669    cells.jobs.send_replace(Arc::new(next.jobs.clone()));
670    assembly.jobs.set_max_attempts(next.jobs.max_attempts);
671
672    cells.acme.set_tls(built.tls);
673    cells.admin.set_tls(built.admin_tls);
674    let opened = sockets.bound.clone();
675    sockets.publish(cells);
676
677    cells
678        .acme_router
679        .send_replace(built.acme_app.into_service::<axum::body::Body>());
680    // An empty router when the panel is off, which is what a request arriving
681    // on a connection established a moment before it was switched off now gets:
682    // closing the socket stops the next client, and this stops that one.
683    cells.admin_router.send_replace(
684        built
685            .admin_app
686            .unwrap_or_default()
687            .into_service::<axum::body::Body>(),
688    );
689
690    // Said here rather than by the supervisor because it needs nothing but a
691    // name, unlike the mounting half, which dispatches a notification.
692    for profile in unmounted {
693        warn!(
694            event = "profile_unmounted",
695            outcome = "advisory",
696            profile = %profile,
697            "the endpoint is no longer served: its accounts and orders stay in the \
698             database and come back if it is mounted again, but any issuance still in \
699             flight for it has no handler left to finish it"
700        );
701    }
702
703    // `--log-level` and `RUST_LOG` both outrank `logging.filter` on a reload
704    // exactly as they do at startup — the two disagreeing would be worse — but
705    // that makes an edited `logging.filter` a silent no-op, which is the one
706    // outcome an operator would read as "my reload did not land". Said only
707    // when both halves hold: something outranked the file, *and* the file's
708    // filter actually moved. `source` names which, since the two are unset in
709    // different places.
710    if logging_filter_source.outranks_config() && applied.logging.filter != next.logging.filter {
711        warn!(
712            event = "server_logging_filter_overridden",
713            outcome = "advisory",
714            source = logging_filter_source.as_str(),
715            configured = %next.logging.filter,
716        );
717    }
718
719    Reloaded {
720        report,
721        config: next,
722        resolved: next_resolved,
723        logins: next_logins,
724        opened,
725        mounted,
726    }
727}
728
729/// The one announcement an endpoint that has just come up makes: a log line and
730/// a `[notify]` lifecycle event.
731///
732/// Shared by startup and by a reload that mounted a new endpoint, so the two
733/// cannot drift — before the profile set could reload there was only one caller
734/// and the sharing was not needed.
735pub(super) async fn announce_profile(profile: &Arc<Profile>) {
736    info!(
737        event = "profile_mounted",
738        outcome = "success",
739        profile = %profile.name,
740        directory = %profile.directory_url(),
741        challenge_bypass = profile.challenges.is_bypassed(),
742        eab_enabled = profile.eab.enabled
743    );
744    profile
745        .notify
746        .dispatch(acme_proxy_jobs::notify::NotifyEvent::ProfileMounted(
747            acme_proxy_jobs::notify::ProfileMountedData {
748                profile: profile.name.clone(),
749            },
750        ))
751        .await;
752}