acme-proxy-server 0.6.1

The runtime of acme-proxy: role processes, listeners, configuration reload and logging (internal crate, no semver promise)
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
//! One configuration generation: built and validated in full, then published
//! in one uninterruptible run. Startup builds the first; a reload builds and
//! publishes every later one.

use std::sync::Arc;
use std::time::Duration;

use tracing::{error, info, warn};

use super::logging;
use acme_proxy_core::config::Config;
use acme_proxy_net::tls;

use super::sockets::{Role, SocketPlans, check_metrics_config, plan_sockets};
use super::supervisor::Cells;
use super::{Assembly, GenerationParts};
use acme_proxy_protocol::profile::Profile;
use acme_proxy_protocol::router::build_app;

/// Everything one configuration generation contributes, built and validated
/// before any of it is published.
///
/// The unit exists so startup and reload cannot drift: both go through
/// [`build_generation`], so a subsystem added to one is added to the other by
/// construction rather than by remembering.
pub(crate) struct Generation {
    /// Kept so startup can announce them; a reload drops them on the floor,
    /// since `profile_mounted` is a lifecycle event and not a heartbeat.
    pub(super) profiles: Vec<Arc<Profile>>,
    pub(super) acme_app: axum::Router,
    pub(super) admin_app: Option<axum::Router>,
    pub(super) job_registry: acme_proxy_jobs::jobs::JobRegistry,
    pub(super) tls: Option<tls::TlsSettings>,
    pub(super) admin_tls: Option<tls::TlsSettings>,
    /// The limiter this generation ended up with, for the next one to carry.
    pub(super) logins: Option<Arc<acme_proxy_admin::webadmin::LoginLimiter>>,
}

/// Builds one generation: profiles, both routers, the job registry, the audit
/// trail and both TLS acceptors.
///
/// Fallible throughout and side-effect-free on the *serving* state: nothing here
/// touches a cell, so a failure leaves whatever is already running exactly as it
/// was. That is what makes an atomic reload possible — everything is built
/// first, and only a complete success publishes anything.
///
/// `dispatchers` is passed in rather than built here because a reload needs to
/// hold it back: it is published to the long-lived [`acme_proxy_jobs::notify::Notifiers`]
/// handle at swap time, *before* the routers, so a request served by the new
/// generation cannot queue a delivery the job runner's map does not know.
///
/// Whether the panel is part of this generation is read from `admin.enabled`
/// rather than from whether a socket exists. It used to be the latter, which
/// was the same answer while the key was frozen and is the wrong one now that a
/// reload can turn the panel on: the app, its TLS, its session sweep and its
/// login limiter all have to appear in the generation *before* there is a
/// listener to serve them.
pub(crate) fn build_generation(
    roles: crate::RoleSet,
    config: &Arc<Config>,
    resolved: &[acme_proxy_core::config::ProfileConfig],
    assembly: &Assembly,
    parts: &GenerationParts,
    previous_logins: Option<&acme_proxy_admin::webadmin::LoginLimiter>,
) -> anyhow::Result<Generation> {
    // A process that does not run the `admin` role binds no admin socket
    // (`sockets::bind_admin`), so building its router, its TLS and its login
    // limiter would be work for a listener that will never exist — and it would
    // make `[admin]` settings this process never serves able to fail its
    // reloads.
    let admin_enabled = config.admin.enabled && roles.has(crate::ProcessRole::Admin);
    let database = assembly.database.clone();
    let profiles = crate::profile::build_all_with(config, resolved, parts)?;

    let tls = tls::from_config(&config.server)
        .inspect_err(|error| {
            error!(event = "tls_init_failed", outcome = "failure", error = %error);
        })?
        .map(|acceptor| {
            tls::TlsSettings::new(
                acceptor,
                Duration::from_millis(config.server.tls.handshake_timeout_ms),
            )
        });

    let admin_tls = match admin_enabled {
        false => None,
        true => tls::admin_from_config(&config.admin)
            .inspect_err(|error| {
                error!(event = "admin_tls_init_failed", outcome = "failure", error = %error);
            })?
            .map(|acceptor| {
                tls::TlsSettings::new(
                    acceptor,
                    Duration::from_millis(config.admin.tls.handshake_timeout_ms),
                )
            }),
    };

    let job_registry =
        job_registry_for(config, resolved, assembly, parts, &profiles, admin_enabled)?;

    // The CA's audit trail: one per process, shared by every profile's router
    // and by the web admin listener, because `[audit]` is process-wide. Built
    // here rather than in `server::profile::build_all` for exactly that reason — it is
    // not a per-endpoint subsystem.
    let auditor = Arc::new(
        acme_proxy_jobs::auditor::Auditor::from_config(
            &config.audit,
            &config.dns,
            database.clone(),
            // The registry the certificate counters land in. Carried by
            // `Assembly`, so it is the same one across every generation and the
            // same one `metrics_app` serves from.
            assembly.metrics.clone(),
        )
        .inspect_err(|error| {
            error!(event = "audit_init_failed", outcome = "failure", error = %error);
        })?,
    );

    // Built **before** `build_app`, which consumes `profiles`. The admin state
    // needs the same profiles (revoking an order resolves that order's own
    // signer), and `build_admin_app` takes a slice precisely so the ordering
    // is a signature constraint rather than a borrow error to rediscover.
    let (admin_app, logins) = match admin_enabled {
        false => (None, None),
        true => {
            // Built again, although `check_config` already did: that call is
            // the refusal, this one is the policy the router keeps.
            let policy =
                acme_proxy_admin::webadmin::filter::build(config).inspect_err(|error| {
                    error!(event = "admin_filter_init_failed", outcome = "failure", error = %error);
                })?;
            let (router, logins) = acme_proxy_admin::webadmin::build_admin_app_with_logins(
                database.clone(),
                config.clone(),
                &profiles,
                auditor.clone(),
                assembly.notifiers.clone(),
                assembly.jobs.clone(),
                policy,
                previous_logins,
            );
            (Some(router), Some(logins))
        }
    };
    let acme_app = build_app(
        database,
        config.clone(),
        profiles.clone(),
        auditor,
        assembly.metrics.clone(),
        assembly.jobs.clone(),
    );

    Ok(Generation {
        profiles,
        acme_app,
        admin_app,
        job_registry,
        tls,
        admin_tls,
        logins,
    })
}

/// Every subsystem with background work, registered for one generation.
///
/// A sweep or handler that is never registered never runs, and nothing else
/// notices: the table it should prune just grows. Which ones a configuration
/// registers is pinned by `tests::a_generation_registers_the_sweeps_its_configuration_asks_for`.
fn job_registry_for(
    config: &Config,
    resolved: &[acme_proxy_core::config::ProfileConfig],
    assembly: &Assembly,
    parts: &GenerationParts,
    profiles: &[Arc<Profile>],
    admin_enabled: bool,
) -> anyhow::Result<acme_proxy_jobs::jobs::JobRegistry> {
    let database = assembly.database.clone();
    // Every subsystem with background work, in one registry. **Nothing here
    // registers a handler per backend**: the registry refuses a second handler
    // for one kind outright, since two would each claim about half the rows, and
    // two profiles over different `[signer]` sections are two backends that
    // `build_backends` deliberately does not collapse. So each backend hands
    // over *state* and one handler is built over all of it.
    let mut job_registry = acme_proxy_jobs::jobs::JobRegistry::new();
    // The CRLs are collected once per distinct backend, since two profiles
    // sharing one CA share one CRL and refreshing it twice a day would be
    // pointless work. The identity is kept as a `usize` rather than the
    // pointer itself, so this function's caller stays `Send`; it is spawned.
    let mut registered: Vec<usize> = Vec::new();
    let mut refreshers: Vec<Arc<dyn acme_proxy_signer::CrlRefresher>> = Vec::new();
    // The relay backends are collected per *profile*, because that is the key a
    // job row is dispatched on — and because taking the profile list from the
    // backend would take a stale one: a backend whose configuration did not
    // move is reused verbatim across a reload, so a profile newly mounted onto
    // it is not in any list it remembers.
    //
    // Both come from this generation's **backends**, which only a process
    // running the `worker` role builds: everywhere else the set is empty, so
    // none of these handlers has a backend to reach, and none of them runs
    // there anyway.
    let mut relays: Vec<(String, acme_proxy_signer::relay::RelayState)> = Vec::new();
    let mut backends = parts.signers.by_profile();
    backends.sort_by(|a, b| a.0.cmp(&b.0));
    for (profile, backend) in &backends {
        relays.extend(backend.relay_state().map(|state| (profile.clone(), state)));
        let identity = Arc::as_ptr(backend).cast::<()>() as usize;
        if registered.contains(&identity) {
            continue;
        }
        registered.push(identity);
        refreshers.extend(backend.crl_refresher());
    }
    // The daily CRL refresh, over whichever CAs keep a CRL of their own.
    // Registered only when there is one, the way the audit sweep is registered
    // only for a non-zero retention.
    // Beside it, the handler that signs revocations the CLI recorded without
    // the key.
    if !refreshers.is_empty() {
        job_registry
            .register(Arc::new(
                acme_proxy_signer::local_ca::sweep::CrlRegenerateJob::new(refreshers.clone()),
            ))
            .inspect_err(|error| {
                error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
            })?;
        job_registry
            .register(Arc::new(
                acme_proxy_signer::local_ca::sweep::CrlSweepJob::new(refreshers),
            ))
            .inspect_err(|error| {
                error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
            })?;
    }
    // Relayed issuance, over every relay profile at once. Registered only when
    // some profile relays, for `CrlSweepJob`'s reason: a deployment with none
    // has no row of this kind to claim.
    if !relays.is_empty() {
        job_registry
            .register(Arc::new(acme_proxy_signer::relay::flow::RelayJob::new(
                database.clone(),
                relays,
            )))
            .inspect_err(|error| {
                error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
            })?;
    }
    // Revocations the host CLI queued for a backend it does not build (`relay`,
    // `custom`). Registered unconditionally, `NotifyJob`'s reason below: a row
    // queued before a configuration change must still find a handler, and one
    // naming a profile this generation does not mount is retried rather than
    // lost.
    job_registry
        .register(Arc::new(
            acme_proxy_protocol::acme::revoke::SignerRevokeJob::new(
                database.clone(),
                Arc::new(
                    acme_proxy_jobs::auditor::Auditor::offline(database.clone())
                        .with_metrics(assembly.metrics.clone()),
                ),
                backends.clone(),
                assembly.notifiers.clone(),
            ),
        ))
        .inspect_err(|error| {
            error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
        })?;
    // Issuance. `finalize` claims the order and queues the signing, so the
    // process answering ACME holds no backend. Registered unconditionally for
    // `SignerRevokeJob`'s reason above, over this generation's backends — none
    // at all in a process without the worker role, which never claims a row.
    job_registry
        .register(Arc::new(
            acme_proxy_protocol::acme::issue::SignerIssueJob::new(
                database.clone(),
                Arc::new(
                    acme_proxy_jobs::auditor::Auditor::offline(database.clone())
                        .with_metrics(assembly.metrics.clone()),
                ),
                backends.clone(),
                assembly.notifiers.clone(),
            ),
        ))
        .inspect_err(|error| {
            error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
        })?;
    // Challenge validation. `POST /chall/{id}` claims the challenge and queues
    // the outbound check, so the probe of a client-chosen host no longer holds
    // an admission permit for the length of `challenge.timeout_ms`. Registered
    // unconditionally for `SignerRevokeJob`'s reason above — a row queued
    // before a configuration change must still find a handler — and holding the
    // profiles rather than one profile, since the registry refuses a second
    // handler for one kind.
    job_registry
        .register(Arc::new(
            acme_proxy_protocol::acme::validate::ChallengeValidateJob::new(
                database.clone(),
                Arc::new(
                    acme_proxy_jobs::auditor::Auditor::offline(database.clone())
                        .with_metrics(assembly.metrics.clone()),
                ),
                profiles
                    .iter()
                    .map(|profile| (profile.name.clone(), profile.clone()))
                    .collect(),
            ),
        ))
        .inspect_err(|error| {
            error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
        })?;

    // Notification delivery. The same shape as the two above and the reason it
    // is: one handler for every profile, holding the whole
    // `profile name -> dispatcher` map, with a job row naming its own profile.
    // Registered unconditionally — a profile with no `[notify]`
    // backends queues nothing, so the handler simply never claims a row, and
    // making the registration conditional would mean a row queued before a
    // configuration change had nobody to run it.
    //
    // It takes the *handle*, not this generation's map: the handler is
    // registered per generation but must read whichever map is current, and a
    // row queued by a reloaded router names a slot id only the new one has.
    job_registry
        .register(Arc::new(acme_proxy_jobs::notify::NotifyJob::new(
            assembly.notifiers.clone(),
        )))
        .inspect_err(|error| {
            error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
        })?;

    // The expiry digest, registered only when some profile asked for one
    // (`notify.expiry.lead_days`), the way `CrlSweepJob` is registered only
    // when there is a ledger to prune. It takes the `Notifiers` handle rather
    // than the profiles' own dispatchers for `NotifyJob`'s reason above, and
    // the queue because its per-profile rows are something it maintains on
    // every pass rather than only at `recover`.
    if let Some(digest) = acme_proxy_jobs::notify::expiry::ExpiryDigestJob::from_profiles(
        resolved,
        assembly.notifiers.clone(),
        database.clone(),
        assembly.jobs.clone(),
    ) {
        job_registry
            .register(Arc::new(digest))
            .inspect_err(|error| {
                error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
            })?;
    }

    // The periodic table sweeps. Each is one self-rescheduling row rather than
    // its own interval loop, so a sweep that dies is reclaimed by lease expiry
    // and its schedule survives a restart. Their `recover` is also the startup
    // sweep — it queues at `run_at = now`, so the runner performs the first pass
    // on its way into the loop and there is nothing to run separately here.
    let ttl = Duration::from_secs(config.nonce.ttl_seconds);
    let mut sweeps = vec![acme_proxy_jobs::jobs::SweepJob::nonces(
        database.clone(),
        ttl,
    )];
    // `0` keeps everything for ever on both of these, and is a handler not
    // registered rather than a sweep with a cutoff at the epoch.
    if config.audit.retention_days > 0 {
        sweeps.push(acme_proxy_jobs::jobs::SweepJob::audit(
            database.clone(),
            config.audit.retention_days,
        ));
    }
    if config.jobs.retention_days > 0 {
        sweeps.push(acme_proxy_jobs::jobs::SweepJob::jobs(
            database.clone(),
            config.jobs.retention_days,
        ));
    }
    // One handler covering every mounted profile, since the registry refuses a
    // second handler for one kind. A profile keeping everything (`0`) is left
    // out of the list rather than swept with a cutoff at the epoch.
    let order_retention: Vec<(String, u64)> = resolved
        .iter()
        .filter(|profile| profile.sections.order.retention_days > 0)
        .map(|profile| (profile.name.clone(), profile.sections.order.retention_days))
        .collect();
    if !order_retention.is_empty() {
        sweeps.push(acme_proxy_jobs::jobs::SweepJob::orders(
            database.clone(),
            order_retention,
        ));
    }
    // Only where some backend publishes http-01 tokens: the table stays empty
    // otherwise, and the CRL refresh's rule applies.
    if profiles
        .iter()
        .any(|profile| profile.signer_info.http01_tokens().is_some())
    {
        sweeps.push(acme_proxy_jobs::jobs::SweepJob::http01_tokens(
            database.clone(),
        ));
    }
    if admin_enabled {
        sweeps.push(acme_proxy_jobs::jobs::SweepJob::admin_sessions(
            database.clone(),
            Duration::from_secs(config.admin.session_idle_timeout_seconds),
            config.admin.session_ttl_seconds,
        ));
    }
    for sweep in sweeps {
        job_registry
            .register(Arc::new(sweep))
            .inspect_err(|error| {
                error!(event = "job_registry_init_failed", outcome = "failure", error = %error);
            })?;
    }
    Ok(job_registry)
}

/// What one successful reload hands back to the supervisor: the report to log,
/// and the pieces of state the *next* reload compares against.
///
/// A struct rather than the tuple this used to return — which needed
/// `#[allow(clippy::type_complexity)]` and left the caller destructuring four
/// same-shaped values positionally, where swapping two would still compile.
/// The same move `ProfileParts` made, for the same reason.
pub(super) struct Reloaded {
    pub(super) report: crate::reload::ReloadReport,
    pub(super) config: Arc<Config>,
    pub(super) resolved: Vec<acme_proxy_core::config::ProfileConfig>,
    pub(super) logins: Option<Arc<acme_proxy_admin::webadmin::LoginLimiter>>,
    /// Each socket this reload bound, with the address it landed on. Announced
    /// by the supervisor rather than here, because saying a listener is up
    /// reaches the database (the panel's "nobody can sign in yet" warning) and
    /// [`publish_reload`] has no await point to spend on it — deliberately, that
    /// being what keeps its publishing run uninterruptible.
    pub(super) opened: Vec<(Role, String)>,
    /// The endpoints this reload **added**, for the same reason and announced in
    /// the same place: `profile_mounted` is dispatched to the `[notify]`
    /// backends, which queues a job row.
    pub(super) mounted: Vec<Arc<Profile>>,
}

/// Everything a reload built and validated, waiting to be published.
///
/// The build/publish split is the whole shape of a reload, and making it two
/// values rather than two halves of one function buys the thing the split was
/// always claiming: [`prepare_reload`] can run wherever it likes — it runs on a
/// blocking thread, since building a `relay` backend can contact its upstream —
/// while [`publish_reload`] stays on the supervisor task, where having no await
/// point is what makes a generation unobservable half-applied.
pub(super) struct Prepared {
    config: Arc<Config>,
    resolved: Vec<acme_proxy_core::config::ProfileConfig>,
    parts: GenerationParts,
    generation: Generation,
    sockets: SocketPlans,
    logging: logging::PreparedLogging,
    /// Whether `RUST_LOG` is what the filter came from, so the publish phase can
    /// say when an edited `logging.filter` changed nothing, and which of the
    /// two outranking layers is why.
    logging_filter_source: logging::FilterSource,
    /// The endpoints in this generation that the previous one did not mount.
    mounted: Vec<Arc<Profile>>,
    /// The endpoints the previous generation mounted and this one does not.
    unmounted: Vec<String>,
}

impl Prepared {
    /// The backends this reload built or carried, for the one step the
    /// supervisor takes between building and publishing.
    pub(super) fn signers(&self) -> &acme_proxy_signer::SignerSet {
        &self.parts.signers
    }
}

/// The build half of one reload: everything that can fail, and everything that
/// can block.
///
/// Nothing here touches a cell, so a failure anywhere leaves the running
/// generation exactly as it was — which is the property the whole "atomic,
/// refuse by name" decision exists for. Every socket this reload needs is bound
/// here too, where a port already in use is still a refusal rather than a
/// listener already dropped.
///
/// Run on a blocking thread by
/// [`supervise_reloads`](super::supervisor::supervise_reloads), because
/// building a `relay` backend for the first time contacts its upstream
/// synchronously.
pub(super) fn prepare_reload(
    roles: crate::RoleSet,
    config: &Arc<Config>,
    resolved: &[acme_proxy_core::config::ProfileConfig],
    assembly: &Assembly,
    logins: Option<&acme_proxy_admin::webadmin::LoginLimiter>,
) -> Result<Prepared, crate::reload::ReloadError> {
    use crate::reload::{Applied, ReloadError, check_frozen};

    // Re-read from scratch: `Config::load` consults the file *and* the
    // `ACME_PROXY_*` environment, so a reload sees whatever the process would
    // see if it restarted right now.
    let next = Arc::new(Config::load().map_err(|error| ReloadError::Load(error.to_string()))?);
    let next_resolved = next
        .resolve_profiles()
        .map_err(|error| ReloadError::Load(error.to_string()))?;

    check_frozen(
        &Applied {
            config,
            profiles: resolved,
        },
        &Applied {
            config: &next,
            profiles: &next_resolved,
        },
    )?;

    // Built here rather than published here: a bad `logging.target` must refuse
    // the whole reload with the message startup would have printed, not leave a
    // half-swapped generation behind. The same build-then-publish split
    // `Assembly::build_parts` makes.
    // `flag_override()` is how a `--log-level` typed at startup survives a
    // `SIGHUP`: the stack is rebuilt from the file, so without re-reading it
    // the reload would silently demote the server to `logging.filter`.
    let logging = logging::prepare_logging(&next.logging, logging::flag_override())
        .map_err(ReloadError::Build)?;
    let logging_filter_source = logging.filter_source;

    // The same validation startup runs before either socket binds, so a panel
    // that would refuse to start refuses to be reloaded into. It also compiles
    // every `admin.template_dir` override, which is what keeps a broken one a
    // failed reload rather than a 500 in a browser.
    // Only where this process serves the panel, exactly as startup checks it
    // (`sockets::bind_admin`). An acme-only or worker-only process that can
    // start with an `[admin]` section it never serves must not then refuse
    // every `SIGHUP` over it.
    if roles.has(crate::ProcessRole::Admin) {
        acme_proxy_admin::webadmin::check_config(&next)
            .map_err(|error| ReloadError::Build(error.to_string()))?;
    }
    // Its twin for the third listener: `webadmin::check_config` sees the
    // admin-versus-server pair, this one sees the two it cannot.
    check_metrics_config(&next).map_err(|error| ReloadError::Build(error.to_string()))?;

    // Every socket this reload needs is bound **here**, where a failure is still
    // a refusal: a port already taken, an address that does not resolve, a
    // privileged port after a `setcap` was lost. Past the publish phase nothing
    // can fail, so the running listeners are never dropped for a configuration
    // that then turns out not to work.
    let sockets = plan_sockets(roles, config, &next)?;

    // The egress clients, the notification dispatchers and the signer backends.
    // The last is where a newly mounted endpoint gets a backend, a removed one's
    // is left out, and an edited `[signer]` is rebuilt over the live instance's
    // in-memory state — see `signer::build_backends`. It is also the one step
    // that can make a network call, hence this whole function's blocking thread.
    let parts = assembly
        .build_parts(&next_resolved, &next)
        .map_err(|error| ReloadError::Build(error.to_string()))?;
    let generation = build_generation(roles, &next, &next_resolved, assembly, &parts, logins)
        .map_err(|error| ReloadError::Build(error.to_string()))?;

    // Compared by name against what is running, not against what is written
    // down: `resolve_profiles` has already dropped every `enabled = false`
    // entry, so this is the set of endpoints actually served.
    let running: std::collections::HashSet<&str> = resolved
        .iter()
        .map(|profile| profile.name.as_str())
        .collect();
    let mounted = generation
        .profiles
        .iter()
        .filter(|profile| !running.contains(profile.name.as_str()))
        .cloned()
        .collect();
    let next_names: std::collections::HashSet<&str> = next_resolved
        .iter()
        .map(|profile| profile.name.as_str())
        .collect();
    let unmounted = resolved
        .iter()
        .map(|profile| profile.name.clone())
        .filter(|name| !next_names.contains(name.as_str()))
        .collect();

    Ok(Prepared {
        config: next,
        resolved: next_resolved,
        parts,
        generation,
        sockets,
        logging,
        logging_filter_source,
        mounted,
        unmounted,
    })
}

/// The publish half: infallible, synchronous, and uninterruptible.
///
/// There is no `.await` in here, and that is load-bearing rather than
/// incidental. `watch::Sender::send_replace` and
/// `mpsc::UnboundedSender::send` are both synchronous, so a run of them with no
/// await point between cannot be interleaved — no task can observe a generation
/// half-applied, and no lock is needed to say so.
///
/// The order matters in three places. `[logging]` goes **first**, because an
/// operator who raised the level did it to see what happens next, starting with
/// this reload's own completion line. The notifier map, the signer set and the
/// job registry go **before** the routers: a request served by the new
/// generation queues a `notify_deliver` row naming a slot id from the new
/// configuration, and a `NotifyJob` still holding the old map would retire it —
/// permanently, since an unknown backend id is a `Failed`, not a `Retry`. And
/// the TLS mode goes before the socket, so a freshly bound listener's very first
/// connection is already accepted under this generation's settings.
pub(super) fn publish_reload(
    prepared: Prepared,
    applied: &Arc<Config>,
    assembly: &Assembly,
    cells: &Cells,
    generation: u64,
    started: std::time::Instant,
) -> Reloaded {
    use crate::reload::ReloadReport;

    let Prepared {
        config: next,
        resolved: next_resolved,
        parts,
        generation: built,
        sockets,
        logging,
        logging_filter_source,
        mounted,
        unmounted,
    } = prepared;

    let logging_reloaded = logging::publish_logging(logging);

    let report = ReloadReport {
        generation,
        profiles: built
            .profiles
            .iter()
            .map(|profile| profile.name.clone())
            .collect(),
        job_kinds: built.job_registry.kinds(),
        tls_reloaded: built.tls.is_some(),
        admin_tls_reloaded: built.admin_tls.is_some(),
        listeners_rebound: sockets.rebound(),
        logging_reloaded,
        duration: started.elapsed(),
    };
    let next_logins = built.logins.clone();

    assembly.publish_notifiers(parts.dispatchers);
    // The set the *next* reload compares against, and the point at which the
    // backends this one dropped are finally released — after their replacements
    // were built and adopted their state, never before.
    assembly.publish_signers(parts.signers, parts.infos);
    cells
        .job_registry
        .send_replace(Arc::new(built.job_registry));

    // `[jobs]` in its two halves, both synchronous and neither able to fail —
    // which is what lets them sit in this run rather than needing a build phase
    // of their own. The runner re-derives its pacing from the cell on its next
    // pass; `max_attempts` goes to the queue instead, because it is the enqueue
    // side that reads it, and it sets the budget for work queued from here on
    // rather than for the rows already waiting.
    cells.jobs.send_replace(Arc::new(next.jobs.clone()));
    assembly.jobs.set_max_attempts(next.jobs.max_attempts);

    cells.acme.set_tls(built.tls);
    cells.admin.set_tls(built.admin_tls);
    let opened = sockets.bound.clone();
    sockets.publish(cells);

    cells
        .acme_router
        .send_replace(built.acme_app.into_service::<axum::body::Body>());
    // An empty router when the panel is off, which is what a request arriving
    // on a connection established a moment before it was switched off now gets:
    // closing the socket stops the next client, and this stops that one.
    cells.admin_router.send_replace(
        built
            .admin_app
            .unwrap_or_default()
            .into_service::<axum::body::Body>(),
    );

    // Said here rather than by the supervisor because it needs nothing but a
    // name, unlike the mounting half, which dispatches a notification.
    for profile in unmounted {
        warn!(
            event = "profile_unmounted",
            outcome = "advisory",
            profile = %profile,
            "the endpoint is no longer served: its accounts and orders stay in the \
             database and come back if it is mounted again, but any issuance still in \
             flight for it has no handler left to finish it"
        );
    }

    // `--log-level` and `RUST_LOG` both outrank `logging.filter` on a reload
    // exactly as they do at startup — the two disagreeing would be worse — but
    // that makes an edited `logging.filter` a silent no-op, which is the one
    // outcome an operator would read as "my reload did not land". Said only
    // when both halves hold: something outranked the file, *and* the file's
    // filter actually moved. `source` names which, since the two are unset in
    // different places.
    if logging_filter_source.outranks_config() && applied.logging.filter != next.logging.filter {
        warn!(
            event = "server_logging_filter_overridden",
            outcome = "advisory",
            source = logging_filter_source.as_str(),
            configured = %next.logging.filter,
        );
    }

    Reloaded {
        report,
        config: next,
        resolved: next_resolved,
        logins: next_logins,
        opened,
        mounted,
    }
}

/// The one announcement an endpoint that has just come up makes: a log line and
/// a `[notify]` lifecycle event.
///
/// Shared by startup and by a reload that mounted a new endpoint, so the two
/// cannot drift — before the profile set could reload there was only one caller
/// and the sharing was not needed.
pub(super) async fn announce_profile(profile: &Arc<Profile>) {
    info!(
        event = "profile_mounted",
        outcome = "success",
        profile = %profile.name,
        directory = %profile.directory_url(),
        challenge_bypass = profile.challenges.is_bypassed(),
        eab_enabled = profile.eab.enabled
    );
    profile
        .notify
        .dispatch(acme_proxy_jobs::notify::NotifyEvent::ProfileMounted(
            acme_proxy_jobs::notify::ProfileMountedData {
                profile: profile.name.clone(),
            },
        ))
        .await;
}