acme_proxy/signer/mod.rs
1//! Certificate-issuance abstraction.
2//!
3//! Finalizing an ACME order turns the client's CSR into an issued certificate.
4//! *How* that happens is pluggable: the [`SignerBackend`] trait hides the backend
5//! behind a single [`issue`](SignerBackend::issue) call, and [`from_config`]
6//! builds the configured one at startup. The only backend implemented today is
7//! [`local_ca::LocalCa`] — a persistent local CA.
8//!
9//! Revocation (RFC 8555 §7.6) is part of the same abstraction:
10//! [`revoke`](SignerBackend::revoke) must actually revoke the certificate at
11//! the backend, not just at the ACME/database layer — for [`local_ca::LocalCa`]
12//! that means a real, CA-signed CRL. [`crl_der`](SignerBackend::crl_der) is how
13//! a backend that maintains one serves it (`GET /crl`); it defaults to `None`
14//! for a backend with no CRL of its own to publish here (e.g. one delegating to
15//! an upstream CA that publishes its own).
16//!
17//! ## Asynchronous by design
18//!
19//! [`issue`](SignerBackend::issue) is **async**, so a backend that *delegates*
20//! signing over the network (an upstream ACME CA, a remote signer) can await its
21//! IO instead of blocking a runtime thread. [`local_ca::LocalCa`] never awaits —
22//! its file IO happens once at startup and signing is CPU-bound — but the trait
23//! is shaped for the backends that do. Like [`crate::filter::Check`], it needs
24//! `#[async_trait]`: `Arc<dyn SignerBackend>` with an `async fn` is not dyn-safe.
25//!
26//! Construction stays synchronous: [`from_config`] runs once at startup, where a
27//! failure is fatal anyway.
28//!
29//! ## Certificate validity is a backend policy
30//!
31//! Leaf validity is decided by the backend, not by the caller or the order — see
32//! [`local_ca::LocalCa`], which uses its own `leaf_validity_days`. A delegating
33//! backend would have no validity knob at all (the upstream CA decides).
34//!
35//! ## A backend outlives the configuration it was built from
36//!
37//! A configuration reload rebuilds nearly everything (see [`crate::reload`]),
38//! but a backend that is still configured exactly as it was is **reused
39//! verbatim** — see [`build_backends`], which keys on the configuration's own
40//! `Debug` rendering. Only a backend whose configuration actually moved is
41//! constructed again, and that one adopts the previous instance's in-memory
42//! state through [`CarriedState`]. Both halves matter: without the reuse every
43//! `SIGHUP` would re-read a CA key and re-open a PKCS#11 session for nothing,
44//! and without the adoption a rebuilt backend would start with an empty
45//! revocation ledger and an empty `http-01` token store.
46
47use std::any::Any;
48use std::collections::HashMap;
49use std::sync::Arc;
50
51use async_trait::async_trait;
52use tracing::debug;
53
54use crate::config::SignerConfig;
55use crate::sqlite::db::Database;
56use crate::sqlite::order::Identifier;
57
58pub mod custom;
59pub mod local_ca;
60pub mod relay;
61
62/// Re-exported so [`SignerBackend::http01_tokens`]'s signature — and the route
63/// in [`crate::build_app`] it feeds — do not reach into one backend's module
64/// for a type the generic trait mentions.
65pub use relay::http01::TokenStore as Http01TokenStore;
66
67/// What [`SignerBackend::issue`] produced: a certificate, or a promise of one.
68///
69/// A backend that signs locally answers synchronously with [`Issued`]. A
70/// backend that delegates over the network answers [`Processing`] and finishes
71/// the work in the background, because holding the finalize request open for
72/// an upstream CA's own validation cycle could take minutes — RFC 8555 §7.4
73/// has the `processing` order status for exactly this, and the client polls.
74///
75/// [`Issued`]: IssueOutcome::Issued
76/// [`Processing`]: IssueOutcome::Processing
77#[derive(Debug)]
78pub enum IssueOutcome {
79 /// A finished PEM chain (leaf followed by the issuer).
80 Issued(String),
81 /// The backend accepted the request and will update the `Order` itself
82 /// (via `Order::finalize`/`Order::mark_invalid`) once it resolves. The
83 /// handler moves the order to `processing` and returns it as-is.
84 Processing,
85}
86
87/// A suggested renewal window (RFC 9773 §4.2): when the CA would like this
88/// certificate replaced, and optionally why.
89///
90/// A struct rather than the `(start, end)` tuple this used to be, because
91/// `explanationURL` has nowhere to live in a tuple — and it is precisely the
92/// field a *delegating* backend most wants to pass through, since an upstream
93/// CA setting an unusual window (a mass-revocation event, say) is exactly when
94/// it publishes a page explaining it. §4.2: "Clients SHOULD provide this URL to
95/// their operator, if present."
96#[derive(Debug, Clone, PartialEq, Eq)]
97pub struct RenewalWindow {
98 /// Start of the window, epoch seconds.
99 pub start: i64,
100 /// End of the window, epoch seconds. §4.2 makes a window whose `end` equals
101 /// or precedes its `start` invalid, and servers "MUST NOT serve such a
102 /// response" — see `get_renewal_info`, which enforces that on the way out
103 /// no matter which backend produced the window.
104 pub end: i64,
105 /// A page explaining why the window has this value, if the backend has one.
106 pub explanation_url: Option<String>,
107}
108
109impl RenewalWindow {
110 /// A window with no explanation — what a backend computing its own answer
111 /// from certificate validity returns.
112 #[must_use]
113 pub fn new(start: i64, end: i64) -> Self {
114 Self {
115 start,
116 end,
117 explanation_url: None,
118 }
119 }
120}
121
122/// The validity window an order asked for (RFC 8555 §7.4's `notBefore` /
123/// `notAfter`), in epoch seconds. Either half may be absent, and usually both
124/// are — most clients let the CA decide.
125///
126/// A request, not an instruction: §7.4 lets the server override, and
127/// [`local_ca::LocalCa`] clamps it to its own `leaf_validity_days` rather than
128/// letting a client mint a ten-year certificate. But before this existed the
129/// fields were stored and echoed in the order object while being dropped on the
130/// way to the signer — so a client that asked for a window, and read one back,
131/// got a certificate with a different one and no way to tell.
132#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
133pub struct RequestedValidity {
134 pub not_before: Option<i64>,
135 pub not_after: Option<i64>,
136}
137
138impl RequestedValidity {
139 /// Whether the order asked for anything at all.
140 #[must_use]
141 pub fn is_empty(&self) -> bool {
142 self.not_before.is_none() && self.not_after.is_none()
143 }
144}
145
146/// In-memory state a signer backend owns that has no durable home, carried from
147/// one configuration generation to the next.
148///
149/// Keyed by the **resource** the state describes — a `crl_path`, a relay account
150/// key path — never by profile name and never by configuration identity. That
151/// choice is the whole safety argument: a backend rebuilt over the same files is
152/// the same CA and must not start with an empty ledger, while one rebuilt over
153/// *different* files must never adopt state describing somebody else's. A key
154/// cannot collide, because [`signer_paths`] already refuses two live backends
155/// over one path.
156///
157/// A map of `Arc<dyn Any>` rather than an enum naming each backend's internals,
158/// so `signer/mod.rs` keeps knowing nothing about what a backend holds — the
159/// same line [`SignerBackend::crl_der`] and [`SignerBackend::http01_tokens`]
160/// draw. And a map of *state* rather than a `fn adopt(&self, previous: &dyn
161/// SignerBackend)`, which would need every backend downcast to itself and would
162/// still have to answer "is that previous backend the same thing I am?" — a
163/// question the key already answers, in the open, one resource at a time.
164#[derive(Default)]
165pub struct CarriedState(HashMap<String, Arc<dyn Any + Send + Sync>>);
166
167impl CarriedState {
168 #[must_use]
169 pub fn new() -> Self {
170 Self::default()
171 }
172
173 /// Offers `value` to whichever backend is built over `resource` next.
174 pub fn insert<T: Any + Send + Sync>(&mut self, resource: String, value: Arc<T>) {
175 self.0.insert(resource, value);
176 }
177
178 /// Takes the state recorded for `resource`, if the previous generation left
179 /// any *and* it is of the expected type.
180 ///
181 /// A type mismatch answers `None` rather than panicking: it can only mean a
182 /// backend changed kind over one path (a `local_ca` where a `relay` used to
183 /// be), which is a legitimate reload and should start from disk, not abort.
184 #[must_use]
185 pub fn get<T: Any + Send + Sync>(&self, resource: &str) -> Option<Arc<T>> {
186 self.0.get(resource)?.clone().downcast::<T>().ok()
187 }
188
189 /// Folds another backend's contribution in.
190 pub fn absorb(&mut self, other: Self) {
191 self.0.extend(other.0);
192 }
193
194 /// The resources this state covers, for a caller that wants to log what it
195 /// is carrying.
196 #[must_use]
197 pub fn resources(&self) -> Vec<&str> {
198 let mut names: Vec<&str> = self.0.keys().map(String::as_str).collect();
199 names.sort_unstable();
200 names
201 }
202}
203
204/// A pluggable certificate-issuance backend.
205#[async_trait]
206pub trait SignerBackend: Send + Sync {
207 /// Issues a leaf certificate for the PKCS#10 CSR in `csr_der`.
208 /// `identifiers` are the order's identifiers, so the backend can check the
209 /// CSR requests exactly them.
210 ///
211 /// `order_id` names the local order this issuance belongs to. A
212 /// synchronous backend ignores it; an asynchronous one needs it to find
213 /// the `Order` again from its background task, since by then the handler
214 /// that called this has long returned.
215 ///
216 /// `validity` is what the order asked for (RFC 8555 §7.4). A backend is
217 /// free to ignore it — one that delegates has no say over the upstream's
218 /// policy — but must not silently *contradict* its own advertised limits;
219 /// see [`RequestedValidity`].
220 async fn issue(
221 &self,
222 order_id: &str,
223 csr_der: &[u8],
224 identifiers: &[Identifier],
225 validity: RequestedValidity,
226 ) -> Result<IssueOutcome, SignerError>;
227
228 /// Revokes the certificate `cert_der` (RFC 8555 §7.6), with an optional
229 /// RFC 5280 §5.3.1 `CRLReason` code. Must be idempotent: revoking an
230 /// already-revoked certificate is not an error.
231 ///
232 /// Deliberately takes no `order_id`: revocation needs no per-order state.
233 /// The backend identifies the certificate from its DER, and a delegating
234 /// backend's upstream account already owns the corresponding upstream
235 /// order — the same `kid`-authenticated path `post_revoke_cert` implements
236 /// on this server's own side.
237 async fn revoke(&self, cert_der: &[u8], reason: Option<u32>) -> Result<(), SignerError>;
238
239 /// The backend's current certificate revocation list (RFC 5280), DER
240 /// encoded, if it maintains one servable here. `None` means the backend
241 /// has no CRL of its own (e.g. a delegating backend whose CRL is only
242 /// ever published by the upstream CA it defers to, at a URL of the
243 /// upstream's choosing).
244 async fn crl_der(&self) -> Option<Vec<u8>> {
245 None
246 }
247
248 /// The certificates a client needs to trust what this backend issues, PEM
249 /// encoded, anchor last — served unauthenticated at `GET /ca.pem`.
250 ///
251 /// `None` means the backend has no trust anchor of its own to hand out, and
252 /// the route answers `404`. That is the honest answer for both delegating
253 /// backends: [`relay`]'s anchor belongs to the upstream CA and is published
254 /// wherever that CA chooses, and a `custom` script's is wherever its
255 /// operator put it. Only [`local_ca::LocalCa`] overrides this, which is
256 /// also the only backend that generates an anchor nothing else knows about
257 /// — the case where "fetch it over HTTP" is the difference between one
258 /// `curl` and finding a file on the server's disk.
259 ///
260 /// A getter on the trait for the same reason
261 /// [`crl_der`](SignerBackend::crl_der) is one.
262 async fn ca_chain_pem(&self) -> Option<String> {
263 None
264 }
265
266 /// The backend's opinion on when `cert_der` should be renewed (ACME
267 /// Renewal Information, RFC 9773) — the same
268 /// [`RenewalWindow`] [`crate::handlers::calculate_suggested_window`]
269 /// produces, so the handler can use either interchangeably.
270 ///
271 /// `Ok(None)` — the default, which [`local_ca::LocalCa`] keeps — means
272 /// "no opinion, compute it locally". Only a backend delegating to an
273 /// upstream CA that publishes its own ARI has anything better to say.
274 async fn renewal_info(&self, _cert_der: &[u8]) -> Result<Option<RenewalWindow>, SignerError> {
275 Ok(None)
276 }
277
278 /// This backend's in-flight issuances, as the process-wide relay handler
279 /// sees them, if it resolves issuance asynchronously at all.
280 ///
281 /// Only a backend whose work outlives the request that started it has
282 /// anything to hand over; a synchronous backend like [`local_ca::LocalCa`]
283 /// never has a half-finished issuance, so the default is `None`.
284 ///
285 /// **State, not a [`JobHandler`](crate::jobs::JobHandler)** — the same
286 /// distinction, and for the same reason, as
287 /// [`crl_pruner`](SignerBackend::crl_pruner) above. This method replaced a
288 /// `jobs()` returning one handler per backend, which made two profiles
289 /// relaying to *different* upstreams — two backends, since
290 /// [`build_backends`] deliberately does not collapse them — a startup
291 /// error, `JobRegistry::register` refusing the second handler for
292 /// `signer_relay_issue`. `cli::build_generation` now builds one
293 /// [`relay::flow::RelayJob`] over every relay profile in the process, which
294 /// picks the backend per row from the profile the row names.
295 ///
296 /// There is deliberately no general "here are my job handlers" hook left on
297 /// this trait: every one it could return has this problem, and a subsystem
298 /// that wants a queue registers one handler covering every backend of its
299 /// kind. Recovery is a case of that queue rather than a mechanism of its own
300 /// — see [`crate::jobs::JobHandler::recover`].
301 fn relay_state(&self) -> Option<relay::RelayState> {
302 None
303 }
304
305 /// The `http-01` token store this backend answers the *upstream's* own
306 /// challenge from, if it has one.
307 ///
308 /// [`crate::build_app`] mounts `GET /.well-known/acme-challenge/{token}`
309 /// on the root router when any profile's backend returns `Some`, and not
310 /// at all otherwise — the same "a backend that has something to publish
311 /// over HTTP says so" shape as [`crl_der`](SignerBackend::crl_der), and the
312 /// reason this is a getter on the trait rather than a parameter threaded
313 /// through `build_app`.
314 ///
315 /// Only [`relay`] with `challenge_strategy = "http01"` overrides it.
316 fn http01_tokens(&self) -> Option<Arc<dyn Http01TokenStore>> {
317 None
318 }
319
320 /// This backend's revocation ledger, if it keeps one that grows and can be
321 /// swept (RFC 5280 §3.3).
322 ///
323 /// A getter handing over *state* rather than a
324 /// [`JobHandler`](crate::jobs::JobHandler), and the distinction is not
325 /// cosmetic: [`crate::jobs::JobRegistry::register`] refuses two handlers for
326 /// one `kind`, and two profiles with *different* `[signer.local_ca]`
327 /// sections are two distinct backends — so a handler returned from here
328 /// would make a supported configuration a startup error. Handing over the
329 /// state instead lets `cli::build_generation` build one handler over every
330 /// CA in the process, the shape
331 /// [`http01_tokens`](SignerBackend::http01_tokens) already has for the same
332 /// reason. [`relay_state`](SignerBackend::relay_state) below is the second
333 /// method of this shape, and the trait deliberately has no third form: a
334 /// backend never returns a handler of its own.
335 ///
336 /// Only [`local_ca::LocalCa`] overrides it, and only when it has files to
337 /// persist to. The delegating backends have no ledger of their own — the
338 /// upstream or the script keeps it.
339 fn crl_pruner(&self) -> Option<Arc<dyn CrlPruner>> {
340 None
341 }
342
343 /// What this backend hands to whichever backend replaces it on a
344 /// configuration reload, keyed by the resource each piece describes.
345 ///
346 /// A getter on the trait for the fourth time and for the same reason as
347 /// [`crl_der`](SignerBackend::crl_der) and
348 /// [`relay_state`](SignerBackend::relay_state): what a backend owns is the
349 /// backend's own business. The default is empty,
350 /// which is the honest answer for [`custom::CustomScriptSigner`] — it holds
351 /// nothing between calls — and for any state that already has a durable
352 /// home.
353 ///
354 /// **Durability is not the test, though; a race is.** `local_ca`'s ledger
355 /// *is* persisted, and it is still carried, because a revocation landing on
356 /// the outgoing instance between the incoming one's read of the sidecar and
357 /// the swap would otherwise be lost. Sharing the `Arc` means both instances
358 /// see one ledger for the whole window, so there is nothing to diverge.
359 fn carried_state(&self) -> CarriedState {
360 CarriedState::default()
361 }
362}
363
364/// One backend's revocation ledger, as the periodic sweep sees it.
365///
366/// Deliberately narrow: the sweep has no business knowing what a `LocalCa` is,
367/// and this is the whole of what it needs — something to name in a log line and
368/// something to call. See [`SignerBackend::crl_pruner`] for why the state
369/// travels rather than a [`JobHandler`](crate::jobs::JobHandler).
370#[async_trait]
371pub trait CrlPruner: Send + Sync {
372 /// Which ledger this is, for logging. The same
373 /// [`CarriedState`] key the reload path files it under, so one CA reads as
374 /// one resource wherever it is named.
375 fn state_key(&self) -> String;
376
377 /// Drops entries whose certificates have expired and re-signs the CRL if
378 /// any went, returning how many. Must be cheap and write nothing when
379 /// there was nothing to drop — it runs daily on every CA in the process.
380 async fn prune_expired(&self) -> Result<usize, SignerError>;
381}
382
383/// Why issuance failed, mapped by the handler to the right ACME error:
384/// a client-side CSR problem versus an internal signing failure.
385#[derive(Debug, thiserror::Error)]
386pub enum SignerError {
387 /// The CSR was unparsable or did not match the order's identifiers.
388 /// Maps to `Problem::bad_csr` (400).
389 #[error("Bad CSR")]
390 BadCsr,
391 /// The backend failed to sign (should not happen in normal operation).
392 /// Maps to `Problem::server_internal` (500).
393 #[error("Internal signer error: {0}")]
394 Internal(String),
395}
396
397/// The dependencies every backend is built from, minus its own `[signer]`
398/// section.
399///
400/// A struct for [`ProfileParts`](crate::ProfileParts)' reason: [`from_config`]
401/// took seven positional parameters and needed an eighth for [`CarriedState`],
402/// which is where a reader starts counting commas and clippy starts complaining.
403/// Taken by reference and cloned field by field, since [`build_backends`] calls
404/// [`from_config`] in a loop.
405///
406/// `database` is for the backends that resolve issuance asynchronously: they own
407/// the `Order` update once the answer arrives, long after the handler that asked
408/// for it returned. `local_ca` ignores it. `notifiers` is the same kind of
409/// dependency for the same reason — a backend whose completion happens in a
410/// background task has no `Profile`/`AppState` to reach a notifier through, so it
411/// is handed the whole `profile name -> dispatcher` map and looks up the right
412/// one by `Order.profile` once it has something to report. It arrives as
413/// [`crate::notify::Notifiers`] rather than a bare `Arc` because a backend
414/// outlives the generation that built it while the map does not: a captured
415/// `Arc` would pin the backend to the dispatchers that existed when it was
416/// constructed.
417#[derive(Clone)]
418pub struct SignerParts {
419 pub database: Arc<Database>,
420 pub notifiers: crate::notify::Notifiers,
421 pub metrics: Arc<crate::metrics::Metrics>,
422 /// This generation's outbound plumbing **and** the configuration identity of
423 /// it, held whole rather than as a bare
424 /// [`Outbound`](crate::http_client::Outbound). The two cannot then disagree,
425 /// and a value that disagreed would make a `dns.resolver` edit a silent
426 /// no-op for every signer — see [`build_backends`].
427 pub egress: Arc<crate::Egress>,
428 pub jobs: crate::jobs::JobQueue,
429}
430
431/// Builds the configured signer backend, adopting whatever the generation before
432/// it left for this backend's own resources.
433///
434/// Called at startup and again for any backend a reload rebuilds; a failure is
435/// fatal to whichever of the two it is (the process exits, or the reload is
436/// refused with the running generation untouched).
437pub fn from_config(
438 cfg: &SignerConfig,
439 parts: &SignerParts,
440 carried: &CarriedState,
441) -> anyhow::Result<Arc<dyn SignerBackend>> {
442 match cfg.backend.as_str() {
443 "local_ca" => Ok(Arc::new(local_ca::LocalCa::load_or_generate(
444 &cfg.local_ca,
445 carried,
446 )?)),
447 // The one backend handed the metrics registry, because it is the one
448 // that finishes an issuance from a background task: `post_finalize`
449 // answered `processing` and returned, so no `Auditor` — and no request
450 // — is in scope when the certificate actually arrives.
451 "relay" => Ok(Arc::new(relay::RelaySigner::from_config(
452 &cfg.relay, parts, carried,
453 )?)),
454 "custom" => Ok(Arc::new(custom::CustomScriptSigner::from_config(
455 &cfg.custom,
456 )?)),
457 // The one name worth explaining rather than merely refusing: it was
458 // this backend's own until it was renamed away from the host program's
459 // name, so an operator hitting it has a written-down configuration and
460 // a one-line fix, not a typo. A diagnostic, not a compatibility path —
461 // nothing reads the old spelling, and this arm goes at 1.0.0.
462 "acme_proxy" => anyhow::bail!(
463 "unknown signer backend: acme_proxy — renamed to `relay`. Set \
464 signer.backend = \"relay\" and rename the [signer.acme_proxy] table to \
465 [signer.relay] (environment: ACME_PROXY_SIGNER__ACME_PROXY__* becomes \
466 ACME_PROXY_SIGNER__RELAY__*)"
467 ),
468 other => anyhow::bail!("unknown signer backend: {other}"),
469 }
470}
471
472/// The backends one configuration generation runs, in the two views that are
473/// needed of them.
474///
475/// `by_profile` is what a [`Profile`](crate::Profile) is handed and the only
476/// thing that serves a request. `by_identity` exists purely so the **next**
477/// reload can ask "is this one already built?" — see [`build_backends`], where
478/// answering yes is what keeps a `SIGHUP` from re-reading a CA key and
479/// re-opening a PKCS#11 session for a configuration that did not move.
480#[derive(Default, Clone)]
481pub struct SignerSet {
482 by_profile: HashMap<String, Arc<dyn SignerBackend>>,
483 by_identity: HashMap<String, Arc<dyn SignerBackend>>,
484}
485
486impl SignerSet {
487 /// The backend serving `profile`, if that endpoint is mounted.
488 #[must_use]
489 pub fn get(&self, profile: &str) -> Option<&Arc<dyn SignerBackend>> {
490 self.by_profile.get(profile)
491 }
492
493 /// How many distinct backend instances this set holds — one per distinct
494 /// `[signer]` configuration, not one per profile.
495 #[must_use]
496 pub fn len(&self) -> usize {
497 self.by_identity.len()
498 }
499
500 #[must_use]
501 pub fn is_empty(&self) -> bool {
502 self.by_identity.is_empty()
503 }
504
505 /// Everything these backends would hand to their replacements, folded into
506 /// one map.
507 ///
508 /// Folded across *all* of them rather than matched backend to backend,
509 /// because the keys are resources and [`signer_paths`] already refuses two
510 /// live backends over one path — so a rebuilt backend finds its own state by
511 /// naming its own files, and no ownership analysis is needed here.
512 #[must_use]
513 pub fn carried(&self) -> CarriedState {
514 let mut carried = CarriedState::new();
515 for backend in self.by_identity.values() {
516 carried.absorb(backend.carried_state());
517 }
518 carried
519 }
520}
521
522/// Builds one backend per profile, **sharing** the instance between profiles
523/// whose signer configuration is identical, and **reusing** the instance the
524/// previous generation built for a configuration that has not moved.
525///
526/// Sharing is not an optimization, it is a correctness requirement. Two
527/// `LocalCa` instances over the same files each keep their own in-memory
528/// revocation ledger and rewrite the CRL from it, so the second one to revoke
529/// silently drops the first one's entries. Two `RelaySigner`s over the same
530/// account key would likewise each register a job handler for one kind, which
531/// the registry refuses outright. Hence also
532/// the check below: identical configuration shares one instance, but *different*
533/// configuration touching the same file is refused outright rather than
534/// half-working.
535///
536/// Reuse is the same requirement in the time dimension, and `previous` is what
537/// makes a reload able to touch this at all. Three outcomes per distinct
538/// configuration:
539///
540/// 1. **Already built** — the very same `Arc` comes back. Nothing is adopted
541/// because nothing is constructed; this is the ordinary case, since most
542/// reloads touch `[filter]` or `[notify]` and leave every signer alone.
543/// 2. **New** — built, and handed everything the outgoing generation offered
544/// ([`SignerSet::carried`]). This covers both a profile mounted for the first
545/// time and a live profile whose `[signer]` an operator edited.
546/// 3. **Gone** — no longer named by any profile, so it is simply absent from the
547/// result and dropped once the caller publishes it.
548///
549/// The identity a configuration is keyed by is its `Debug` rendering — every
550/// config type derives `Debug`, the output is deterministic for equal values,
551/// and it is only ever compared to another one, never parsed and never shown —
552/// **plus [`SignerParts::egress`]**. That second half is what lets `[dns]` and
553/// `[proxy]` reload: they are not `[signer]` keys, but every backend that
554/// reaches the network caches them at construction, so a backend reused across a
555/// reload that changed either would keep dialling through the old policy with
556/// nothing saying so.
557pub fn build_backends(
558 profiles: &[crate::config::ProfileConfig],
559 parts: &SignerParts,
560 previous: &SignerSet,
561) -> anyhow::Result<SignerSet> {
562 let key_of = |cfg: &SignerConfig| format!("{cfg:?}|{}", parts.egress.identity);
563
564 let mut owners: HashMap<String, String> = HashMap::new();
565 for profile in profiles {
566 let key = key_of(&profile.sections.signer);
567 for path in signer_paths(&profile.sections.signer) {
568 match owners.get(&path) {
569 Some(existing) if *existing != key => anyhow::bail!(
570 "profile `{}` reuses `{path}` with a different signer configuration: \
571 two backends over one file would overwrite each other's state \
572 (give each profile its own paths, or make their [signer] sections identical)",
573 profile.name
574 ),
575 _ => {
576 owners.insert(path, key.clone());
577 }
578 }
579 }
580 }
581
582 // Gathered once, before anything is built: a backend rebuilt over the same
583 // files must find the live ledger, and the outgoing instances are still
584 // holding it at this point — which is the whole reason the handover is a
585 // shared `Arc` and not a copy.
586 let carried = previous.carried();
587
588 let mut set = SignerSet::default();
589 for profile in profiles {
590 let key = key_of(&profile.sections.signer);
591 let backend = match (set.by_identity.get(&key), previous.by_identity.get(&key)) {
592 (Some(backend), _) => backend.clone(),
593 (None, Some(backend)) => {
594 debug!(
595 event = "signer_backend_reused",
596 outcome = "success",
597 profile = %profile.name,
598 "the configuration did not move, so the running backend is carried \
599 whole rather than rebuilt"
600 );
601 let backend = backend.clone();
602 set.by_identity.insert(key, backend.clone());
603 backend
604 }
605 (None, None) => {
606 let backend = from_config(&profile.sections.signer, parts, &carried)
607 .map_err(|error| anyhow::anyhow!("profile `{}`: {error}", profile.name))?;
608 set.by_identity.insert(key, backend.clone());
609 backend
610 }
611 };
612 set.by_profile.insert(profile.name.clone(), backend);
613 }
614 Ok(set)
615}
616
617/// The files a signer configuration owns — what two profiles must not share
618/// unless they share the whole configuration.
619fn signer_paths(cfg: &SignerConfig) -> Vec<String> {
620 match cfg.backend.as_str() {
621 "local_ca" => {
622 let mut paths = vec![
623 cfg.local_ca.cert_path.clone(),
624 cfg.local_ca.key_path.clone(),
625 cfg.local_ca.crl_path.clone(),
626 ];
627 // A PKCS#11 key is shared state in exactly the way this check
628 // exists for: two `LocalCa`s over one token key would each keep
629 // their own revocation ledger and rewrite the CRL from it. Not a
630 // file, but the same hazard, so it goes in the same list under a
631 // pseudo-path that cannot collide with a real one.
632 if cfg.local_ca.key_source == "pkcs11" {
633 paths.push(format!(
634 "pkcs11:{}#{}#{}#{}",
635 cfg.local_ca.pkcs11.module_path,
636 cfg.local_ca.pkcs11.token_label,
637 cfg.local_ca.pkcs11.key_label,
638 cfg.local_ca.pkcs11.key_id,
639 ));
640 }
641 paths
642 }
643 "relay" => vec![cfg.relay.account_key_path.clone()],
644 _ => Vec::new(),
645 }
646}
647
648#[cfg(test)]
649mod tests {
650 use super::*;
651
652 /// The shared resolver `Profile::build_all` supplies at startup. These
653 /// tests reach loopback by IP literal, which `dns::connect` short-circuits.
654 fn test_resolver() -> std::sync::Arc<dyn crate::dns::Resolver> {
655 std::sync::Arc::new(crate::dns::HickoryResolver::from_system_uncached().unwrap())
656 }
657 use crate::config::LocalCaConfig;
658
659 /// A `SignerConfig` writing its CA material into a throwaway directory, so
660 /// the `local_ca` arm can run without touching the repository's `ca.pem`.
661 fn config(backend: &str) -> (SignerConfig, crate::testutil::TempDir) {
662 let dir = crate::testutil::TempDir::new("signer");
663 let cfg = SignerConfig {
664 backend: backend.to_string(),
665 local_ca: LocalCaConfig {
666 cert_path: dir.join("ca.pem").to_string_lossy().into_owned(),
667 key_path: dir.join("ca.key").to_string_lossy().into_owned(),
668 crl_path: dir.join("ca.crl").to_string_lossy().into_owned(),
669 ..LocalCaConfig::default()
670 },
671 ..SignerConfig::default()
672 };
673 (cfg, dir)
674 }
675
676 async fn database() -> Arc<Database> {
677 Arc::new(Database::connect_in_memory().await.unwrap())
678 }
679
680 /// The dependencies a backend is built from, none of which these tests are
681 /// about. `egress` carries a fixed identity, so a case that wants to prove
682 /// `[dns]`/`[proxy]` reach the identity key overrides it deliberately.
683 async fn parts() -> SignerParts {
684 crate::testutil::signer_parts(database().await, test_resolver())
685 }
686
687 /// `parts()` with a different egress identity — what a reload that changed
688 /// `dns.resolver` or `[proxy]` hands `build_backends`.
689 async fn parts_with_egress(identity: &str) -> SignerParts {
690 let mut parts = parts().await;
691 parts.egress = Arc::new(crate::Egress {
692 resolver: test_resolver(),
693 proxies: crate::testutil::no_proxies(),
694 identity: identity.to_string(),
695 });
696 parts
697 }
698
699 fn profile(name: &str, signer: SignerConfig) -> crate::config::ProfileConfig {
700 crate::config::ProfileConfig {
701 name: name.to_string(),
702 sections: crate::config::ProfileSections {
703 signer,
704 ..crate::config::ProfileSections::default()
705 },
706 }
707 }
708
709 /// A configuration that did not move is **not rebuilt**: the reload gets the
710 /// very same instance back.
711 ///
712 /// The ordinary case, and the one that matters most for cost — most reloads
713 /// touch `[filter]` or `[notify]` and leave every signer alone, and
714 /// rebuilding one there would re-read a CA key and, under
715 /// `key_source = "pkcs11"`, log in to a token again per `SIGHUP`.
716 ///
717 /// `Arc::ptr_eq` against the *previous* set is the only assertion that can
718 /// tell reuse from a rebuild that happened to adopt everything: both produce
719 /// a backend that behaves identically.
720 #[tokio::test]
721 async fn a_configuration_that_did_not_move_is_reused_rather_than_rebuilt() {
722 let (cfg, _dir) = config("local_ca");
723 let profiles = vec![profile("le", cfg)];
724
725 let parts = parts().await;
726 let first = build_backends(&profiles, &parts, &SignerSet::default()).unwrap();
727 let second = build_backends(&profiles, &parts, &first).unwrap();
728
729 assert!(
730 Arc::ptr_eq(first.get("le").unwrap(), second.get("le").unwrap()),
731 "an unchanged `[signer]` must hand back the running instance"
732 );
733 }
734
735 /// A configuration that *did* move is rebuilt — and the new instance shares
736 /// the old one's revocation ledger.
737 ///
738 /// The whole point of [`CarriedState`]. Proven through the CRL rather than
739 /// by inspecting the ledger: a revocation recorded on the outgoing backend
740 /// is visible in the incoming one's CRL, which is what an operator would
741 /// notice if it were not.
742 #[tokio::test]
743 async fn an_edited_configuration_is_rebuilt_over_the_running_ledger() {
744 let (cfg, _dir) = config("local_ca");
745 let parts = parts().await;
746 let running =
747 build_backends(&[profile("le", cfg.clone())], &parts, &SignerSet::default()).unwrap();
748
749 // Something to lose: a certificate issued and revoked by the instance
750 // that is about to be replaced.
751 let outgoing = running.get("le").unwrap().clone();
752 let key_pair = rcgen::KeyPair::generate().unwrap();
753 let params = rcgen::CertificateParams::new(vec!["example.com".to_string()]).unwrap();
754 let csr = params.serialize_request(&key_pair).unwrap();
755 let chain = match outgoing
756 .issue(
757 "ord-1",
758 csr.der(),
759 &[Identifier::dns("example.com")],
760 RequestedValidity::default(),
761 )
762 .await
763 .unwrap()
764 {
765 IssueOutcome::Issued(chain) => chain,
766 IssueOutcome::Processing => panic!("local_ca issues synchronously"),
767 };
768 let leaf = crate::cert::leaf_der_from_chain(&chain).unwrap();
769 outgoing.revoke(&leaf, Some(1)).await.unwrap();
770 let before = outgoing.crl_der().await.expect("a local CA has a CRL");
771
772 // The operator edits one key of `[signer]` and signals.
773 let mut edited = cfg;
774 edited.local_ca.leaf_validity_days = 30;
775 let reloaded = build_backends(&[profile("le", edited)], &parts, &running).unwrap();
776
777 let incoming = reloaded.get("le").unwrap();
778 assert!(
779 !Arc::ptr_eq(&outgoing, incoming),
780 "an edited `[signer]` must really be rebuilt, or the edit did nothing"
781 );
782 assert_eq!(
783 incoming.crl_der().await.expect("a local CA has a CRL"),
784 before,
785 "the rebuilt CA must serve the same CRL, ledger and all",
786 );
787
788 // And the sharing is live in both directions, which is what closes the
789 // window between building the replacement and publishing it.
790 let second = params.serialize_request(&key_pair).unwrap();
791 let chain = match outgoing
792 .issue(
793 "ord-2",
794 second.der(),
795 &[Identifier::dns("example.com")],
796 RequestedValidity::default(),
797 )
798 .await
799 .unwrap()
800 {
801 IssueOutcome::Issued(chain) => chain,
802 IssueOutcome::Processing => unreachable!(),
803 };
804 let leaf = crate::cert::leaf_der_from_chain(&chain).unwrap();
805 outgoing.revoke(&leaf, None).await.unwrap();
806 assert_ne!(
807 incoming.crl_der().await.unwrap(),
808 before,
809 "a revocation landing on the outgoing instance mid-reload must reach \
810 the incoming one — that is the case reading the sidecar back cannot cover",
811 );
812 }
813
814 /// `[dns]` and `[proxy]` are not `[signer]` keys, but a change to either
815 /// still rebuilds every backend.
816 ///
817 /// This is the whole reason those two keys could come off `reload::FROZEN`.
818 /// A backend caches the resolver and proxy policy it was built with, so
819 /// reuse keyed on `[signer]` alone would leave a `dns.resolver` edit
820 /// applying to every subsystem *except* the signers, silently.
821 #[tokio::test]
822 async fn a_changed_egress_rebuilds_a_backend_whose_signer_section_did_not_move() {
823 let (cfg, _dir) = config("local_ca");
824 let profiles = vec![profile("le", cfg)];
825
826 let first = build_backends(
827 &profiles,
828 &parts_with_egress("before").await,
829 &SignerSet::default(),
830 )
831 .unwrap();
832 let second = build_backends(&profiles, &parts_with_egress("after").await, &first).unwrap();
833
834 assert!(
835 !Arc::ptr_eq(first.get("le").unwrap(), second.get("le").unwrap()),
836 "a moved `[dns]`/`[proxy]` must reach the signers, which cache it",
837 );
838 }
839
840 /// A profile mounted by a reload gets a backend; one unmounted leaves its
841 /// backend behind, and it is not carried into the next generation.
842 ///
843 /// Between them these are "mount an endpoint without a restart", which was
844 /// the visible half of the whole freeze.
845 #[tokio::test]
846 async fn mounting_and_unmounting_a_profile_adds_and_drops_its_backend() {
847 let (first_cfg, _first_dir) = config("local_ca");
848 let (second_cfg, _second_dir) = config("local_ca");
849
850 let parts = parts().await;
851 let one = build_backends(
852 &[profile("le", first_cfg.clone())],
853 &parts,
854 &SignerSet::default(),
855 )
856 .unwrap();
857 assert_eq!(one.len(), 1);
858
859 let two = build_backends(
860 &[
861 profile("le", first_cfg),
862 profile("staging", second_cfg.clone()),
863 ],
864 &parts,
865 &one,
866 )
867 .unwrap();
868 assert_eq!(two.len(), 2, "the new endpoint got a backend of its own");
869 assert!(
870 Arc::ptr_eq(one.get("le").unwrap(), two.get("le").unwrap()),
871 "and the endpoint that was already running kept its instance"
872 );
873
874 let back_to_one = build_backends(&[profile("staging", second_cfg)], &parts, &two).unwrap();
875 assert_eq!(back_to_one.len(), 1);
876 assert!(back_to_one.get("le").is_none(), "the endpoint is unmounted");
877 assert!(
878 Arc::ptr_eq(
879 two.get("staging").unwrap(),
880 back_to_one.get("staging").unwrap()
881 ),
882 "the survivor is untouched by its neighbour going away"
883 );
884 }
885
886 /// A rebuild over *different* files starts from those files, never from the
887 /// state describing the old ones.
888 ///
889 /// The safety half of keying [`CarriedState`] on a resource: an operator
890 /// repointing a profile at a second CA must get that CA's revocation
891 /// history, not the first one's.
892 #[tokio::test]
893 async fn a_backend_rebuilt_over_different_files_adopts_nothing() {
894 let (cfg, _dir) = config("local_ca");
895 let parts = parts().await;
896 let running = build_backends(&[profile("le", cfg)], &parts, &SignerSet::default()).unwrap();
897
898 let outgoing = running.get("le").unwrap().clone();
899 let key_pair = rcgen::KeyPair::generate().unwrap();
900 let params = rcgen::CertificateParams::new(vec!["example.com".to_string()]).unwrap();
901 let csr = params.serialize_request(&key_pair).unwrap();
902 let IssueOutcome::Issued(chain) = outgoing
903 .issue(
904 "ord-1",
905 csr.der(),
906 &[Identifier::dns("example.com")],
907 RequestedValidity::default(),
908 )
909 .await
910 .unwrap()
911 else {
912 panic!("local_ca issues synchronously")
913 };
914 outgoing
915 .revoke(&crate::cert::leaf_der_from_chain(&chain).unwrap(), None)
916 .await
917 .unwrap();
918
919 let (elsewhere, _other_dir) = config("local_ca");
920 let reloaded = build_backends(&[profile("le", elsewhere)], &parts, &running).unwrap();
921
922 assert_ne!(
923 reloaded.get("le").unwrap().crl_der().await.unwrap(),
924 outgoing.crl_der().await.unwrap(),
925 "a different `crl_path` is a different CA and starts from its own sidecar",
926 );
927 }
928
929 /// Two endpoints configured identically share **one** backend instance.
930 ///
931 /// Not an optimization: a second `LocalCa` over the same files would keep
932 /// its own revocation ledger and rewrite the CRL from it, silently dropping
933 /// the first one's entries.
934 #[tokio::test]
935 async fn identical_signer_configuration_yields_one_shared_backend() {
936 let (cfg, _dir) = config("local_ca");
937 let profiles = vec![profile("a", cfg.clone()), profile("b", cfg)];
938
939 let backends = build_backends(&profiles, &parts().await, &SignerSet::default()).unwrap();
940 assert_eq!(backends.len(), 1);
941 assert!(
942 Arc::ptr_eq(backends.get("a").unwrap(), backends.get("b").unwrap()),
943 "one configuration must mean one instance"
944 );
945 }
946
947 #[tokio::test]
948 async fn differing_signer_configuration_yields_separate_backends() {
949 let (first, dir_a) = config("local_ca");
950 let (second, dir_b) = config("local_ca");
951 let profiles = vec![profile("a", first), profile("b", second)];
952
953 let backends = build_backends(&profiles, &parts().await, &SignerSet::default()).unwrap();
954 assert!(
955 !Arc::ptr_eq(backends.get("a").unwrap(), backends.get("b").unwrap()),
956 "different CA material must mean different CAs"
957 );
958
959 std::fs::remove_dir_all(dir_a).ok();
960 std::fs::remove_dir_all(dir_b).ok();
961 }
962
963 /// Sharing files while disagreeing about anything else is refused outright:
964 /// the two instances would overwrite each other's state, and the failure
965 /// would only show up as a mysteriously short CRL much later.
966 #[tokio::test]
967 async fn sharing_ca_files_with_a_different_configuration_is_a_startup_error() {
968 let (first, _dir) = config("local_ca");
969 let mut second = first.clone();
970 second.local_ca.leaf_validity_days = 7;
971
972 let profiles = vec![profile("a", first), profile("b", second)];
973 let error = match build_backends(&profiles, &parts().await, &SignerSet::default()) {
974 Err(error) => error.to_string(),
975 Ok(_) => panic!("two backends over one key file must not both be built"),
976 };
977 assert!(error.contains("different signer configuration"), "{error}");
978 assert!(
979 error.contains("ca.key") || error.contains("ca.pem"),
980 "{error}"
981 );
982 }
983
984 #[tokio::test]
985 async fn a_backend_failure_names_the_profile_it_came_from() {
986 let profiles = vec![profile(
987 "le",
988 SignerConfig {
989 backend: "nope".to_string(),
990 ..SignerConfig::default()
991 },
992 )];
993
994 let error = match build_backends(&profiles, &parts().await, &SignerSet::default()) {
995 Err(error) => error.to_string(),
996 Ok(_) => panic!("an unknown backend is a startup error"),
997 };
998 assert!(error.contains("profile `le`"), "{error}");
999 }
1000
1001 #[tokio::test]
1002 async fn builds_the_local_ca_backend_and_it_can_issue() {
1003 let (cfg, _dir) = config("local_ca");
1004 let signer = from_config(&cfg, &parts().await, &CarriedState::new())
1005 .expect("local_ca is a known backend");
1006
1007 // Reached through the trait object, which is how handlers see it.
1008 let key_pair = rcgen::KeyPair::generate().unwrap();
1009 let params = rcgen::CertificateParams::new(vec!["example.com".to_string()]).unwrap();
1010 let csr = params.serialize_request(&key_pair).unwrap();
1011 let outcome = signer
1012 .issue(
1013 "ord-1",
1014 csr.der(),
1015 &[Identifier::dns("example.com")],
1016 RequestedValidity::default(),
1017 )
1018 .await
1019 .unwrap();
1020 // A local CA answers synchronously; only a delegating backend defers.
1021 let chain = match outcome {
1022 IssueOutcome::Issued(chain) => chain,
1023 IssueOutcome::Processing => panic!("local_ca must issue synchronously"),
1024 };
1025 assert_eq!(chain.matches("-----BEGIN CERTIFICATE-----").count(), 2);
1026 }
1027
1028 #[tokio::test]
1029 async fn builds_the_custom_backend_and_it_can_issue() {
1030 let dir = crate::testutil::TempDir::new("signer");
1031 let script_path = dir.join("issue.sh");
1032 std::fs::write(
1033 &script_path,
1034 "#!/bin/sh\ncat > /dev/null\necho '-----BEGIN CERTIFICATE-----leaf-----END CERTIFICATE-----'\nexit 0\n",
1035 )
1036 .unwrap();
1037 #[cfg(unix)]
1038 {
1039 use std::os::unix::fs::PermissionsExt;
1040 std::fs::set_permissions(&script_path, std::fs::Permissions::from_mode(0o755)).unwrap();
1041 }
1042
1043 let cfg = SignerConfig {
1044 backend: "custom".to_string(),
1045 custom: crate::config::CustomSignerConfig {
1046 script_path: script_path.to_string_lossy().into_owned(),
1047 ..Default::default()
1048 },
1049 ..SignerConfig::default()
1050 };
1051 let signer = from_config(&cfg, &parts().await, &CarriedState::new())
1052 .expect("custom is a known backend");
1053
1054 let outcome = signer
1055 .issue(
1056 "ord-1",
1057 &[0x30, 0x00],
1058 &[Identifier::dns("example.com")],
1059 RequestedValidity::default(),
1060 )
1061 .await
1062 .unwrap();
1063 assert!(matches!(outcome, IssueOutcome::Issued(chain) if chain.contains("leaf")));
1064 }
1065
1066 /// `local_ca` has no upstream to ask, so it must keep the trait's default
1067 /// "no opinion" answer — that is what makes `get_renewal_info` fall back to
1068 /// its own local computation.
1069 #[tokio::test]
1070 async fn the_local_ca_backend_has_no_renewal_info_opinion() {
1071 let (cfg, _dir) = config("local_ca");
1072 let signer = from_config(&cfg, &parts().await, &CarriedState::new()).unwrap();
1073 assert!(matches!(signer.renewal_info(&[0x30, 0x00]).await, Ok(None)));
1074 }
1075
1076 /// A typo in `signer.backend` stops the server rather than silently leaving
1077 /// it unable to issue.
1078 #[tokio::test]
1079 async fn an_unknown_backend_is_a_startup_error() {
1080 let (cfg, _dir) = config("hashicorp-vault");
1081 // `Arc<dyn SignerBackend>` is not `Debug`, so `unwrap_err` is unavailable.
1082 let error = match from_config(&cfg, &parts().await, &CarriedState::new()) {
1083 Err(error) => error.to_string(),
1084 Ok(_) => panic!("an unknown backend must not build"),
1085 };
1086 assert!(
1087 error.contains("unknown signer backend") && error.contains("hashicorp-vault"),
1088 "{error}"
1089 );
1090 }
1091
1092 /// The one unknown backend that is a renamed key rather than a typo:
1093 /// `acme_proxy` was this backend's own name until it was renamed away from
1094 /// the host program's. The refusal has to carry the new name and the new
1095 /// environment prefix, since neither is guessable from "unknown signer
1096 /// backend" alone.
1097 #[tokio::test]
1098 async fn the_old_acme_proxy_backend_name_is_refused_by_its_new_one() {
1099 let (cfg, _dir) = config("acme_proxy");
1100 let error = match from_config(&cfg, &parts().await, &CarriedState::new()) {
1101 Err(error) => error.to_string(),
1102 Ok(_) => panic!("the old backend name must not build"),
1103 };
1104 for expected in [
1105 "acme_proxy",
1106 "`relay`",
1107 "[signer.relay]",
1108 "ACME_PROXY_SIGNER__RELAY__",
1109 ] {
1110 assert!(error.contains(expected), "{expected} missing from: {error}");
1111 }
1112 }
1113
1114 /// Both variants render. `SignerError` is what a handler logs when
1115 /// issuance fails, so a variant with no message would leave nothing behind.
1116 #[test]
1117 fn signer_errors_render_their_kind() {
1118 assert_eq!(SignerError::BadCsr.to_string(), "Bad CSR");
1119 assert_eq!(
1120 SignerError::Internal("ca offline".to_string()).to_string(),
1121 "Internal signer error: ca offline"
1122 );
1123 }
1124}