acme_proxy/signer/mod.rs
1//! Certificate-issuance abstraction.
2//!
3//! Finalizing an ACME order turns the client's CSR into an issued certificate.
4//! *How* that happens is pluggable: the [`SignerBackend`] trait hides the backend
5//! behind a single [`issue`](SignerBackend::issue) call, and [`from_config`]
6//! builds the configured one at startup. The only backend implemented today is
7//! [`local_ca::LocalCa`] — a persistent local CA.
8//!
9//! Revocation (RFC 8555 §7.6) is part of the same abstraction:
10//! [`revoke`](SignerBackend::revoke) must actually revoke the certificate at
11//! the backend, not just at the ACME/database layer — for [`local_ca::LocalCa`]
12//! that means a real, CA-signed CRL. [`crl_der`](SignerBackend::crl_der) is how
13//! a backend that maintains one serves it (`GET /crl`); it defaults to `None`
14//! for a backend with no CRL of its own to publish here (e.g. one delegating to
15//! an upstream CA that publishes its own).
16//!
17//! ## Asynchronous by design
18//!
19//! [`issue`](SignerBackend::issue) is **async**, so a backend that *delegates*
20//! signing over the network (an upstream ACME CA, a remote signer) can await its
21//! IO instead of blocking a runtime thread. [`local_ca::LocalCa`] never awaits —
22//! its file IO happens once at startup and signing is CPU-bound — but the trait
23//! is shaped for the backends that do. Like [`crate::filter::Check`], it needs
24//! `#[async_trait]`: `Arc<dyn SignerBackend>` with an `async fn` is not dyn-safe.
25//!
26//! Construction stays synchronous: [`from_config`] runs once at startup, where a
27//! failure is fatal anyway.
28//!
29//! ## Certificate validity is a backend policy
30//!
31//! Leaf validity is decided by the backend, not by the caller or the order — see
32//! [`local_ca::LocalCa`], which uses its own `leaf_validity_days`. A delegating
33//! backend would have no validity knob at all (the upstream CA decides).
34//!
35//! ## A backend outlives the configuration it was built from
36//!
37//! A configuration reload rebuilds nearly everything (see [`crate::reload`]),
38//! but a backend that is still configured exactly as it was is **reused
39//! verbatim** — see [`build_backends`], which keys on the configuration's own
40//! `Debug` rendering. Only a backend whose configuration actually moved is
41//! constructed again, and that one adopts the previous instance's in-memory
42//! state through [`CarriedState`]. Both halves matter: without the reuse every
43//! `SIGHUP` would re-read a CA key and re-open a PKCS#11 session for nothing,
44//! and without the adoption a rebuilt backend would start with an empty
45//! revocation ledger and an empty `http-01` token store.
46
47use std::any::Any;
48use std::collections::HashMap;
49use std::sync::Arc;
50
51use async_trait::async_trait;
52use tracing::debug;
53
54use crate::config::SignerConfig;
55use crate::sqlite::db::Database;
56use crate::sqlite::order::Identifier;
57
58pub mod custom;
59pub mod local_ca;
60pub mod relay;
61
62/// Re-exported so [`SignerBackend::http01_tokens`]'s signature — and the route
63/// in [`crate::build_app`] it feeds — do not reach into one backend's module
64/// for a type the generic trait mentions.
65pub use relay::http01::TokenStore as Http01TokenStore;
66
67/// What [`SignerBackend::issue`] produced: a certificate, or a promise of one.
68///
69/// A backend that signs locally answers synchronously with [`Issued`]. A
70/// backend that delegates over the network answers [`Processing`] and finishes
71/// the work in the background, because holding the finalize request open for
72/// an upstream CA's own validation cycle could take minutes — RFC 8555 §7.4
73/// has the `processing` order status for exactly this, and the client polls.
74///
75/// [`Issued`]: IssueOutcome::Issued
76/// [`Processing`]: IssueOutcome::Processing
77#[derive(Debug)]
78pub enum IssueOutcome {
79 /// A finished PEM chain (leaf followed by the issuer).
80 Issued(String),
81 /// The backend accepted the request and will update the `Order` itself
82 /// (via `Order::finalize`/`Order::mark_invalid`) once it resolves. The
83 /// handler moves the order to `processing` and returns it as-is.
84 Processing,
85}
86
87/// A suggested renewal window (RFC 9773 §4.2): when the CA would like this
88/// certificate replaced, and optionally why.
89///
90/// A struct rather than the `(start, end)` tuple this used to be, because
91/// `explanationURL` has nowhere to live in a tuple — and it is precisely the
92/// field a *delegating* backend most wants to pass through, since an upstream
93/// CA setting an unusual window (a mass-revocation event, say) is exactly when
94/// it publishes a page explaining it. §4.2: "Clients SHOULD provide this URL to
95/// their operator, if present."
96#[derive(Debug, Clone, PartialEq, Eq)]
97pub struct RenewalWindow {
98 /// Start of the window, epoch seconds.
99 pub start: i64,
100 /// End of the window, epoch seconds. §4.2 makes a window whose `end` equals
101 /// or precedes its `start` invalid, and servers "MUST NOT serve such a
102 /// response" — see `get_renewal_info`, which enforces that on the way out
103 /// no matter which backend produced the window.
104 pub end: i64,
105 /// A page explaining why the window has this value, if the backend has one.
106 pub explanation_url: Option<String>,
107}
108
109impl RenewalWindow {
110 /// A window with no explanation — what a backend computing its own answer
111 /// from certificate validity returns.
112 #[must_use]
113 pub fn new(start: i64, end: i64) -> Self {
114 Self {
115 start,
116 end,
117 explanation_url: None,
118 }
119 }
120}
121
122/// The validity window an order asked for (RFC 8555 §7.4's `notBefore` /
123/// `notAfter`), in epoch seconds. Either half may be absent, and usually both
124/// are — most clients let the CA decide.
125///
126/// A request, not an instruction: §7.4 lets the server override, and
127/// [`local_ca::LocalCa`] clamps it to its own `leaf_validity_days` rather than
128/// letting a client mint a ten-year certificate. But before this existed the
129/// fields were stored and echoed in the order object while being dropped on the
130/// way to the signer — so a client that asked for a window, and read one back,
131/// got a certificate with a different one and no way to tell.
132#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
133pub struct RequestedValidity {
134 pub not_before: Option<i64>,
135 pub not_after: Option<i64>,
136}
137
138impl RequestedValidity {
139 /// Whether the order asked for anything at all.
140 #[must_use]
141 pub fn is_empty(&self) -> bool {
142 self.not_before.is_none() && self.not_after.is_none()
143 }
144}
145
146/// In-memory state a signer backend owns that has no durable home, carried from
147/// one configuration generation to the next.
148///
149/// Keyed by the **resource** the state describes — a `crl_path`, a relay account
150/// key path — never by profile name and never by configuration identity. That
151/// choice is the whole safety argument: a backend rebuilt over the same files is
152/// the same CA and must not start with an empty ledger, while one rebuilt over
153/// *different* files must never adopt state describing somebody else's. A key
154/// cannot collide, because [`signer_paths`] already refuses two live backends
155/// over one path.
156///
157/// A map of `Arc<dyn Any>` rather than an enum naming each backend's internals,
158/// so `signer/mod.rs` keeps knowing nothing about what a backend holds — the
159/// same line [`SignerBackend::crl_der`] and [`SignerBackend::http01_tokens`]
160/// draw. And a map of *state* rather than a `fn adopt(&self, previous: &dyn
161/// SignerBackend)`, which would need every backend downcast to itself and would
162/// still have to answer "is that previous backend the same thing I am?" — a
163/// question the key already answers, in the open, one resource at a time.
164#[derive(Default)]
165pub struct CarriedState(HashMap<String, Arc<dyn Any + Send + Sync>>);
166
167impl CarriedState {
168 #[must_use]
169 pub fn new() -> Self {
170 Self::default()
171 }
172
173 /// Offers `value` to whichever backend is built over `resource` next.
174 pub fn insert<T: Any + Send + Sync>(&mut self, resource: String, value: Arc<T>) {
175 self.0.insert(resource, value);
176 }
177
178 /// Takes the state recorded for `resource`, if the previous generation left
179 /// any *and* it is of the expected type.
180 ///
181 /// A type mismatch answers `None` rather than panicking: it can only mean a
182 /// backend changed kind over one path (a `local_ca` where a `relay` used to
183 /// be), which is a legitimate reload and should start from disk, not abort.
184 #[must_use]
185 pub fn get<T: Any + Send + Sync>(&self, resource: &str) -> Option<Arc<T>> {
186 self.0.get(resource)?.clone().downcast::<T>().ok()
187 }
188
189 /// Folds another backend's contribution in.
190 pub fn absorb(&mut self, other: Self) {
191 self.0.extend(other.0);
192 }
193
194 /// The resources this state covers, for a caller that wants to log what it
195 /// is carrying.
196 #[must_use]
197 pub fn resources(&self) -> Vec<&str> {
198 let mut names: Vec<&str> = self.0.keys().map(String::as_str).collect();
199 names.sort_unstable();
200 names
201 }
202}
203
204/// A pluggable certificate-issuance backend.
205#[async_trait]
206pub trait SignerBackend: Send + Sync {
207 /// Issues a leaf certificate for the PKCS#10 CSR in `csr_der`.
208 /// `identifiers` are the order's identifiers, so the backend can check the
209 /// CSR requests exactly them.
210 ///
211 /// `order_id` names the local order this issuance belongs to. A
212 /// synchronous backend ignores it; an asynchronous one needs it to find
213 /// the `Order` again from its background task, since by then the handler
214 /// that called this has long returned.
215 ///
216 /// `validity` is what the order asked for (RFC 8555 §7.4). A backend is
217 /// free to ignore it — one that delegates has no say over the upstream's
218 /// policy — but must not silently *contradict* its own advertised limits;
219 /// see [`RequestedValidity`].
220 async fn issue(
221 &self,
222 order_id: &str,
223 csr_der: &[u8],
224 identifiers: &[Identifier],
225 validity: RequestedValidity,
226 ) -> Result<IssueOutcome, SignerError>;
227
228 /// Revokes the certificate `cert_der` (RFC 8555 §7.6), with an optional
229 /// RFC 5280 §5.3.1 `CRLReason` code. Must be idempotent: revoking an
230 /// already-revoked certificate is not an error.
231 ///
232 /// Deliberately takes no `order_id`: revocation needs no per-order state.
233 /// The backend identifies the certificate from its DER, and a delegating
234 /// backend's upstream account already owns the corresponding upstream
235 /// order — the same `kid`-authenticated path `post_revoke_cert` implements
236 /// on this server's own side.
237 async fn revoke(&self, cert_der: &[u8], reason: Option<u32>) -> Result<(), SignerError>;
238
239 /// The backend's current certificate revocation list (RFC 5280), DER
240 /// encoded, if it maintains one servable here. `None` means the backend
241 /// has no CRL of its own (e.g. a delegating backend whose CRL is only
242 /// ever published by the upstream CA it defers to, at a URL of the
243 /// upstream's choosing).
244 async fn crl_der(&self) -> Option<Vec<u8>> {
245 None
246 }
247
248 /// The certificates a client needs to trust what this backend issues, PEM
249 /// encoded, anchor last — served unauthenticated at `GET /ca.pem`.
250 ///
251 /// `None` means the backend has no trust anchor of its own to hand out, and
252 /// the route answers `404`. That is the honest answer for both delegating
253 /// backends: [`relay`]'s anchor belongs to the upstream CA and is published
254 /// wherever that CA chooses, and a `custom` script's is wherever its
255 /// operator put it. Only [`local_ca::LocalCa`] overrides this, which is
256 /// also the only backend that generates an anchor nothing else knows about
257 /// — the case where "fetch it over HTTP" is the difference between one
258 /// `curl` and finding a file on the server's disk.
259 ///
260 /// A getter on the trait for the same reason
261 /// [`crl_der`](SignerBackend::crl_der) is one.
262 async fn ca_chain_pem(&self) -> Option<String> {
263 None
264 }
265
266 /// The backend's opinion on when `cert_der` should be renewed (ACME
267 /// Renewal Information, RFC 9773) — the same
268 /// [`RenewalWindow`] [`crate::handlers::calculate_suggested_window`]
269 /// produces, so the handler can use either interchangeably.
270 ///
271 /// `Ok(None)` — the default, which [`local_ca::LocalCa`] keeps — means
272 /// "no opinion, compute it locally". Only a backend delegating to an
273 /// upstream CA that publishes its own ARI has anything better to say.
274 async fn renewal_info(&self, _cert_der: &[u8]) -> Result<Option<RenewalWindow>, SignerError> {
275 Ok(None)
276 }
277
278 /// The background work this backend needs a durable queue for.
279 ///
280 /// Only a backend that resolves issuance asynchronously has anything to
281 /// register: its work outlives the request that started it and must survive
282 /// the process. A synchronous backend like [`local_ca::LocalCa`] never has a
283 /// half-finished issuance at all, so the default is an empty list.
284 ///
285 /// A getter on the trait for the same reason
286 /// [`crl_der`](SignerBackend::crl_der) and
287 /// [`http01_tokens`](SignerBackend::http01_tokens) are: whether a backend has
288 /// something to publish, or to run later, is the backend's own business, and
289 /// the alternative is threading a registry through `build_backends` for the
290 /// two implementations that have nothing to put in it.
291 ///
292 /// This replaced a `resume` hook that took no arguments and returned
293 /// nothing, which each asynchronous backend implemented by re-spawning its
294 /// own tasks at startup. Recovery is now one case of a queue rather than a
295 /// mechanism of its own — see [`crate::jobs::JobHandler::recover`], which is
296 /// where that logic went.
297 fn jobs(&self) -> Vec<Arc<dyn crate::jobs::JobHandler>> {
298 Vec::new()
299 }
300
301 /// The `http-01` token store this backend answers the *upstream's* own
302 /// challenge from, if it has one.
303 ///
304 /// [`crate::build_app`] mounts `GET /.well-known/acme-challenge/{token}`
305 /// on the root router when any profile's backend returns `Some`, and not
306 /// at all otherwise — the same "a backend that has something to publish
307 /// over HTTP says so" shape as [`crl_der`](SignerBackend::crl_der), and the
308 /// reason this is a getter on the trait rather than a parameter threaded
309 /// through `build_app`.
310 ///
311 /// Only [`relay`] with `challenge_strategy = "http01"` overrides it.
312 fn http01_tokens(&self) -> Option<Arc<dyn Http01TokenStore>> {
313 None
314 }
315
316 /// This backend's revocation ledger, if it keeps one that grows and can be
317 /// swept (RFC 5280 §3.3).
318 ///
319 /// A getter rather than an entry in [`jobs`](SignerBackend::jobs), and the
320 /// distinction is not cosmetic: [`crate::jobs::JobRegistry::register`]
321 /// refuses two handlers for one `kind`, and two profiles with *different*
322 /// `[signer.local_ca]` sections are two distinct backends — so a handler
323 /// returned from `jobs()` would make a supported configuration a startup
324 /// error. Handing over the *state* instead lets `cli::build_generation`
325 /// build one handler over every CA in the process, the shape
326 /// [`http01_tokens`](SignerBackend::http01_tokens) already has for the same
327 /// reason.
328 ///
329 /// Only [`local_ca::LocalCa`] overrides it, and only when it has files to
330 /// persist to. The delegating backends have no ledger of their own — the
331 /// upstream or the script keeps it.
332 fn crl_pruner(&self) -> Option<Arc<dyn CrlPruner>> {
333 None
334 }
335
336 /// What this backend hands to whichever backend replaces it on a
337 /// configuration reload, keyed by the resource each piece describes.
338 ///
339 /// A getter on the trait for the third time and for the same reason as
340 /// [`crl_der`](SignerBackend::crl_der) and [`jobs`](SignerBackend::jobs):
341 /// what a backend owns is the backend's own business. The default is empty,
342 /// which is the honest answer for [`custom::CustomScriptSigner`] — it holds
343 /// nothing between calls — and for any state that already has a durable
344 /// home.
345 ///
346 /// **Durability is not the test, though; a race is.** `local_ca`'s ledger
347 /// *is* persisted, and it is still carried, because a revocation landing on
348 /// the outgoing instance between the incoming one's read of the sidecar and
349 /// the swap would otherwise be lost. Sharing the `Arc` means both instances
350 /// see one ledger for the whole window, so there is nothing to diverge.
351 fn carried_state(&self) -> CarriedState {
352 CarriedState::default()
353 }
354}
355
356/// One backend's revocation ledger, as the periodic sweep sees it.
357///
358/// Deliberately narrow: the sweep has no business knowing what a `LocalCa` is,
359/// and this is the whole of what it needs — something to name in a log line and
360/// something to call. See [`SignerBackend::crl_pruner`] for why the state
361/// travels rather than a [`JobHandler`](crate::jobs::JobHandler).
362#[async_trait]
363pub trait CrlPruner: Send + Sync {
364 /// Which ledger this is, for logging. The same
365 /// [`CarriedState`] key the reload path files it under, so one CA reads as
366 /// one resource wherever it is named.
367 fn state_key(&self) -> String;
368
369 /// Drops entries whose certificates have expired and re-signs the CRL if
370 /// any went, returning how many. Must be cheap and write nothing when
371 /// there was nothing to drop — it runs daily on every CA in the process.
372 async fn prune_expired(&self) -> Result<usize, SignerError>;
373}
374
375/// Why issuance failed, mapped by the handler to the right ACME error:
376/// a client-side CSR problem versus an internal signing failure.
377#[derive(Debug, thiserror::Error)]
378pub enum SignerError {
379 /// The CSR was unparsable or did not match the order's identifiers.
380 /// Maps to `Problem::bad_csr` (400).
381 #[error("Bad CSR")]
382 BadCsr,
383 /// The backend failed to sign (should not happen in normal operation).
384 /// Maps to `Problem::server_internal` (500).
385 #[error("Internal signer error: {0}")]
386 Internal(String),
387}
388
389/// The dependencies every backend is built from, minus its own `[signer]`
390/// section.
391///
392/// A struct for [`ProfileParts`](crate::ProfileParts)' reason: [`from_config`]
393/// took seven positional parameters and needed an eighth for [`CarriedState`],
394/// which is where a reader starts counting commas and clippy starts complaining.
395/// Taken by reference and cloned field by field, since [`build_backends`] calls
396/// [`from_config`] in a loop.
397///
398/// `database` is for the backends that resolve issuance asynchronously: they own
399/// the `Order` update once the answer arrives, long after the handler that asked
400/// for it returned. `local_ca` ignores it. `notifiers` is the same kind of
401/// dependency for the same reason — a backend whose completion happens in a
402/// background task has no `Profile`/`AppState` to reach a notifier through, so it
403/// is handed the whole `profile name -> dispatcher` map and looks up the right
404/// one by `Order.profile` once it has something to report. It arrives as
405/// [`crate::notify::Notifiers`] rather than a bare `Arc` because a backend
406/// outlives the generation that built it while the map does not: a captured
407/// `Arc` would pin the backend to the dispatchers that existed when it was
408/// constructed.
409#[derive(Clone)]
410pub struct SignerParts {
411 pub database: Arc<Database>,
412 pub notifiers: crate::notify::Notifiers,
413 pub metrics: Arc<crate::metrics::Metrics>,
414 /// This generation's outbound plumbing **and** the configuration identity of
415 /// it, held whole rather than as a bare
416 /// [`Outbound`](crate::http_client::Outbound). The two cannot then disagree,
417 /// and a value that disagreed would make a `dns.resolver` edit a silent
418 /// no-op for every signer — see [`build_backends`].
419 pub egress: Arc<crate::Egress>,
420 pub jobs: crate::jobs::JobQueue,
421}
422
423/// Builds the configured signer backend, adopting whatever the generation before
424/// it left for this backend's own resources.
425///
426/// Called at startup and again for any backend a reload rebuilds; a failure is
427/// fatal to whichever of the two it is (the process exits, or the reload is
428/// refused with the running generation untouched).
429pub fn from_config(
430 cfg: &SignerConfig,
431 profiles: Vec<String>,
432 parts: &SignerParts,
433 carried: &CarriedState,
434) -> anyhow::Result<Arc<dyn SignerBackend>> {
435 match cfg.backend.as_str() {
436 "local_ca" => Ok(Arc::new(local_ca::LocalCa::load_or_generate(
437 &cfg.local_ca,
438 carried,
439 )?)),
440 // The one backend handed the metrics registry, because it is the one
441 // that finishes an issuance from a background task: `post_finalize`
442 // answered `processing` and returned, so no `Auditor` — and no request
443 // — is in scope when the certificate actually arrives.
444 "relay" => Ok(Arc::new(relay::RelaySigner::from_config(
445 &cfg.relay, profiles, parts, carried,
446 )?)),
447 "custom" => Ok(Arc::new(custom::CustomScriptSigner::from_config(
448 &cfg.custom,
449 )?)),
450 // The one name worth explaining rather than merely refusing: it was
451 // this backend's own until it was renamed away from the host program's
452 // name, so an operator hitting it has a written-down configuration and
453 // a one-line fix, not a typo. A diagnostic, not a compatibility path —
454 // nothing reads the old spelling, and this arm goes at 1.0.0.
455 "acme_proxy" => anyhow::bail!(
456 "unknown signer backend: acme_proxy — renamed to `relay`. Set \
457 signer.backend = \"relay\" and rename the [signer.acme_proxy] table to \
458 [signer.relay] (environment: ACME_PROXY_SIGNER__ACME_PROXY__* becomes \
459 ACME_PROXY_SIGNER__RELAY__*)"
460 ),
461 other => anyhow::bail!("unknown signer backend: {other}"),
462 }
463}
464
465/// The backends one configuration generation runs, in the two views that are
466/// needed of them.
467///
468/// `by_profile` is what a [`Profile`](crate::Profile) is handed and the only
469/// thing that serves a request. `by_identity` exists purely so the **next**
470/// reload can ask "is this one already built?" — see [`build_backends`], where
471/// answering yes is what keeps a `SIGHUP` from re-reading a CA key and
472/// re-opening a PKCS#11 session for a configuration that did not move.
473#[derive(Default, Clone)]
474pub struct SignerSet {
475 by_profile: HashMap<String, Arc<dyn SignerBackend>>,
476 by_identity: HashMap<String, Arc<dyn SignerBackend>>,
477}
478
479impl SignerSet {
480 /// The backend serving `profile`, if that endpoint is mounted.
481 #[must_use]
482 pub fn get(&self, profile: &str) -> Option<&Arc<dyn SignerBackend>> {
483 self.by_profile.get(profile)
484 }
485
486 /// How many distinct backend instances this set holds — one per distinct
487 /// `[signer]` configuration, not one per profile.
488 #[must_use]
489 pub fn len(&self) -> usize {
490 self.by_identity.len()
491 }
492
493 #[must_use]
494 pub fn is_empty(&self) -> bool {
495 self.by_identity.is_empty()
496 }
497
498 /// Everything these backends would hand to their replacements, folded into
499 /// one map.
500 ///
501 /// Folded across *all* of them rather than matched backend to backend,
502 /// because the keys are resources and [`signer_paths`] already refuses two
503 /// live backends over one path — so a rebuilt backend finds its own state by
504 /// naming its own files, and no ownership analysis is needed here.
505 #[must_use]
506 pub fn carried(&self) -> CarriedState {
507 let mut carried = CarriedState::new();
508 for backend in self.by_identity.values() {
509 carried.absorb(backend.carried_state());
510 }
511 carried
512 }
513}
514
515/// Builds one backend per profile, **sharing** the instance between profiles
516/// whose signer configuration is identical, and **reusing** the instance the
517/// previous generation built for a configuration that has not moved.
518///
519/// Sharing is not an optimization, it is a correctness requirement. Two
520/// `LocalCa` instances over the same files each keep their own in-memory
521/// revocation ledger and rewrite the CRL from it, so the second one to revoke
522/// silently drops the first one's entries. Two `RelaySigner`s over the same
523/// account key would likewise each register a job handler for one kind, which
524/// the registry refuses outright. Hence also
525/// the check below: identical configuration shares one instance, but *different*
526/// configuration touching the same file is refused outright rather than
527/// half-working.
528///
529/// Reuse is the same requirement in the time dimension, and `previous` is what
530/// makes a reload able to touch this at all. Three outcomes per distinct
531/// configuration:
532///
533/// 1. **Already built** — the very same `Arc` comes back. Nothing is adopted
534/// because nothing is constructed; this is the ordinary case, since most
535/// reloads touch `[filter]` or `[notify]` and leave every signer alone.
536/// 2. **New** — built, and handed everything the outgoing generation offered
537/// ([`SignerSet::carried`]). This covers both a profile mounted for the first
538/// time and a live profile whose `[signer]` an operator edited.
539/// 3. **Gone** — no longer named by any profile, so it is simply absent from the
540/// result and dropped once the caller publishes it.
541///
542/// The identity a configuration is keyed by is its `Debug` rendering — every
543/// config type derives `Debug`, the output is deterministic for equal values,
544/// and it is only ever compared to another one, never parsed and never shown —
545/// **plus [`SignerParts::egress`]**. That second half is what lets `[dns]` and
546/// `[proxy]` reload: they are not `[signer]` keys, but every backend that
547/// reaches the network caches them at construction, so a backend reused across a
548/// reload that changed either would keep dialling through the old policy with
549/// nothing saying so.
550pub fn build_backends(
551 profiles: &[crate::config::ProfileConfig],
552 parts: &SignerParts,
553 previous: &SignerSet,
554) -> anyhow::Result<SignerSet> {
555 let key_of = |cfg: &SignerConfig| format!("{cfg:?}|{}", parts.egress.identity);
556
557 let mut owners: HashMap<String, String> = HashMap::new();
558 for profile in profiles {
559 let key = key_of(&profile.sections.signer);
560 for path in signer_paths(&profile.sections.signer) {
561 match owners.get(&path) {
562 Some(existing) if *existing != key => anyhow::bail!(
563 "profile `{}` reuses `{path}` with a different signer configuration: \
564 two backends over one file would overwrite each other's state \
565 (give each profile its own paths, or make their [signer] sections identical)",
566 profile.name
567 ),
568 _ => {
569 owners.insert(path, key.clone());
570 }
571 }
572 }
573 }
574
575 // Which profiles each distinct configuration serves — what a relaying
576 // backend needs to know to recover the right orders and only those.
577 let mut served: HashMap<String, Vec<String>> = HashMap::new();
578 for profile in profiles {
579 served
580 .entry(key_of(&profile.sections.signer))
581 .or_default()
582 .push(profile.name.clone());
583 }
584
585 // Gathered once, before anything is built: a backend rebuilt over the same
586 // files must find the live ledger, and the outgoing instances are still
587 // holding it at this point — which is the whole reason the handover is a
588 // shared `Arc` and not a copy.
589 let carried = previous.carried();
590
591 let mut set = SignerSet::default();
592 for profile in profiles {
593 let key = key_of(&profile.sections.signer);
594 let backend = match (set.by_identity.get(&key), previous.by_identity.get(&key)) {
595 (Some(backend), _) => backend.clone(),
596 (None, Some(backend)) => {
597 debug!(
598 event = "signer_backend_reused",
599 outcome = "success",
600 profile = %profile.name,
601 "the configuration did not move, so the running backend is carried \
602 whole rather than rebuilt"
603 );
604 let backend = backend.clone();
605 set.by_identity.insert(key, backend.clone());
606 backend
607 }
608 (None, None) => {
609 let backend = from_config(
610 &profile.sections.signer,
611 served.get(&key).cloned().unwrap_or_default(),
612 parts,
613 &carried,
614 )
615 .map_err(|error| anyhow::anyhow!("profile `{}`: {error}", profile.name))?;
616 set.by_identity.insert(key, backend.clone());
617 backend
618 }
619 };
620 set.by_profile.insert(profile.name.clone(), backend);
621 }
622 Ok(set)
623}
624
625/// The files a signer configuration owns — what two profiles must not share
626/// unless they share the whole configuration.
627fn signer_paths(cfg: &SignerConfig) -> Vec<String> {
628 match cfg.backend.as_str() {
629 "local_ca" => {
630 let mut paths = vec![
631 cfg.local_ca.cert_path.clone(),
632 cfg.local_ca.key_path.clone(),
633 cfg.local_ca.crl_path.clone(),
634 ];
635 // A PKCS#11 key is shared state in exactly the way this check
636 // exists for: two `LocalCa`s over one token key would each keep
637 // their own revocation ledger and rewrite the CRL from it. Not a
638 // file, but the same hazard, so it goes in the same list under a
639 // pseudo-path that cannot collide with a real one.
640 if cfg.local_ca.key_source == "pkcs11" {
641 paths.push(format!(
642 "pkcs11:{}#{}#{}#{}",
643 cfg.local_ca.pkcs11.module_path,
644 cfg.local_ca.pkcs11.token_label,
645 cfg.local_ca.pkcs11.key_label,
646 cfg.local_ca.pkcs11.key_id,
647 ));
648 }
649 paths
650 }
651 "relay" => vec![cfg.relay.account_key_path.clone()],
652 _ => Vec::new(),
653 }
654}
655
656#[cfg(test)]
657mod tests {
658 use super::*;
659
660 /// The shared resolver `Profile::build_all` supplies at startup. These
661 /// tests reach loopback by IP literal, which `dns::connect` short-circuits.
662 fn test_resolver() -> std::sync::Arc<dyn crate::dns::Resolver> {
663 std::sync::Arc::new(crate::dns::HickoryResolver::from_system_uncached().unwrap())
664 }
665 use crate::config::LocalCaConfig;
666
667 /// A `SignerConfig` writing its CA material into a throwaway directory, so
668 /// the `local_ca` arm can run without touching the repository's `ca.pem`.
669 fn config(backend: &str) -> (SignerConfig, crate::testutil::TempDir) {
670 let dir = crate::testutil::TempDir::new("signer");
671 let cfg = SignerConfig {
672 backend: backend.to_string(),
673 local_ca: LocalCaConfig {
674 cert_path: dir.join("ca.pem").to_string_lossy().into_owned(),
675 key_path: dir.join("ca.key").to_string_lossy().into_owned(),
676 crl_path: dir.join("ca.crl").to_string_lossy().into_owned(),
677 ..LocalCaConfig::default()
678 },
679 ..SignerConfig::default()
680 };
681 (cfg, dir)
682 }
683
684 async fn database() -> Arc<Database> {
685 Arc::new(Database::connect_in_memory().await.unwrap())
686 }
687
688 /// The dependencies a backend is built from, none of which these tests are
689 /// about. `egress` carries a fixed identity, so a case that wants to prove
690 /// `[dns]`/`[proxy]` reach the identity key overrides it deliberately.
691 async fn parts() -> SignerParts {
692 crate::testutil::signer_parts(database().await, test_resolver())
693 }
694
695 /// `parts()` with a different egress identity — what a reload that changed
696 /// `dns.resolver` or `[proxy]` hands `build_backends`.
697 async fn parts_with_egress(identity: &str) -> SignerParts {
698 let mut parts = parts().await;
699 parts.egress = Arc::new(crate::Egress {
700 resolver: test_resolver(),
701 proxies: crate::testutil::no_proxies(),
702 identity: identity.to_string(),
703 });
704 parts
705 }
706
707 fn profile(name: &str, signer: SignerConfig) -> crate::config::ProfileConfig {
708 crate::config::ProfileConfig {
709 name: name.to_string(),
710 sections: crate::config::ProfileSections {
711 signer,
712 ..crate::config::ProfileSections::default()
713 },
714 }
715 }
716
717 /// A configuration that did not move is **not rebuilt**: the reload gets the
718 /// very same instance back.
719 ///
720 /// The ordinary case, and the one that matters most for cost — most reloads
721 /// touch `[filter]` or `[notify]` and leave every signer alone, and
722 /// rebuilding one there would re-read a CA key and, under
723 /// `key_source = "pkcs11"`, log in to a token again per `SIGHUP`.
724 ///
725 /// `Arc::ptr_eq` against the *previous* set is the only assertion that can
726 /// tell reuse from a rebuild that happened to adopt everything: both produce
727 /// a backend that behaves identically.
728 #[tokio::test]
729 async fn a_configuration_that_did_not_move_is_reused_rather_than_rebuilt() {
730 let (cfg, _dir) = config("local_ca");
731 let profiles = vec![profile("le", cfg)];
732
733 let parts = parts().await;
734 let first = build_backends(&profiles, &parts, &SignerSet::default()).unwrap();
735 let second = build_backends(&profiles, &parts, &first).unwrap();
736
737 assert!(
738 Arc::ptr_eq(first.get("le").unwrap(), second.get("le").unwrap()),
739 "an unchanged `[signer]` must hand back the running instance"
740 );
741 }
742
743 /// A configuration that *did* move is rebuilt — and the new instance shares
744 /// the old one's revocation ledger.
745 ///
746 /// The whole point of [`CarriedState`]. Proven through the CRL rather than
747 /// by inspecting the ledger: a revocation recorded on the outgoing backend
748 /// is visible in the incoming one's CRL, which is what an operator would
749 /// notice if it were not.
750 #[tokio::test]
751 async fn an_edited_configuration_is_rebuilt_over_the_running_ledger() {
752 let (cfg, _dir) = config("local_ca");
753 let parts = parts().await;
754 let running =
755 build_backends(&[profile("le", cfg.clone())], &parts, &SignerSet::default()).unwrap();
756
757 // Something to lose: a certificate issued and revoked by the instance
758 // that is about to be replaced.
759 let outgoing = running.get("le").unwrap().clone();
760 let key_pair = rcgen::KeyPair::generate().unwrap();
761 let params = rcgen::CertificateParams::new(vec!["example.com".to_string()]).unwrap();
762 let csr = params.serialize_request(&key_pair).unwrap();
763 let chain = match outgoing
764 .issue(
765 "ord-1",
766 csr.der(),
767 &[Identifier::dns("example.com")],
768 RequestedValidity::default(),
769 )
770 .await
771 .unwrap()
772 {
773 IssueOutcome::Issued(chain) => chain,
774 IssueOutcome::Processing => panic!("local_ca issues synchronously"),
775 };
776 let leaf = crate::cert::leaf_der_from_chain(&chain).unwrap();
777 outgoing.revoke(&leaf, Some(1)).await.unwrap();
778 let before = outgoing.crl_der().await.expect("a local CA has a CRL");
779
780 // The operator edits one key of `[signer]` and signals.
781 let mut edited = cfg;
782 edited.local_ca.leaf_validity_days = 30;
783 let reloaded = build_backends(&[profile("le", edited)], &parts, &running).unwrap();
784
785 let incoming = reloaded.get("le").unwrap();
786 assert!(
787 !Arc::ptr_eq(&outgoing, incoming),
788 "an edited `[signer]` must really be rebuilt, or the edit did nothing"
789 );
790 assert_eq!(
791 incoming.crl_der().await.expect("a local CA has a CRL"),
792 before,
793 "the rebuilt CA must serve the same CRL, ledger and all",
794 );
795
796 // And the sharing is live in both directions, which is what closes the
797 // window between building the replacement and publishing it.
798 let second = params.serialize_request(&key_pair).unwrap();
799 let chain = match outgoing
800 .issue(
801 "ord-2",
802 second.der(),
803 &[Identifier::dns("example.com")],
804 RequestedValidity::default(),
805 )
806 .await
807 .unwrap()
808 {
809 IssueOutcome::Issued(chain) => chain,
810 IssueOutcome::Processing => unreachable!(),
811 };
812 let leaf = crate::cert::leaf_der_from_chain(&chain).unwrap();
813 outgoing.revoke(&leaf, None).await.unwrap();
814 assert_ne!(
815 incoming.crl_der().await.unwrap(),
816 before,
817 "a revocation landing on the outgoing instance mid-reload must reach \
818 the incoming one — that is the case reading the sidecar back cannot cover",
819 );
820 }
821
822 /// `[dns]` and `[proxy]` are not `[signer]` keys, but a change to either
823 /// still rebuilds every backend.
824 ///
825 /// This is the whole reason those two keys could come off `reload::FROZEN`.
826 /// A backend caches the resolver and proxy policy it was built with, so
827 /// reuse keyed on `[signer]` alone would leave a `dns.resolver` edit
828 /// applying to every subsystem *except* the signers, silently.
829 #[tokio::test]
830 async fn a_changed_egress_rebuilds_a_backend_whose_signer_section_did_not_move() {
831 let (cfg, _dir) = config("local_ca");
832 let profiles = vec![profile("le", cfg)];
833
834 let first = build_backends(
835 &profiles,
836 &parts_with_egress("before").await,
837 &SignerSet::default(),
838 )
839 .unwrap();
840 let second = build_backends(&profiles, &parts_with_egress("after").await, &first).unwrap();
841
842 assert!(
843 !Arc::ptr_eq(first.get("le").unwrap(), second.get("le").unwrap()),
844 "a moved `[dns]`/`[proxy]` must reach the signers, which cache it",
845 );
846 }
847
848 /// A profile mounted by a reload gets a backend; one unmounted leaves its
849 /// backend behind, and it is not carried into the next generation.
850 ///
851 /// Between them these are "mount an endpoint without a restart", which was
852 /// the visible half of the whole freeze.
853 #[tokio::test]
854 async fn mounting_and_unmounting_a_profile_adds_and_drops_its_backend() {
855 let (first_cfg, _first_dir) = config("local_ca");
856 let (second_cfg, _second_dir) = config("local_ca");
857
858 let parts = parts().await;
859 let one = build_backends(
860 &[profile("le", first_cfg.clone())],
861 &parts,
862 &SignerSet::default(),
863 )
864 .unwrap();
865 assert_eq!(one.len(), 1);
866
867 let two = build_backends(
868 &[
869 profile("le", first_cfg),
870 profile("staging", second_cfg.clone()),
871 ],
872 &parts,
873 &one,
874 )
875 .unwrap();
876 assert_eq!(two.len(), 2, "the new endpoint got a backend of its own");
877 assert!(
878 Arc::ptr_eq(one.get("le").unwrap(), two.get("le").unwrap()),
879 "and the endpoint that was already running kept its instance"
880 );
881
882 let back_to_one = build_backends(&[profile("staging", second_cfg)], &parts, &two).unwrap();
883 assert_eq!(back_to_one.len(), 1);
884 assert!(back_to_one.get("le").is_none(), "the endpoint is unmounted");
885 assert!(
886 Arc::ptr_eq(
887 two.get("staging").unwrap(),
888 back_to_one.get("staging").unwrap()
889 ),
890 "the survivor is untouched by its neighbour going away"
891 );
892 }
893
894 /// A rebuild over *different* files starts from those files, never from the
895 /// state describing the old ones.
896 ///
897 /// The safety half of keying [`CarriedState`] on a resource: an operator
898 /// repointing a profile at a second CA must get that CA's revocation
899 /// history, not the first one's.
900 #[tokio::test]
901 async fn a_backend_rebuilt_over_different_files_adopts_nothing() {
902 let (cfg, _dir) = config("local_ca");
903 let parts = parts().await;
904 let running = build_backends(&[profile("le", cfg)], &parts, &SignerSet::default()).unwrap();
905
906 let outgoing = running.get("le").unwrap().clone();
907 let key_pair = rcgen::KeyPair::generate().unwrap();
908 let params = rcgen::CertificateParams::new(vec!["example.com".to_string()]).unwrap();
909 let csr = params.serialize_request(&key_pair).unwrap();
910 let IssueOutcome::Issued(chain) = outgoing
911 .issue(
912 "ord-1",
913 csr.der(),
914 &[Identifier::dns("example.com")],
915 RequestedValidity::default(),
916 )
917 .await
918 .unwrap()
919 else {
920 panic!("local_ca issues synchronously")
921 };
922 outgoing
923 .revoke(&crate::cert::leaf_der_from_chain(&chain).unwrap(), None)
924 .await
925 .unwrap();
926
927 let (elsewhere, _other_dir) = config("local_ca");
928 let reloaded = build_backends(&[profile("le", elsewhere)], &parts, &running).unwrap();
929
930 assert_ne!(
931 reloaded.get("le").unwrap().crl_der().await.unwrap(),
932 outgoing.crl_der().await.unwrap(),
933 "a different `crl_path` is a different CA and starts from its own sidecar",
934 );
935 }
936
937 /// Two endpoints configured identically share **one** backend instance.
938 ///
939 /// Not an optimization: a second `LocalCa` over the same files would keep
940 /// its own revocation ledger and rewrite the CRL from it, silently dropping
941 /// the first one's entries.
942 #[tokio::test]
943 async fn identical_signer_configuration_yields_one_shared_backend() {
944 let (cfg, _dir) = config("local_ca");
945 let profiles = vec![profile("a", cfg.clone()), profile("b", cfg)];
946
947 let backends = build_backends(&profiles, &parts().await, &SignerSet::default()).unwrap();
948 assert_eq!(backends.len(), 1);
949 assert!(
950 Arc::ptr_eq(backends.get("a").unwrap(), backends.get("b").unwrap()),
951 "one configuration must mean one instance"
952 );
953 }
954
955 #[tokio::test]
956 async fn differing_signer_configuration_yields_separate_backends() {
957 let (first, dir_a) = config("local_ca");
958 let (second, dir_b) = config("local_ca");
959 let profiles = vec![profile("a", first), profile("b", second)];
960
961 let backends = build_backends(&profiles, &parts().await, &SignerSet::default()).unwrap();
962 assert!(
963 !Arc::ptr_eq(backends.get("a").unwrap(), backends.get("b").unwrap()),
964 "different CA material must mean different CAs"
965 );
966
967 std::fs::remove_dir_all(dir_a).ok();
968 std::fs::remove_dir_all(dir_b).ok();
969 }
970
971 /// Sharing files while disagreeing about anything else is refused outright:
972 /// the two instances would overwrite each other's state, and the failure
973 /// would only show up as a mysteriously short CRL much later.
974 #[tokio::test]
975 async fn sharing_ca_files_with_a_different_configuration_is_a_startup_error() {
976 let (first, _dir) = config("local_ca");
977 let mut second = first.clone();
978 second.local_ca.leaf_validity_days = 7;
979
980 let profiles = vec![profile("a", first), profile("b", second)];
981 let error = match build_backends(&profiles, &parts().await, &SignerSet::default()) {
982 Err(error) => error.to_string(),
983 Ok(_) => panic!("two backends over one key file must not both be built"),
984 };
985 assert!(error.contains("different signer configuration"), "{error}");
986 assert!(
987 error.contains("ca.key") || error.contains("ca.pem"),
988 "{error}"
989 );
990 }
991
992 #[tokio::test]
993 async fn a_backend_failure_names_the_profile_it_came_from() {
994 let profiles = vec![profile(
995 "le",
996 SignerConfig {
997 backend: "nope".to_string(),
998 ..SignerConfig::default()
999 },
1000 )];
1001
1002 let error = match build_backends(&profiles, &parts().await, &SignerSet::default()) {
1003 Err(error) => error.to_string(),
1004 Ok(_) => panic!("an unknown backend is a startup error"),
1005 };
1006 assert!(error.contains("profile `le`"), "{error}");
1007 }
1008
1009 #[tokio::test]
1010 async fn builds_the_local_ca_backend_and_it_can_issue() {
1011 let (cfg, _dir) = config("local_ca");
1012 let signer = from_config(
1013 &cfg,
1014 vec!["default".to_string()],
1015 &parts().await,
1016 &CarriedState::new(),
1017 )
1018 .expect("local_ca is a known backend");
1019
1020 // Reached through the trait object, which is how handlers see it.
1021 let key_pair = rcgen::KeyPair::generate().unwrap();
1022 let params = rcgen::CertificateParams::new(vec!["example.com".to_string()]).unwrap();
1023 let csr = params.serialize_request(&key_pair).unwrap();
1024 let outcome = signer
1025 .issue(
1026 "ord-1",
1027 csr.der(),
1028 &[Identifier::dns("example.com")],
1029 RequestedValidity::default(),
1030 )
1031 .await
1032 .unwrap();
1033 // A local CA answers synchronously; only a delegating backend defers.
1034 let chain = match outcome {
1035 IssueOutcome::Issued(chain) => chain,
1036 IssueOutcome::Processing => panic!("local_ca must issue synchronously"),
1037 };
1038 assert_eq!(chain.matches("-----BEGIN CERTIFICATE-----").count(), 2);
1039 }
1040
1041 #[tokio::test]
1042 async fn builds_the_custom_backend_and_it_can_issue() {
1043 let dir = crate::testutil::TempDir::new("signer");
1044 let script_path = dir.join("issue.sh");
1045 std::fs::write(
1046 &script_path,
1047 "#!/bin/sh\ncat > /dev/null\necho '-----BEGIN CERTIFICATE-----leaf-----END CERTIFICATE-----'\nexit 0\n",
1048 )
1049 .unwrap();
1050 #[cfg(unix)]
1051 {
1052 use std::os::unix::fs::PermissionsExt;
1053 std::fs::set_permissions(&script_path, std::fs::Permissions::from_mode(0o755)).unwrap();
1054 }
1055
1056 let cfg = SignerConfig {
1057 backend: "custom".to_string(),
1058 custom: crate::config::CustomSignerConfig {
1059 script_path: script_path.to_string_lossy().into_owned(),
1060 ..Default::default()
1061 },
1062 ..SignerConfig::default()
1063 };
1064 let signer = from_config(
1065 &cfg,
1066 vec!["default".to_string()],
1067 &parts().await,
1068 &CarriedState::new(),
1069 )
1070 .expect("custom is a known backend");
1071
1072 let outcome = signer
1073 .issue(
1074 "ord-1",
1075 &[0x30, 0x00],
1076 &[Identifier::dns("example.com")],
1077 RequestedValidity::default(),
1078 )
1079 .await
1080 .unwrap();
1081 assert!(matches!(outcome, IssueOutcome::Issued(chain) if chain.contains("leaf")));
1082 }
1083
1084 /// `local_ca` has no upstream to ask, so it must keep the trait's default
1085 /// "no opinion" answer — that is what makes `get_renewal_info` fall back to
1086 /// its own local computation.
1087 #[tokio::test]
1088 async fn the_local_ca_backend_has_no_renewal_info_opinion() {
1089 let (cfg, _dir) = config("local_ca");
1090 let signer = from_config(
1091 &cfg,
1092 vec!["default".to_string()],
1093 &parts().await,
1094 &CarriedState::new(),
1095 )
1096 .unwrap();
1097 assert!(matches!(signer.renewal_info(&[0x30, 0x00]).await, Ok(None)));
1098 }
1099
1100 /// A typo in `signer.backend` stops the server rather than silently leaving
1101 /// it unable to issue.
1102 #[tokio::test]
1103 async fn an_unknown_backend_is_a_startup_error() {
1104 let (cfg, _dir) = config("hashicorp-vault");
1105 // `Arc<dyn SignerBackend>` is not `Debug`, so `unwrap_err` is unavailable.
1106 let error = match from_config(
1107 &cfg,
1108 vec!["default".to_string()],
1109 &parts().await,
1110 &CarriedState::new(),
1111 ) {
1112 Err(error) => error.to_string(),
1113 Ok(_) => panic!("an unknown backend must not build"),
1114 };
1115 assert!(
1116 error.contains("unknown signer backend") && error.contains("hashicorp-vault"),
1117 "{error}"
1118 );
1119 }
1120
1121 /// The one unknown backend that is a renamed key rather than a typo:
1122 /// `acme_proxy` was this backend's own name until it was renamed away from
1123 /// the host program's. The refusal has to carry the new name and the new
1124 /// environment prefix, since neither is guessable from "unknown signer
1125 /// backend" alone.
1126 #[tokio::test]
1127 async fn the_old_acme_proxy_backend_name_is_refused_by_its_new_one() {
1128 let (cfg, _dir) = config("acme_proxy");
1129 let error = match from_config(
1130 &cfg,
1131 vec!["default".to_string()],
1132 &parts().await,
1133 &CarriedState::new(),
1134 ) {
1135 Err(error) => error.to_string(),
1136 Ok(_) => panic!("the old backend name must not build"),
1137 };
1138 for expected in [
1139 "acme_proxy",
1140 "`relay`",
1141 "[signer.relay]",
1142 "ACME_PROXY_SIGNER__RELAY__",
1143 ] {
1144 assert!(error.contains(expected), "{expected} missing from: {error}");
1145 }
1146 }
1147
1148 /// Both variants render. `SignerError` is what a handler logs when
1149 /// issuance fails, so a variant with no message would leave nothing behind.
1150 #[test]
1151 fn signer_errors_render_their_kind() {
1152 assert_eq!(SignerError::BadCsr.to_string(), "Bad CSR");
1153 assert_eq!(
1154 SignerError::Internal("ca offline".to_string()).to_string(),
1155 "Internal signer error: ca offline"
1156 );
1157 }
1158}