Skip to main content

feather_reader/oauth/
runtime.rs

1//! The Rust OAuth client's long-lived state, assembled once at startup.
2//!
3//! Everything here is expensive to build, unsafe to rebuild per request, or
4//! both: the signing key anchors `client_id` and must be the same key across
5//! every request; the refresh locks are only useful if every caller shares one
6//! map; the DNS resolver holds the host's configuration.
7
8use std::path::Path;
9
10use anyhow::{Context as _, Result};
11
12use super::client_auth::AuthMethod;
13use super::crypto::Codec;
14use super::keys::SigningKey;
15use super::metadata::ClientConfig;
16use super::session::RefreshLocks;
17
18/// The `kid` for the client's signing key.
19///
20/// Published in the JWKS and echoed in every client assertion's header, so a
21/// server that has cached our JWKS looks the key up by this. It matches the
22/// sidecar's so a rollback finds the same key under the same name.
23pub const CLIENT_KID: &str = "featherreader-oauth-1";
24
25/// Everything the Rust OAuth client needs that outlives a request.
26pub struct OauthRuntime {
27    /// At-rest encryption for sessions and the signing key.
28    pub codec: Codec,
29    /// The validated client identity.
30    pub client: ClientConfig,
31    /// `client_id`, precomputed — it is derived, and recomputing it per request
32    /// invites a divergence between what we send and what we publish.
33    pub client_id: String,
34    /// The ES256 client key. `None` for the dev client, which is a public
35    /// client and publishes no JWKS — and for a confidential client whose
36    /// backend is not selected and which has no key file yet, since creating one
37    /// it will never use is key material at rest for nothing.
38    pub client_key: Option<SigningKey>,
39    /// How this client authenticates to the authorization server.
40    pub auth_method: AuthMethod,
41    /// Per-subject refresh serialization. One map process-wide, or the locking
42    /// does nothing.
43    pub locks: RefreshLocks,
44    /// The PLC directory for `did:plc` resolution.
45    pub plc_directory: String,
46    /// The system DNS resolver, for handle → DID.
47    pub resolver: hickory_resolver::TokioResolver,
48}
49
50impl OauthRuntime {
51    /// Build the runtime from configuration.
52    ///
53    /// **Dev is inferred from the public URL, exactly as the sidecar infers it**,
54    /// so the two agree on which client identity they present. A localhost
55    /// public URL means atproto's localhost development client: a public client
56    /// with no JWKS.
57    ///
58    /// The signing key is loaded (or created) only for a confidential client
59    /// **that is actually going to use it** — i.e. when the Rust backend is the
60    /// selected one. Two reasons, and the second is the one that bit:
61    ///
62    /// * a dev client is public, so a key there is never used and never
63    ///   published, and later reads as "the key exists, so it must be in play";
64    /// * the runtime is built on EVERY start so configuration errors surface
65    ///   early, including when the sidecar is serving. Creating the key as part
66    ///   of that validation meant a sidecar deployment wrote an ES256 private
67    ///   key it would never use — key material at rest, for nothing. In tests it
68    ///   also meant any `AppState` built with a production-like `public_url`
69    ///   dropped a private key into the working directory.
70    pub fn new(cfg: &crate::config::Config) -> Result<Self> {
71        let dev = is_loopback_url(&cfg.public_url);
72        let client = ClientConfig::new(&cfg.public_url, &cfg.oauth.scope, dev)
73            .context("building the OAuth client identity")?;
74        let client_id = super::metadata::client_id(&client);
75        let auth_method = AuthMethod::negotiate(dev);
76        let codec = Codec::new(cfg.oauth.encryption_key.as_deref())
77            .context("building the at-rest encryption codec")?;
78
79        // CREATION is gated on the backend; LOADING is not.
80        //
81        // Gating both was wrong, and dangerously so: `revoke_everywhere` signs a
82        // user out of BOTH backends on purpose, because after a flip their
83        // tokens can be in either store. With the sidecar selected and a key
84        // already on disk from a previous rust deployment, refusing to load it
85        // left `auth_method` as `private_key_jwt` with no key — so revocation
86        // bailed while the local row was deleted anyway, and the PDS-side
87        // refresh token stayed live forever with no local record left to retry
88        // from. An in-flight rust login completing after a flip died the same
89        // way.
90        //
91        // So: create a key only for the backend that will use it, but adopt one
92        // that already exists whatever the backend.
93        let creates_key = cfg.repo_backend == crate::metrics::Backend::Rust;
94        let key_exists = cfg.oauth.key_path.exists();
95        let client_key = match auth_method {
96            AuthMethod::PrivateKeyJwt if creates_key || key_exists => Some(
97                super::keys::load_or_create(Path::new(&cfg.oauth.key_path), &codec, CLIENT_KID)
98                    .with_context(|| {
99                        format!(
100                            "loading the OAuth signing key at {}",
101                            cfg.oauth.key_path.display()
102                        )
103                    })?,
104            ),
105            // Either a public client, or a confidential one whose backend is not
106            // selected AND which has no key on disk to adopt.
107            AuthMethod::PrivateKeyJwt | AuthMethod::None => None,
108        };
109
110        Ok(Self {
111            codec,
112            client,
113            client_id,
114            client_key,
115            auth_method,
116            locks: RefreshLocks::default(),
117            plc_directory: cfg.oauth.plc_directory.clone(),
118            resolver: super::resolve::resolver()?,
119        })
120    }
121}
122
123impl OauthRuntime {
124    /// [`OauthRuntime::new`], refusing instead of CREATING a missing signing
125    /// key. For the operator's `--revoke-all-sessions`.
126    ///
127    /// `new` creates the key when a confidential client has none — right for
128    /// the app's first boot, wrong here. Run from the wrong directory (the
129    /// default `FEATHERREADER_OAUTH_KEY_PATH` is relative) or with the wrong
130    /// environment, it minted a fresh key under the same `kid` while the live
131    /// app kept publishing the old JWKS. Every client assertion was then
132    /// rejected, every row deleted anyway, and the teardown proceeded over
133    /// live tokens. Revocation can only work with the key the PDSes know, so a
134    /// missing one is an error.
135    ///
136    /// App startup is unchanged: it still calls `new`.
137    pub fn without_creating_key(cfg: &crate::config::Config) -> Result<Self> {
138        let confidential =
139            AuthMethod::negotiate(is_loopback_url(&cfg.public_url)) == AuthMethod::PrivateKeyJwt;
140        if confidential && !cfg.oauth.key_path.exists() {
141            anyhow::bail!(
142                "the OAuth signing key {} does not exist, and this mode will not create one \
143                 (a new key is one no PDS can verify). Point FEATHERREADER_OAUTH_KEY_PATH at the \
144                 key the running app uses",
145                cfg.oauth.key_path.display()
146            );
147        }
148        Self::new(cfg)
149    }
150}
151
152/// Whether a public URL names the local machine.
153///
154/// The sidecar infers its dev mode the same way (`SIDECAR_DEV` defaults to "the
155/// public URL is localhost"). Matching that inference matters because dev and
156/// production publish DIFFERENT `client_id`s — disagreeing would mean the two
157/// implementations present themselves as different clients from identical
158/// configuration.
159fn is_loopback_url(public_url: &str) -> bool {
160    let Ok(parsed) = url::Url::parse(public_url) else {
161        return false;
162    };
163    match parsed.host() {
164        Some(url::Host::Domain(host)) => host == "localhost" || host.ends_with(".localhost"),
165        Some(url::Host::Ipv4(ip)) => ip.is_loopback(),
166        Some(url::Host::Ipv6(ip)) => ip.is_loopback(),
167        None => false,
168    }
169}
170
171#[cfg(test)]
172mod tests {
173    use super::*;
174
175    /// The dev inference must match the sidecar's, because the two publish
176    /// different `client_id`s in the two modes. If they disagreed, identical
177    /// configuration would produce two different clients and the cutover would
178    /// invalidate every grant.
179    #[test]
180    fn dev_is_inferred_from_a_loopback_public_url() {
181        assert!(is_loopback_url("http://localhost:8080"));
182        assert!(is_loopback_url("http://127.0.0.1:8080"));
183        assert!(is_loopback_url("http://[::1]:8080"));
184        assert!(is_loopback_url("http://app.localhost:8080"));
185
186        assert!(!is_loopback_url("https://feather-reader.com"));
187        assert!(!is_loopback_url("https://localhost.evil.com"));
188    }
189
190    /// A host merely CONTAINING "localhost" is not loopback. `localhost.evil.com`
191    /// resolving as dev would hand the production deployment the dev client
192    /// identity — a public client with no client authentication at all.
193    #[test]
194    fn a_hostname_containing_localhost_is_not_loopback() {
195        assert!(!is_loopback_url("https://localhost.evil.com"));
196        assert!(!is_loopback_url("https://notlocalhost"));
197        assert!(!is_loopback_url("https://mylocalhost.net"));
198    }
199
200    /// An unparseable URL is NOT dev. Failing open here would drop client
201    /// authentication on a malformed production config.
202    #[test]
203    fn an_unparseable_url_is_not_treated_as_dev() {
204        assert!(!is_loopback_url("not a url"));
205        assert!(!is_loopback_url(""));
206    }
207}
208
209#[cfg(test)]
210mod key_creation_tests {
211    use super::*;
212
213    fn cfg(
214        backend: crate::metrics::Backend,
215        public_url: &str,
216        key_path: &std::path::Path,
217    ) -> crate::config::Config {
218        crate::config::Config {
219            repo_backend: backend,
220            public_url: public_url.to_string(),
221            oauth: crate::config::OauthConfig {
222                key_path: key_path.to_path_buf(),
223                ..crate::config::OauthConfig::default()
224            },
225            ..crate::config::Config::default()
226        }
227    }
228
229    fn temp_key_path(name: &str) -> std::path::PathBuf {
230        std::env::temp_dir().join(format!("fr-test-key-{name}-{}.json", std::process::id()))
231    }
232
233    /// **Building the runtime must not write a key the deployment will not use.**
234    ///
235    /// The runtime is constructed on every start, whatever the backend, so
236    /// configuration errors surface early. Creating the signing key as part of
237    /// that meant a SIDECAR deployment — the default — wrote an ES256 private
238    /// key it never touches: key material at rest for nothing.
239    ///
240    /// It surfaced as a unit test dropping a private key into the repo root,
241    /// because any `AppState` built with a production-like `public_url` did it.
242    #[test]
243    fn the_sidecar_backend_writes_no_signing_key() {
244        let path = temp_key_path("sidecar");
245        let _ = std::fs::remove_file(&path);
246
247        let runtime = OauthRuntime::new(&cfg(
248            crate::metrics::Backend::Sidecar,
249            "https://feather-reader.com",
250            &path,
251        ))
252        .expect("must build");
253
254        assert!(runtime.client_key.is_none());
255        assert!(
256            !path.exists(),
257            "the sidecar backend wrote a signing key it will never use"
258        );
259    }
260
261    /// The Rust backend on a production URL DOES need the key, and creates it.
262    #[test]
263    fn the_rust_backend_creates_its_signing_key() {
264        let path = temp_key_path("rust");
265        let _ = std::fs::remove_file(&path);
266
267        let runtime = OauthRuntime::new(&cfg(
268            crate::metrics::Backend::Rust,
269            "https://feather-reader.com",
270            &path,
271        ))
272        .expect("must build");
273
274        assert!(
275            runtime.client_key.is_some(),
276            "a confidential client needs its key"
277        );
278        assert!(path.exists(), "the key was not persisted");
279        let _ = std::fs::remove_file(&path);
280    }
281
282    /// **An EXISTING key is adopted even when the backend will not create one.**
283    ///
284    /// This is the flip-back case. `revoke_everywhere` signs a user out of both
285    /// backends deliberately, because after a flip their tokens can be in either
286    /// store. Refusing to load a key that is already on disk left the sidecar
287    /// deployment with `private_key_jwt` and no key, so revocation bailed while
288    /// the local row was deleted regardless — the PDS-side refresh token then
289    /// stayed live with nothing left to retry from.
290    #[test]
291    fn an_existing_key_is_adopted_on_the_sidecar_backend() {
292        let path = temp_key_path("adopt");
293        let _ = std::fs::remove_file(&path);
294
295        // A previous rust deployment left a key behind.
296        OauthRuntime::new(&cfg(
297            crate::metrics::Backend::Rust,
298            "https://feather-reader.com",
299            &path,
300        ))
301        .expect("must build");
302        assert!(path.exists(), "precondition: the key was created");
303
304        // Flip back to the sidecar. The key must still be loaded.
305        let runtime = OauthRuntime::new(&cfg(
306            crate::metrics::Backend::Sidecar,
307            "https://feather-reader.com",
308            &path,
309        ))
310        .expect("must build");
311        assert!(
312            runtime.client_key.is_some(),
313            "an existing key was ignored, so rust sessions could never be revoked"
314        );
315        let _ = std::fs::remove_file(&path);
316    }
317
318    /// A loopback deployment is a PUBLIC client: no key, on either backend.
319    #[test]
320    fn a_loopback_deployment_is_public_and_keyless() {
321        let path = temp_key_path("dev");
322        let _ = std::fs::remove_file(&path);
323
324        let runtime = OauthRuntime::new(&cfg(
325            crate::metrics::Backend::Rust,
326            "http://localhost:8080",
327            &path,
328        ))
329        .expect("must build");
330
331        assert_eq!(runtime.auth_method, AuthMethod::None);
332        assert!(runtime.client_key.is_none());
333        assert!(!path.exists(), "a public client wrote a signing key");
334    }
335}