Skip to main content

feather_reader/oauth/
runtime.rs

1//! The Rust OAuth client's long-lived state, assembled once at startup.
2//!
3//! Everything here is expensive to build, unsafe to rebuild per request, or
4//! both: the signing key anchors `client_id` and must be the same key across
5//! every request; the refresh locks are only useful if every caller shares one
6//! map; the DNS resolver holds the host's configuration.
7
8use std::path::Path;
9
10use anyhow::{Context as _, Result};
11
12use super::client_auth::AuthMethod;
13use super::crypto::Codec;
14use super::keys::SigningKey;
15use super::metadata::ClientConfig;
16use super::session::RefreshLocks;
17
18/// The `kid` for the client's signing key.
19///
20/// Published in the JWKS and echoed in every client assertion's header, so a
21/// server that has cached our JWKS looks the key up by this. It matches the
22/// sidecar's so a rollback finds the same key under the same name.
23pub const CLIENT_KID: &str = "featherreader-oauth-1";
24
25/// Everything the Rust OAuth client needs that outlives a request.
26pub struct OauthRuntime {
27    /// At-rest encryption for sessions and the signing key.
28    pub codec: Codec,
29    /// The validated client identity.
30    pub client: ClientConfig,
31    /// `client_id`, precomputed — it is derived, and recomputing it per request
32    /// invites a divergence between what we send and what we publish.
33    pub client_id: String,
34    /// The ES256 client key. `None` for the dev client, which is a public
35    /// client and publishes no JWKS — and for a confidential client whose
36    /// backend is not selected and which has no key file yet, since creating one
37    /// it will never use is key material at rest for nothing.
38    pub client_key: Option<SigningKey>,
39    /// How this client authenticates to the authorization server.
40    pub auth_method: AuthMethod,
41    /// Per-subject refresh serialization. One map process-wide, or the locking
42    /// does nothing.
43    pub locks: RefreshLocks,
44    /// The PLC directory for `did:plc` resolution.
45    pub plc_directory: String,
46    /// The system DNS resolver, for handle → DID.
47    pub resolver: hickory_resolver::TokioResolver,
48}
49
50impl OauthRuntime {
51    /// Build the runtime from configuration.
52    ///
53    /// **Dev is inferred from the public URL, exactly as the sidecar infers it**,
54    /// so the two agree on which client identity they present. A localhost
55    /// public URL means atproto's localhost development client: a public client
56    /// with no JWKS.
57    ///
58    /// The signing key is loaded (or created) only for a confidential client
59    /// **that is actually going to use it** — i.e. when the Rust backend is the
60    /// selected one. Two reasons, and the second is the one that bit:
61    ///
62    /// * a dev client is public, so a key there is never used and never
63    ///   published, and later reads as "the key exists, so it must be in play";
64    /// * the runtime is built on EVERY start so configuration errors surface
65    ///   early, including when the sidecar is serving. Creating the key as part
66    ///   of that validation meant a sidecar deployment wrote an ES256 private
67    ///   key it would never use — key material at rest, for nothing. In tests it
68    ///   also meant any `AppState` built with a production-like `public_url`
69    ///   dropped a private key into the working directory.
70    pub fn new(cfg: &crate::config::Config) -> Result<Self> {
71        let dev = is_loopback_url(&cfg.public_url);
72        let client = ClientConfig::new(&cfg.public_url, &cfg.oauth.scope, dev)
73            .context("building the OAuth client identity")?;
74        let client_id = super::metadata::client_id(&client);
75        let auth_method = AuthMethod::negotiate(dev);
76        let codec = Codec::new(cfg.oauth.encryption_key.as_deref())
77            .context("building the at-rest encryption codec")?;
78
79        // CREATION is gated on the backend; LOADING is not.
80        //
81        // Gating both was wrong, and dangerously so: `revoke_everywhere` signs a
82        // user out of BOTH backends on purpose, because after a flip their
83        // tokens can be in either store. With the sidecar selected and a key
84        // already on disk from a previous rust deployment, refusing to load it
85        // left `auth_method` as `private_key_jwt` with no key — so revocation
86        // bailed while the local row was deleted anyway, and the PDS-side
87        // refresh token stayed live forever with no local record left to retry
88        // from. An in-flight rust login completing after a flip died the same
89        // way.
90        //
91        // So: create a key only for the backend that will use it, but adopt one
92        // that already exists whatever the backend.
93        let creates_key = cfg.repo_backend == crate::metrics::Backend::Rust;
94        let key_exists = cfg.oauth.key_path.exists();
95        let client_key = match auth_method {
96            AuthMethod::PrivateKeyJwt if creates_key || key_exists => Some(
97                super::keys::load_or_create(Path::new(&cfg.oauth.key_path), &codec, CLIENT_KID)
98                    .with_context(|| {
99                        format!(
100                            "loading the OAuth signing key at {}",
101                            cfg.oauth.key_path.display()
102                        )
103                    })?,
104            ),
105            // Either a public client, or a confidential one whose backend is not
106            // selected AND which has no key on disk to adopt.
107            AuthMethod::PrivateKeyJwt | AuthMethod::None => None,
108        };
109
110        Ok(Self {
111            codec,
112            client,
113            client_id,
114            client_key,
115            auth_method,
116            locks: RefreshLocks::default(),
117            plc_directory: cfg.oauth.plc_directory.clone(),
118            resolver: super::resolve::resolver()?,
119        })
120    }
121}
122
123/// Whether a public URL names the local machine.
124///
125/// The sidecar infers its dev mode the same way (`SIDECAR_DEV` defaults to "the
126/// public URL is localhost"). Matching that inference matters because dev and
127/// production publish DIFFERENT `client_id`s — disagreeing would mean the two
128/// implementations present themselves as different clients from identical
129/// configuration.
130fn is_loopback_url(public_url: &str) -> bool {
131    let Ok(parsed) = url::Url::parse(public_url) else {
132        return false;
133    };
134    match parsed.host() {
135        Some(url::Host::Domain(host)) => host == "localhost" || host.ends_with(".localhost"),
136        Some(url::Host::Ipv4(ip)) => ip.is_loopback(),
137        Some(url::Host::Ipv6(ip)) => ip.is_loopback(),
138        None => false,
139    }
140}
141
142#[cfg(test)]
143mod tests {
144    use super::*;
145
146    /// The dev inference must match the sidecar's, because the two publish
147    /// different `client_id`s in the two modes. If they disagreed, identical
148    /// configuration would produce two different clients and the cutover would
149    /// invalidate every grant.
150    #[test]
151    fn dev_is_inferred_from_a_loopback_public_url() {
152        assert!(is_loopback_url("http://localhost:8080"));
153        assert!(is_loopback_url("http://127.0.0.1:8080"));
154        assert!(is_loopback_url("http://[::1]:8080"));
155        assert!(is_loopback_url("http://app.localhost:8080"));
156
157        assert!(!is_loopback_url("https://feather-reader.com"));
158        assert!(!is_loopback_url("https://localhost.evil.com"));
159    }
160
161    /// A host merely CONTAINING "localhost" is not loopback. `localhost.evil.com`
162    /// resolving as dev would hand the production deployment the dev client
163    /// identity — a public client with no client authentication at all.
164    #[test]
165    fn a_hostname_containing_localhost_is_not_loopback() {
166        assert!(!is_loopback_url("https://localhost.evil.com"));
167        assert!(!is_loopback_url("https://notlocalhost"));
168        assert!(!is_loopback_url("https://mylocalhost.net"));
169    }
170
171    /// An unparseable URL is NOT dev. Failing open here would drop client
172    /// authentication on a malformed production config.
173    #[test]
174    fn an_unparseable_url_is_not_treated_as_dev() {
175        assert!(!is_loopback_url("not a url"));
176        assert!(!is_loopback_url(""));
177    }
178}
179
180#[cfg(test)]
181mod key_creation_tests {
182    use super::*;
183
184    fn cfg(
185        backend: crate::metrics::Backend,
186        public_url: &str,
187        key_path: &std::path::Path,
188    ) -> crate::config::Config {
189        crate::config::Config {
190            repo_backend: backend,
191            public_url: public_url.to_string(),
192            oauth: crate::config::OauthConfig {
193                key_path: key_path.to_path_buf(),
194                ..crate::config::OauthConfig::default()
195            },
196            ..crate::config::Config::default()
197        }
198    }
199
200    fn temp_key_path(name: &str) -> std::path::PathBuf {
201        std::env::temp_dir().join(format!("fr-test-key-{name}-{}.json", std::process::id()))
202    }
203
204    /// **Building the runtime must not write a key the deployment will not use.**
205    ///
206    /// The runtime is constructed on every start, whatever the backend, so
207    /// configuration errors surface early. Creating the signing key as part of
208    /// that meant a SIDECAR deployment — the default — wrote an ES256 private
209    /// key it never touches: key material at rest for nothing.
210    ///
211    /// It surfaced as a unit test dropping a private key into the repo root,
212    /// because any `AppState` built with a production-like `public_url` did it.
213    #[test]
214    fn the_sidecar_backend_writes_no_signing_key() {
215        let path = temp_key_path("sidecar");
216        let _ = std::fs::remove_file(&path);
217
218        let runtime = OauthRuntime::new(&cfg(
219            crate::metrics::Backend::Sidecar,
220            "https://feather-reader.com",
221            &path,
222        ))
223        .expect("must build");
224
225        assert!(runtime.client_key.is_none());
226        assert!(
227            !path.exists(),
228            "the sidecar backend wrote a signing key it will never use"
229        );
230    }
231
232    /// The Rust backend on a production URL DOES need the key, and creates it.
233    #[test]
234    fn the_rust_backend_creates_its_signing_key() {
235        let path = temp_key_path("rust");
236        let _ = std::fs::remove_file(&path);
237
238        let runtime = OauthRuntime::new(&cfg(
239            crate::metrics::Backend::Rust,
240            "https://feather-reader.com",
241            &path,
242        ))
243        .expect("must build");
244
245        assert!(
246            runtime.client_key.is_some(),
247            "a confidential client needs its key"
248        );
249        assert!(path.exists(), "the key was not persisted");
250        let _ = std::fs::remove_file(&path);
251    }
252
253    /// **An EXISTING key is adopted even when the backend will not create one.**
254    ///
255    /// This is the flip-back case. `revoke_everywhere` signs a user out of both
256    /// backends deliberately, because after a flip their tokens can be in either
257    /// store. Refusing to load a key that is already on disk left the sidecar
258    /// deployment with `private_key_jwt` and no key, so revocation bailed while
259    /// the local row was deleted regardless — the PDS-side refresh token then
260    /// stayed live with nothing left to retry from.
261    #[test]
262    fn an_existing_key_is_adopted_on_the_sidecar_backend() {
263        let path = temp_key_path("adopt");
264        let _ = std::fs::remove_file(&path);
265
266        // A previous rust deployment left a key behind.
267        OauthRuntime::new(&cfg(
268            crate::metrics::Backend::Rust,
269            "https://feather-reader.com",
270            &path,
271        ))
272        .expect("must build");
273        assert!(path.exists(), "precondition: the key was created");
274
275        // Flip back to the sidecar. The key must still be loaded.
276        let runtime = OauthRuntime::new(&cfg(
277            crate::metrics::Backend::Sidecar,
278            "https://feather-reader.com",
279            &path,
280        ))
281        .expect("must build");
282        assert!(
283            runtime.client_key.is_some(),
284            "an existing key was ignored, so rust sessions could never be revoked"
285        );
286        let _ = std::fs::remove_file(&path);
287    }
288
289    /// A loopback deployment is a PUBLIC client: no key, on either backend.
290    #[test]
291    fn a_loopback_deployment_is_public_and_keyless() {
292        let path = temp_key_path("dev");
293        let _ = std::fs::remove_file(&path);
294
295        let runtime = OauthRuntime::new(&cfg(
296            crate::metrics::Backend::Rust,
297            "http://localhost:8080",
298            &path,
299        ))
300        .expect("must build");
301
302        assert_eq!(runtime.auth_method, AuthMethod::None);
303        assert!(runtime.client_key.is_none());
304        assert!(!path.exists(), "a public client wrote a signing key");
305    }
306}