feather_reader/oauth/fetch.rs
1//! Fetching the JSON documents OAuth discovery depends on.
2//!
3//! Three rules apply to every document here, and each exists because the
4//! reference client enforces it:
5//!
6//! * **No redirects.** The mix-up defence compares a document's `issuer` against
7//! the URL it was fetched from; a `302` would make that comparison meaningless
8//! while still appearing to pass. See [`crate::net::guarded_get_no_redirect`].
9//! * **Status exactly 200.** Not "2xx", not "whatever parsed". A `204` or a `206`
10//! is not a metadata document.
11//! * **Content type must be JSON.** A server answering `text/html` is not
12//! serving the document we asked for, whatever the bytes happen to parse as.
13//!
14//! On top of the SSRF guard and the body cap those bring with them.
15
16use anyhow::{bail, Context as _, Result};
17use reqwest::Client;
18use serde_json::Value;
19
20use crate::net;
21
22/// Content types accepted for OAuth metadata documents.
23pub const JSON: &[&str] = &["application/json"];
24
25/// Content types accepted for DID documents. `did+ld+json` is what
26/// `plc.directory` actually serves.
27pub const DID_JSON: &[&str] = &[
28 "application/json",
29 "application/did+ld+json",
30 "application/did+json",
31];
32
33/// Validate a response's status and content type before its body is trusted.
34///
35/// Split out from the fetch so it is testable without a network: this is the
36/// part with the rules in it.
37fn check_meta(status: u16, content_type: Option<&str>, allowed: &[&str]) -> Result<()> {
38 if status != 200 {
39 bail!("expected status 200, got {status}");
40 }
41 let Some(raw) = content_type else {
42 bail!("response has no Content-Type; refusing to parse it as JSON");
43 };
44 // `application/json; charset=utf-8` — compare only the media type, and
45 // case-insensitively (RFC 9110 §8.3.1 makes it case-insensitive).
46 let media = raw
47 .split(';')
48 .next()
49 .unwrap_or_default()
50 .trim()
51 .to_ascii_lowercase();
52 if !allowed.iter().any(|a| *a == media) {
53 bail!("unexpected Content-Type {media:?}; expected one of {allowed:?}");
54 }
55 Ok(())
56}
57
58/// Fetch a JSON document, requiring 200 and an acceptable content type.
59pub async fn get_json(client: &Client, url: &str, allowed: &[&str]) -> Result<Value> {
60 get_json_optional(client, url, allowed)
61 .await?
62 .with_context(|| format!("document not found at {url}"))
63}
64
65/// As [`get_json`], but a `404` means **absent** rather than failed.
66///
67/// `/.well-known/oauth-protected-resource` legitimately does not exist on some
68/// hosts, and "absent" is a different answer from "the fetch went wrong".
69pub async fn get_json_optional(
70 client: &Client,
71 url: &str,
72 allowed: &[&str],
73) -> Result<Option<Value>> {
74 let resp = net::guarded_get_no_redirect(client, url, &[])
75 .await
76 .with_context(|| format!("fetching {url}"))?;
77
78 let status = resp.status().as_u16();
79 if status == 404 {
80 return Ok(None);
81 }
82 let content_type = resp
83 .headers()
84 .get(reqwest::header::CONTENT_TYPE)
85 .and_then(|v| v.to_str().ok())
86 .map(str::to_string);
87
88 check_meta(status, content_type.as_deref(), allowed)
89 .with_context(|| format!("fetching {url}"))?;
90
91 let body = net::read_capped(resp)
92 .await
93 .with_context(|| format!("reading {url}"))?;
94 // **The host here is chosen by whoever typed the handle, and this runs before
95 // anyone is authenticated.** `resolve::did_document` fetches a `did:web`
96 // document from the host named in the DID, and `discovery::discover` fetches
97 // `/.well-known/oauth-authorization-server` from the PDS URL that came with
98 // it. `read_capped` bounds the wire at 8 MB, which is the INPUT to the
99 // amplification, not a limit on it — 8 MB of `{"":0}` objects measured 789 MB
100 // of `Value`.
101 //
102 // The node bound rather than a length bound, because unlike an error peek
103 // these bodies are legitimately structural: a DID document carries services
104 // and verification methods, and authorization-server metadata is a few dozen
105 // arrays. Bounding structure permits any amount of text.
106 crate::atproto::refuse_a_structure_explosion(&body, &format!("the body of {url}"))?;
107 let value =
108 serde_json::from_slice(&body).with_context(|| format!("{url} is not valid JSON"))?;
109 Ok(Some(value))
110}
111
112#[cfg(test)]
113mod tests {
114
115 /// **The pre-auth path, and the one a stranger can reach.**
116 ///
117 /// `get_json` returns a `Value` from a host chosen by whoever typed the
118 /// handle: `resolve::did_document` fetches a `did:web` document from the host
119 /// named in the DID, and `discovery::discover` fetches
120 /// `/.well-known/oauth-authorization-server` from the PDS URL that came with
121 /// it. Neither needs a session. `read_capped` bounds the wire at 8 MiB, which
122 /// is the INPUT to the amplification rather than a limit on it — measured, 824
123 /// MB of `Value` for that body — so before the guard a stranger could spend
124 /// most of a 512 MB box by submitting a handle.
125 ///
126 /// Asserts the guard through the real `guarded_get`+`read_capped` path rather
127 /// than on the helper, because the point is the wiring.
128 #[tokio::test]
129 async fn a_did_document_that_is_a_structure_explosion_is_refused() {
130 let mut body = String::from(r#"{"id":"did:web:probe.test","service":["#);
131 for _ in 0..700_000 {
132 body.push_str("{},");
133 }
134 body.push_str("{}]}");
135 assert!(
136 crate::atproto::count_structural_chars(body.as_bytes())
137 > crate::atproto::MAX_LIST_STRUCTURAL_CHARS,
138 "the probe body is not over the cap, so this test proves nothing",
139 );
140 // `serve_bodies_in_sequence` answers `application/json`, which `DID_JSON`
141 // accepts — so the refusal under test is the body's, not `check_meta`'s.
142 // The host override is how a loopback server gets past the SSRF guard:
143 // without it this fails on the address rather than on the body, which is
144 // exactly the "failed for the wrong reason" trap the assertion below
145 // exists to catch.
146 let base = crate::net::tests::serve_bodies_in_sequence(vec![body.into_bytes()]).await;
147 let port: u16 = base
148 .trim_end_matches('/')
149 .rsplit(':')
150 .next()
151 .unwrap()
152 .parse()
153 .unwrap();
154 crate::net::test_host_override(
155 "did-explosion.test",
156 std::net::SocketAddr::from(([127, 0, 0, 1], port)),
157 );
158 let err = get_json(
159 &Client::new(),
160 &format!("http://did-explosion.test:{port}/did.json"),
161 DID_JSON,
162 )
163 .await
164 .expect_err("a structure explosion was parsed rather than refused");
165 let rendered = format!("{err:#}");
166 assert!(
167 rendered.contains("structural characters"),
168 "failed for the wrong reason: {rendered}"
169 );
170 }
171 use super::*;
172
173 #[test]
174 fn a_200_with_json_is_accepted() {
175 assert!(check_meta(200, Some("application/json"), JSON).is_ok());
176 }
177
178 /// Parameters and casing are both legal on a real content type.
179 #[test]
180 fn the_media_type_is_compared_without_parameters_or_case() {
181 for ct in [
182 "application/json; charset=utf-8",
183 "application/json;charset=UTF-8",
184 "APPLICATION/JSON",
185 " application/json ",
186 ] {
187 assert!(check_meta(200, Some(ct), JSON).is_ok(), "rejected {ct:?}");
188 }
189 }
190
191 /// A server answering HTML is not serving the document we asked for, even if
192 /// the bytes would happen to parse.
193 #[test]
194 fn a_non_json_content_type_is_rejected() {
195 for ct in [
196 "text/html",
197 "text/plain",
198 "application/xml",
199 "application/jsonp",
200 "application/json-seq",
201 ] {
202 assert!(check_meta(200, Some(ct), JSON).is_err(), "accepted {ct:?}");
203 }
204 }
205
206 #[test]
207 fn a_missing_content_type_is_rejected() {
208 assert!(check_meta(200, None, JSON).is_err());
209 }
210
211 /// Exactly 200 — a redirect reaching here at all would mean the no-redirect
212 /// guard failed, and 2xx-but-not-200 is not a metadata document.
213 #[test]
214 fn only_status_200_is_accepted() {
215 for status in [201u16, 204, 206, 301, 302, 400, 401, 403, 500] {
216 assert!(
217 check_meta(status, Some("application/json"), JSON).is_err(),
218 "accepted status {status}"
219 );
220 }
221 }
222
223 /// DID documents are served as `did+ld+json` by plc.directory; metadata
224 /// documents must NOT be.
225 #[test]
226 fn the_allowed_set_is_per_document_kind() {
227 assert!(check_meta(200, Some("application/did+ld+json"), DID_JSON).is_ok());
228 assert!(check_meta(200, Some("application/json"), DID_JSON).is_ok());
229 assert!(check_meta(200, Some("application/did+ld+json"), JSON).is_err());
230 }
231
232 /// These fetches must fail closed on an internal target like every other
233 /// outbound call.
234 ///
235 /// Asserts on the GUARD's error rather than `is_err()`: connecting to
236 /// `127.0.0.1` fails regardless, so an `is_err()`-only assertion would pass
237 /// with the SSRF guard removed entirely.
238 #[tokio::test]
239 async fn discovery_fetches_fail_closed_on_internal_targets() {
240 let client = Client::new();
241 for url in [
242 "http://127.0.0.1/.well-known/oauth-authorization-server",
243 "http://169.254.169.254/.well-known/oauth-protected-resource",
244 "http://10.1.2.3/did.json",
245 ] {
246 for rendered in [
247 format!(
248 "{:#}",
249 get_json(&client, url, JSON).await.expect_err("allowed")
250 ),
251 format!(
252 "{:#}",
253 get_json_optional(&client, url, JSON)
254 .await
255 .expect_err("allowed")
256 ),
257 ] {
258 assert!(
259 rendered.contains("forbidden (internal) address"),
260 "{url} failed for the wrong reason: {rendered}"
261 );
262 }
263 }
264 }
265}