feather_reader/oauth/identity.rs
1//! atproto identity resolution: handle → DID → DID document → PDS.
2//!
3//! The security property that matters here is **bidirectional verification**.
4//! The atproto spec makes it mandatory: *"If starting with a handle, it is
5//! critical (mandatory) to bidirectionally verify the handle by checking that
6//! the DID document claims the handle."* A handle is a DNS name someone else
7//! controls; without the back-check, whoever controls `victim.example` can point
8//! it at any DID they like.
9//!
10//! The comparison is deliberately narrow — equality against the **first**
11//! `at://` entry in `alsoKnownAs`, not membership in the array. A "is the handle
12//! anywhere in the list" check reintroduces the attack, because an attacker's
13//! own DID document can list the victim's handle as a secondary entry.
14
15use anyhow::{bail, Context as _, Result};
16use serde_json::Value;
17use std::collections::HashSet;
18
19/// TLDs the handle spec disallows outright.
20///
21/// `.local` and `.internal` are the ones that matter operationally: a handle
22/// ending in either is an attempt to steer resolution at the internal network,
23/// and rejecting it here means the SSRF guard is a second line of defence rather
24/// than the only one.
25const RESERVED_TLDS: [&str; 8] = [
26 "alt",
27 "arpa",
28 "example",
29 "internal",
30 "invalid",
31 "local",
32 "localhost",
33 "onion",
34];
35
36// The `at://` prefix an `alsoKnownAs` handle claim carries — the crate's one
37// spelling of it.
38use crate::atproto::AT_URI_PREFIX;
39
40/// Normalize and validate a handle: lowercase, then check it against the
41/// handle grammar and the reserved-TLD list.
42///
43/// Normalizing BEFORE any comparison is what makes the bidirectional check in
44/// [`verify_handle_claim`] sound; comparing raw input against a raw claim would
45/// make `Alice.example` and `alice.example` different handles.
46pub fn normalize_handle(input: &str) -> Result<String> {
47 let handle = input.trim().to_ascii_lowercase();
48 if handle.is_empty() || handle.len() > 253 {
49 bail!("handle {input:?} has an invalid length");
50 }
51 let labels: Vec<&str> = handle.split('.').collect();
52 if labels.len() < 2 {
53 bail!("handle {input:?} must have at least two segments");
54 }
55 for label in &labels {
56 if label.is_empty() || label.len() > 63 {
57 bail!("handle {input:?} has an empty or over-long segment");
58 }
59 if !label
60 .bytes()
61 .all(|b| b.is_ascii_alphanumeric() || b == b'-')
62 {
63 bail!("handle {input:?} has a segment with illegal characters");
64 }
65 if label.starts_with('-') || label.ends_with('-') {
66 bail!("handle {input:?} has a segment starting or ending with a hyphen");
67 }
68 }
69 // Spec: "The last segment (the 'top level domain') can not start with a
70 // numeric digit." Rejecting only ALL-numeric TLDs would let `alice.1com`
71 // and `alice.4chan` through.
72 let tld = labels[labels.len() - 1];
73 if tld.starts_with(|c: char| c.is_ascii_digit()) {
74 bail!("handle {input:?} has a TLD starting with a digit");
75 }
76 if RESERVED_TLDS.contains(&tld) {
77 bail!("handle {input:?} uses the reserved TLD .{tld}");
78 }
79 Ok(handle)
80}
81
82/// Whether a `did:web` method-specific id is a bare, canonical hostname.
83///
84/// This is the gate that stops URL construction from being steerable. Two
85/// shapes matter and neither is obvious:
86///
87/// * `good.com@evil.com` builds `https://good.com@evil.com/…`, whose ACTUAL
88/// host is `evil.com` — the plausible-looking part is demoted to userinfo.
89/// * `evil.com/x` is a straight path injection into the well-known path.
90///
91/// `:` is the did:web path separator (a port is spelled `%3A`), and atproto
92/// permits neither, so any of them disqualifies the DID.
93fn is_bare_did_web_host(host: &str) -> bool {
94 if host.is_empty() || host != host.to_ascii_lowercase() {
95 return false;
96 }
97 // Explicitly, rather than relying on the URL parser to object: these are
98 // the characters that change what the host IS.
99 if host.bytes().any(|b| {
100 matches!(b, b':' | b'/' | b'@' | b'%' | b'?' | b'#' | b'\\') || b.is_ascii_whitespace()
101 }) {
102 return false;
103 }
104 if !host.contains('.') {
105 return false; // a bare token is not a resolvable host
106 }
107 // The SAME reserved-TLD policy handles are held to. `normalize_handle`
108 // applies it and this did not, so `did:web:printer.local` and
109 // `did:web:pds.internal` were accepted and fetched. The SSRF guard does stop
110 // them — but the rule this module states for handles is that rejecting here
111 // makes the guard a SECOND line of defence rather than the only one, and for
112 // `did:web` it was the only one.
113 if let Some(tld) = host.rsplit('.').next() {
114 if RESERVED_TLDS.contains(&tld) {
115 return false;
116 }
117 }
118 // **An IP literal is refused by the numeric-final-label rule**, not by a
119 // parse. An earlier version parsed the host as `IpAddr` first and refused
120 // on success; that branch was dead — `IpAddr` accepts only the canonical
121 // dotted-quad, whose final label is numeric and is refused below anyway,
122 // and IPv6 cannot reach here at all because `:` is refused above. A test
123 // named for the IP rule kept passing with it deleted, which is how a
124 // check that does nothing gets mistaken for one that does.
125 //
126 // The URL parser the fetch path uses accepts far more than `IpAddr`:
127 // `127.1`, `0177.0.0.1` and `0x7f.0.0.1` all resolve to 127.0.0.1. A real
128 // TLD is never entirely numeric, so refusing a numeric final label catches
129 // `did:web:169.254.169.254` (cloud metadata) and every non-canonical
130 // spelling of it, without reimplementing the URL parser's arithmetic. It
131 // is the same rule `normalize_handle` applies to handles.
132 if let Some(tld) = host.rsplit('.').next() {
133 if tld.starts_with(|c: char| c.is_ascii_digit()) {
134 return false;
135 }
136 }
137 // A DNS name is at most 253 bytes, and each label at most 63; anything
138 // longer cannot resolve, and an unbounded one is only useful for making us
139 // construct absurd URLs. Both bounds, to match what `normalize_handle`
140 // enforces — the earlier version checked only the total.
141 if host.len() > 253 || host.split('.').any(|label| label.len() > 63) {
142 return false;
143 }
144 host.split('.').all(|label| {
145 !label.is_empty()
146 && label
147 .bytes()
148 .all(|b| b.is_ascii_alphanumeric() || b == b'-')
149 && !label.starts_with('-')
150 && !label.ends_with('-')
151 })
152}
153
154/// Whether `did` is a DID this client can resolve: `did:plc:` or `did:web:`.
155///
156/// This is the ONLY validation applied to a DID arriving from a TXT record or a
157/// well-known document, so anything it waves through becomes the account's
158/// identity and, for `did:web`, part of a URL.
159pub fn is_atproto_did(did: &str) -> bool {
160 if let Some(ident) = did.strip_prefix("did:plc:") {
161 // base32-SORTABLE: `[a-z2-7]`, 24 characters. Not `[a-z0-9]` — `0`,
162 // `1`, `8` and `9` are not in that alphabet.
163 return ident.len() == 24
164 && ident
165 .bytes()
166 .all(|b| b.is_ascii_lowercase() || (b'2'..=b'7').contains(&b));
167 }
168 if let Some(host) = did.strip_prefix("did:web:") {
169 return is_bare_did_web_host(host);
170 }
171 false
172}
173
174/// Join one TXT record's character-strings into its value.
175///
176/// A DNS TXT record is a SEQUENCE of strings, each at most 255 bytes, and the
177/// record's value is their concatenation. A DID that crosses that boundary
178/// arrives as two chunks; passing them along as two separate records produces
179/// two truncated fragments, neither a valid DID, and the handle fails to resolve
180/// with nothing to show for it.
181///
182/// `dig +short` performs this join itself, which is precisely why a shell-based
183/// spike will never surface the bug.
184pub fn join_txt_chunks(chunks: &[&[u8]]) -> String {
185 let joined: Vec<u8> = chunks.iter().flat_map(|c| c.iter().copied()).collect();
186 String::from_utf8_lossy(&joined).into_owned()
187}
188
189/// Extract the DID from a handle's `_atproto` TXT records.
190///
191/// `Ok(None)` means no record — a normal outcome that falls through to the
192/// well-known lookup. An error means the records are present but unusable, which
193/// must NOT fall through.
194///
195/// **Two differing records fail rather than picking one.** Spec: *"If multiple
196/// valid records with different DIDs are present, resolution should fail."*
197/// Taking the first would let anyone able to add a TXT record to a zone hijack a
198/// handle that already resolves.
199///
200/// The value after `did=` is deliberately NOT trimmed, matching the spec and the
201/// reference: a padded record is invalid rather than silently repaired.
202pub fn did_from_txt_records(records: &[String]) -> Result<Option<String>> {
203 let mut candidates: Vec<&str> = records
204 .iter()
205 .filter_map(|r| r.strip_prefix("did="))
206 .collect();
207 // **Deduplicate before counting.** The spec fails resolution when multiple
208 // records name DIFFERENT DIDs; this counted RECORDS. A zone that serves the
209 // same `did=` value twice — routine with split-horizon or multi-provider DNS
210 // — was an unrecoverable error, and because it is an `Err` rather than
211 // `Ok(None)` it does not even fall through to the well-known lookup. That
212 // account simply could not log in.
213 candidates.sort_unstable();
214 candidates.dedup();
215
216 match candidates.len() {
217 0 => Ok(None),
218 1 => {
219 let did = candidates[0];
220 if !is_atproto_did(did) {
221 bail!("_atproto TXT record does not contain a usable DID: {did:?}");
222 }
223 Ok(Some(did.to_string()))
224 }
225 n => bail!("{n} `did=` TXT records present; resolution must fail rather than choose"),
226 }
227}
228
229/// Extract the DID from a `/.well-known/atproto-did` body: first line, trimmed.
230pub fn did_from_well_known(body: &str) -> Result<String> {
231 let did = body.lines().next().unwrap_or_default().trim();
232 if !is_atproto_did(did) {
233 bail!("/.well-known/atproto-did did not contain a usable DID");
234 }
235 Ok(did.to_string())
236}
237
238/// Where a DID's document lives.
239///
240/// atproto restricts `did:web` to a bare hostname — **no path components and no
241/// port** (localhost excepted, which this client has no use for). A `did:web`
242/// carrying extra segments would otherwise let a DID name an arbitrary path on a
243/// host, which is a needlessly large surface for something resolved from user
244/// input.
245pub fn did_document_url(did: &str, plc_directory: &str) -> Result<String> {
246 if let Some(ident) = did.strip_prefix("did:plc:") {
247 if !is_atproto_did(did) {
248 bail!("{did:?} is not a well-formed did:plc identifier ({ident:?})");
249 }
250 return Ok(format!("{}/{did}", plc_directory.trim_end_matches('/')));
251 }
252 if let Some(host) = did.strip_prefix("did:web:") {
253 if !is_bare_did_web_host(host) {
254 bail!(
255 "atproto did:web must be a bare, canonical hostname with no path, \
256 port, credentials or escapes, got {host:?}"
257 );
258 }
259 return Ok(format!("https://{host}/.well-known/did.json"));
260 }
261 bail!("unsupported DID method in {did:?}; only did:plc and did:web are resolvable")
262}
263
264/// Validate a DID document against the DID that was requested.
265///
266/// The `id` check is the one that matters: without it, `plc.directory` — or
267/// whoever answers for a `did:web` host — can return a *different account's*
268/// document and every downstream decision is made about the wrong account.
269pub fn validate_did_document(document: &Value, expected_did: &str) -> Result<()> {
270 let id = document
271 .get("id")
272 .and_then(Value::as_str)
273 .context("DID document has no `id`")?;
274 if id != expected_did {
275 bail!("DID document id {id:?} does not match the requested DID {expected_did:?}");
276 }
277
278 // Duplicate service ids make "the first #atproto_pds" depend on array order.
279 //
280 // Ids are NORMALIZED to absolute form before comparing: `#atproto_pds` and
281 // `did:plc:xxx#atproto_pds` are two spellings of the SAME service, so a
282 // raw-string dedup sees two distinct ids and lets a document list the
283 // service twice with different endpoints.
284 if let Some(services) = document.get("service").and_then(Value::as_array) {
285 let mut seen = HashSet::new();
286 for service in services {
287 if let Some(sid) = service.get("id").and_then(Value::as_str) {
288 let absolute = if let Some(fragment) = sid.strip_prefix('#') {
289 format!("{expected_did}#{fragment}")
290 } else {
291 sid.to_string()
292 };
293 if !seen.insert(absolute.clone()) {
294 bail!("DID document has duplicate service id {absolute:?}");
295 }
296 }
297 }
298 }
299 Ok(())
300}
301
302/// The PDS endpoint from a DID document.
303///
304/// Three conditions must all hold, per the reference's
305/// `isAtprotoPersonalDataServerService`: the id matches `#atproto_pds` in
306/// relative or absolute form, the type is exactly `AtprotoPersonalDataServer`,
307/// and `serviceEndpoint` is a **string** that parses as a URL.
308pub fn pds_endpoint(document: &Value, did: &str) -> Result<String> {
309 let services = document
310 .get("service")
311 .and_then(Value::as_array)
312 .context("DID document has no `service` array")?;
313 let absolute = format!("{did}#atproto_pds");
314
315 for service in services {
316 let id = service
317 .get("id")
318 .and_then(Value::as_str)
319 .unwrap_or_default();
320 // EXACTLY the relative or THIS DID's absolute spelling. An `ends_with`
321 // test would let a document list a decoy service -- `urn:evil#atproto_pds`,
322 // or another DID's `#atproto_pds` -- ahead of the real one and win,
323 // choosing the server for the entire session.
324 let matches_id = id == "#atproto_pds" || id == absolute;
325 let matches_type =
326 service.get("type").and_then(Value::as_str) == Some("AtprotoPersonalDataServer");
327 if !matches_id || !matches_type {
328 continue;
329 }
330 let endpoint = service
331 .get("serviceEndpoint")
332 .and_then(Value::as_str)
333 .context("#atproto_pds serviceEndpoint is not a string")?;
334 let parsed = url::Url::parse(endpoint)
335 .with_context(|| format!("#atproto_pds serviceEndpoint {endpoint:?} is not a URL"))?;
336 if parsed.scheme() != "https" {
337 // The `http`-on-loopback carve-out that used to live here was
338 // unreachable: every fetch against this endpoint goes through the
339 // SSRF guard, which rejects all loopback addresses unconditionally
340 // with no dev flag. It could never serve the local-dev case it
341 // named, and only widened what a hostile DID document could get
342 // past this function.
343 bail!("#atproto_pds serviceEndpoint must be https, got {endpoint:?}");
344 }
345 // Reject userinfo for the same reason `did:web` hosts reject `@`: the
346 // plausible-looking part becomes credentials and the REAL host is
347 // whatever follows. Every security decision downstream re-derives the
348 // host (`origin_of` and `htu` both strip userinfo), so this is not a
349 // trust bypass — but `aud` is what appears in logs and what reqwest
350 // would turn into a `Basic` credential on every XRPC call.
351 if !parsed.username().is_empty() || parsed.password().is_some() {
352 bail!("#atproto_pds serviceEndpoint must not carry credentials, got {endpoint:?}");
353 }
354 return Ok(endpoint.to_string());
355 }
356 bail!("DID document declares no #atproto_pds service for {did}")
357}
358
359/// The handle a DID document claims, normalized.
360///
361/// **The first `at://` entry only.** That entry is the document's claim; later
362/// entries are not. A membership test over the whole array would let an
363/// attacker's document list a victim's handle as a secondary entry and pass
364/// verification for it.
365pub fn declared_handle(document: &Value) -> Option<String> {
366 document
367 .get("alsoKnownAs")?
368 .as_array()?
369 .iter()
370 .filter_map(Value::as_str)
371 .find_map(|entry| entry.strip_prefix(AT_URI_PREFIX))
372 .and_then(|handle| normalize_handle(handle).ok())
373}
374
375/// Bidirectional verification: does this DID document claim `handle`?
376///
377/// Mandatory when a login starts from a handle. A handle is a DNS name under
378/// someone else's control, so without the back-check whoever controls
379/// `victim.example` can point it at any DID at all.
380pub fn verify_handle_claim(document: &Value, handle: &str) -> Result<()> {
381 let wanted = normalize_handle(handle)?;
382 let claimed = declared_handle(document)
383 .context("DID document claims no handle; cannot verify bidirectionally")?;
384 if claimed != wanted {
385 bail!("DID document claims handle {claimed:?}, not {wanted:?}");
386 }
387 Ok(())
388}
389
390#[cfg(test)]
391mod tests {
392 use super::*;
393 use serde_json::json;
394
395 const DID: &str = "did:plc:ewvi7nxzyoun6zhxrhs64oiz";
396
397 // ── handle normalization ─────────────────────────────────────────────────
398
399 #[test]
400 fn handles_are_lowercased() {
401 assert_eq!(
402 normalize_handle("Alice.BSky.Social").unwrap(),
403 "alice.bsky.social"
404 );
405 assert_eq!(
406 normalize_handle(" bob.example.com ").unwrap(),
407 "bob.example.com"
408 );
409 }
410
411 /// **SSRF-adjacent.** `.local` / `.internal` handles are an attempt to steer
412 /// resolution at the internal network dressed up as a login. The spec
413 /// disallows these TLDs outright.
414 #[test]
415 fn reserved_tlds_are_rejected() {
416 for handle in [
417 "alice.local",
418 "alice.localhost",
419 "alice.internal",
420 "alice.arpa",
421 "alice.invalid",
422 "alice.example",
423 "alice.alt",
424 "alice.onion",
425 "deep.sub.local",
426 ] {
427 assert!(normalize_handle(handle).is_err(), "accepted {handle}");
428 }
429 }
430
431 #[test]
432 fn malformed_handles_are_rejected() {
433 for handle in [
434 "",
435 "alice", // no TLD
436 "alice.", // trailing dot
437 ".alice.com", // empty first label
438 "alice..com", // empty middle label
439 "-alice.com", // label starts with hyphen
440 "alice-.com", // label ends with hyphen
441 "alice.com-",
442 "alice_bob.com", // underscore not allowed
443 "alice.123", // all-numeric TLD
444 // Spec: "The last segment (the 'top level domain') can not start
445 // with a numeric digit." Rejecting only ALL-numeric TLDs let these
446 // three through.
447 "alice.1com",
448 "alice.4chan",
449 "alice.0x",
450 "al ice.com",
451 "alice.com/path",
452 "alice.com:443",
453 "https://alice.com",
454 ] {
455 assert!(normalize_handle(handle).is_err(), "accepted {handle:?}");
456 }
457 }
458
459 #[test]
460 fn ordinary_handles_are_accepted() {
461 for handle in [
462 "alice.bsky.social",
463 "a.co",
464 "xn--80akhbyknj4f.com",
465 "very-long-label-with-hyphens.example.org",
466 ] {
467 assert!(normalize_handle(handle).is_ok(), "rejected {handle}");
468 }
469 }
470
471 // ── DNS TXT resolution ───────────────────────────────────────────────────
472
473 /// **A DNS TXT record is a sequence of character-strings**, each capped at
474 /// 255 bytes, and the record's value is their CONCATENATION. A DID that
475 /// crosses that boundary arrives as two chunks; treating them as separate
476 /// records yields two truncated fragments, neither a valid DID, and the
477 /// handle fails to resolve for no visible reason.
478 ///
479 /// `dig +short` hides this by joining for you, which is exactly why the
480 /// spike did not catch it.
481 #[test]
482 fn txt_chunks_are_joined_into_one_record_value() {
483 let long = format!("did={DID}");
484 let (head, tail) = long.split_at(20);
485 assert_eq!(
486 join_txt_chunks(&[head.as_bytes(), tail.as_bytes()]),
487 long,
488 "chunks were not concatenated"
489 );
490 assert_eq!(join_txt_chunks(&[long.as_bytes()]), long);
491 assert_eq!(join_txt_chunks(&[]), "");
492 }
493
494 /// Joined chunks must then resolve exactly as a single-chunk record would.
495 #[test]
496 fn a_did_split_across_txt_chunks_still_resolves() {
497 let long = format!("did={DID}");
498 let (head, tail) = long.split_at(20);
499 let joined = join_txt_chunks(&[head.as_bytes(), tail.as_bytes()]);
500 assert_eq!(
501 did_from_txt_records(&[joined]).unwrap().as_deref(),
502 Some(DID)
503 );
504 }
505
506 #[test]
507 fn a_single_did_record_resolves() {
508 let records = vec![format!("did={DID}")];
509 assert_eq!(
510 did_from_txt_records(&records).unwrap().as_deref(),
511 Some(DID)
512 );
513 }
514
515 #[test]
516 fn unrelated_txt_records_are_ignored() {
517 let records = vec![
518 "v=spf1 -all".to_string(),
519 format!("did={DID}"),
520 "google-site-verification=abc".to_string(),
521 ];
522 assert_eq!(
523 did_from_txt_records(&records).unwrap().as_deref(),
524 Some(DID)
525 );
526 }
527
528 /// Spec: *"If multiple valid records with different DIDs are present,
529 /// resolution should fail."* Picking the first would let anyone who can add
530 /// a TXT record to a zone hijack a handle that already has one.
531 #[test]
532 fn multiple_did_records_fail_rather_than_picking_one() {
533 let records = vec![
534 format!("did={DID}"),
535 "did=did:plc:aaaaaaaaaaaaaaaaaaaaaaaa".to_string(),
536 ];
537 assert!(did_from_txt_records(&records).is_err());
538 }
539
540 #[test]
541 fn no_did_record_is_absent_not_an_error() {
542 let records = vec!["v=spf1 -all".to_string()];
543 assert_eq!(did_from_txt_records(&records).unwrap(), None);
544 }
545
546 /// The value after `did=` is NOT trimmed — deliberately, to stay consistent
547 /// with the spec — so a padded record is invalid rather than silently fixed.
548 #[test]
549 fn a_padded_did_value_is_invalid() {
550 let records = vec![format!("did= {DID}")];
551 assert!(did_from_txt_records(&records).is_err());
552 }
553
554 #[test]
555 fn a_non_did_value_is_rejected() {
556 for value in ["did=notadid", "did=", "did=did:unknown:xyz"] {
557 assert!(
558 did_from_txt_records(&[value.to_string()]).is_err(),
559 "accepted {value}"
560 );
561 }
562 }
563
564 // ── .well-known/atproto-did ──────────────────────────────────────────────
565
566 #[test]
567 fn the_well_known_body_takes_the_first_line_trimmed() {
568 assert_eq!(did_from_well_known(&format!("{DID}\n")).unwrap(), DID);
569 assert_eq!(did_from_well_known(&format!(" {DID} ")).unwrap(), DID);
570 assert_eq!(
571 did_from_well_known(&format!("{DID}\nignored")).unwrap(),
572 DID
573 );
574 }
575
576 #[test]
577 fn a_well_known_body_that_is_not_a_did_is_rejected() {
578 for body in ["", "\n", "not a did", "<html>", "did:unknown:x"] {
579 assert!(did_from_well_known(body).is_err(), "accepted {body:?}");
580 }
581 }
582
583 // ── DID → document URL ───────────────────────────────────────────────────
584
585 #[test]
586 fn plc_dids_map_to_the_directory() {
587 assert_eq!(
588 did_document_url(DID, "https://plc.directory").unwrap(),
589 format!("https://plc.directory/{DID}")
590 );
591 }
592
593 #[test]
594 fn did_web_maps_to_the_hosts_well_known() {
595 assert_eq!(
596 did_document_url("did:web:example.com", "https://plc.directory").unwrap(),
597 "https://example.com/.well-known/did.json"
598 );
599 }
600
601 /// atproto restricts `did:web` to a bare hostname: no path components, no
602 /// port. (`:` is the path separator in a did:web method-specific id; a port
603 /// is spelled `%3A`.)
604 #[test]
605 fn did_web_with_a_path_or_port_is_rejected() {
606 for did in [
607 "did:web:example.com:path",
608 "did:web:example.com:8080",
609 "did:web:example.com%3A8080",
610 "did:web:example.com:path:to:doc",
611 ] {
612 assert!(
613 did_document_url(did, "https://plc.directory").is_err(),
614 "accepted {did}"
615 );
616 assert!(!is_atproto_did(did), "is_atproto_did accepted {did}");
617 }
618 }
619
620 /// **Host confusion.** `did:web:good.com@evil.com` builds
621 /// `https://good.com@evil.com/…`, whose ACTUAL host is `evil.com` — the
622 /// plausible-looking part is demoted to userinfo. A `/` is a straight path
623 /// injection. Neither may survive as far as URL construction.
624 #[test]
625 fn did_web_host_confusion_and_path_injection_are_rejected() {
626 for did in [
627 "did:web:good.com@evil.com",
628 "did:web:evil.com/x",
629 "did:web:evil.com/.well-known/did.json#",
630 "did:web:%00",
631 "did:web:%2e%2e",
632 "did:web:ex ample.com",
633 "did:web:",
634 "did:web:.",
635 "did:web:-example.com",
636 "did:web:Example.com", // must be canonical lowercase
637 ] {
638 assert!(!is_atproto_did(did), "is_atproto_did accepted {did:?}");
639 assert!(
640 did_document_url(did, "https://plc.directory").is_err(),
641 "built a URL for {did:?}"
642 );
643 }
644 }
645
646 /// `did:plc` identifiers are base32-SORTABLE: `[a-z2-7]`. `0`, `1`, `8` and
647 /// `9` are not in that alphabet.
648 #[test]
649 fn did_plc_uses_the_base32_sortable_alphabet() {
650 assert!(is_atproto_did(DID));
651 for did in [
652 "did:plc:aaaaaaaaaaaaaaaaaaaaaa01",
653 "did:plc:aaaaaaaaaaaaaaaaaaaaaa89",
654 "did:plc:AAAAAAAAAAAAAAAAAAAAAAAA",
655 "did:plc:tooshort",
656 "did:plc:aaaaaaaaaaaaaaaaaaaaaaaaa",
657 ] {
658 assert!(!is_atproto_did(did), "accepted {did}");
659 }
660 }
661
662 #[test]
663 fn unsupported_did_methods_are_rejected() {
664 for did in ["did:key:z6Mk", "did:example:123", "notadid", "", "did:"] {
665 assert!(
666 did_document_url(did, "https://plc.directory").is_err(),
667 "accepted {did}"
668 );
669 }
670 }
671
672 // ── DID document validation ──────────────────────────────────────────────
673
674 fn doc() -> serde_json::Value {
675 json!({
676 "id": DID,
677 "alsoKnownAs": ["at://alice.bsky.social"],
678 "service": [{
679 "id": "#atproto_pds",
680 "type": "AtprotoPersonalDataServer",
681 "serviceEndpoint": "https://pds.example.com"
682 }]
683 })
684 }
685
686 /// Without this, `plc.directory` — or whoever answers for a `did:web` host —
687 /// can hand back a different account's document entirely.
688 #[test]
689 fn the_document_id_must_match_the_did_requested() {
690 let mut d = doc();
691 d["id"] = json!("did:plc:someoneelse00000000000");
692 assert!(validate_did_document(&d, DID).is_err());
693 assert!(validate_did_document(&doc(), DID).is_ok());
694 }
695
696 #[test]
697 fn a_document_without_an_id_is_rejected() {
698 let mut d = doc();
699 d.as_object_mut().unwrap().remove("id");
700 assert!(validate_did_document(&d, DID).is_err());
701 }
702
703 /// With duplicate ids, "the first `#atproto_pds`" depends on array order,
704 /// which is not a property anyone should rely on.
705 #[test]
706 fn duplicate_service_ids_are_rejected() {
707 let mut d = doc();
708 d["service"] = json!([
709 {"id": "#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://a.example"},
710 {"id": "#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://b.example"}
711 ]);
712 assert!(validate_did_document(&d, DID).is_err());
713 }
714
715 // ── PDS endpoint extraction ──────────────────────────────────────────────
716
717 #[test]
718 fn the_pds_endpoint_is_extracted() {
719 assert_eq!(
720 pds_endpoint(&doc(), DID).unwrap(),
721 "https://pds.example.com"
722 );
723 }
724
725 /// The id may be relative (`#atproto_pds`) or absolute
726 /// (`did:plc:xxx#atproto_pds`) — but ONLY those two spellings.
727 #[test]
728 fn an_absolute_service_id_is_accepted() {
729 let mut d = doc();
730 d["service"][0]["id"] = json!(format!("{DID}#atproto_pds"));
731 assert_eq!(pds_endpoint(&d, DID).unwrap(), "https://pds.example.com");
732 }
733
734 /// **PDS steering.** Matching on "ends with `#atproto_pds`" lets a DID
735 /// document put a DECOY service first whose id merely has that suffix, and
736 /// win. The PDS is what discovery runs against and what every later XRPC
737 /// call targets, so this chooses the server for the whole session.
738 ///
739 /// The decoy must be FIRST here — with a single service there is nothing to
740 /// beat, which is how the original tests missed this.
741 #[test]
742 fn a_foreign_service_id_ending_in_atproto_pds_does_not_win() {
743 let mut d = doc();
744 d["service"] = json!([
745 {"id": "urn:evil#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://attacker.example"},
746 {"id": "#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://real-pds.example"}
747 ]);
748 assert_eq!(
749 pds_endpoint(&d, DID).unwrap(),
750 "https://real-pds.example",
751 "a decoy service id steered the PDS"
752 );
753
754 // And an id belonging to a DIFFERENT DID is not ours either.
755 d["service"] = json!([
756 {"id": "did:plc:aaaaaaaaaaaaaaaaaaaaaaaa#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://attacker.example"},
757 {"id": "#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://real-pds.example"}
758 ]);
759 assert_eq!(pds_endpoint(&d, DID).unwrap(), "https://real-pds.example");
760 }
761
762 /// The relative and absolute spellings are the SAME service, so listing both
763 /// is a duplicate — which the raw-string dedup did not see.
764 #[test]
765 fn the_relative_and_absolute_spellings_count_as_one_service() {
766 let mut d = doc();
767 d["service"] = json!([
768 {"id": format!("{DID}#atproto_pds"), "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://attacker.example"},
769 {"id": "#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://real-pds.example"}
770 ]);
771 assert!(
772 validate_did_document(&d, DID).is_err(),
773 "two spellings of the same service id were not seen as duplicates"
774 );
775 }
776
777 /// A DID document is attacker-controlled in the `did:web` case, and tokens
778 /// go to this host, so plaintext is refused — **including on loopback**.
779 ///
780 /// This previously carved out `http` on loopback "for a local dev PDS". That
781 /// carve-out was unreachable: every fetch against this endpoint goes through
782 /// the SSRF guard, which rejects all loopback addresses unconditionally with
783 /// no dev flag and no config bypass. It could never serve the case it named,
784 /// and only widened what a hostile document could get past this function.
785 #[test]
786 fn a_plaintext_http_pds_is_rejected_including_on_loopback() {
787 let mut d = doc();
788 for endpoint in [
789 "http://pds.attacker.example",
790 "http://10.0.0.5",
791 "http://localhost:2583",
792 "http://127.0.0.1:2583",
793 ] {
794 d["service"][0]["serviceEndpoint"] = json!(endpoint);
795 assert!(pds_endpoint(&d, DID).is_err(), "accepted {endpoint}");
796 }
797 }
798
799 /// **Credentials in a `serviceEndpoint` are refused.**
800 ///
801 /// `https://good.example@attacker.example` has real host `attacker.example`
802 /// — the exact demotion this module already rejects for `did:web` hosts. The
803 /// downstream security decisions re-derive the host either way, so this is
804 /// not a trust bypass; but this value is `aud`, it is what appears in logs,
805 /// and reqwest would turn the userinfo into a `Basic` credential on every
806 /// XRPC call to the PDS.
807 #[test]
808 fn a_pds_endpoint_carrying_credentials_is_refused() {
809 let mut d = doc();
810 for endpoint in [
811 "https://good.example@attacker.example",
812 "https://u:p@attacker.example/x",
813 ] {
814 d["service"][0]["serviceEndpoint"] = json!(endpoint);
815 assert!(pds_endpoint(&d, DID).is_err(), "accepted {endpoint}");
816 }
817 }
818
819 /// `did:web` hosts are held to the SAME policy as handles: no reserved TLD,
820 /// no IP literal. The SSRF guard blocks these at fetch time, but this module
821 /// states that rejecting here is what makes the guard a second line of
822 /// defence rather than the only one.
823 ///
824 /// Every IP case below is refused by the numeric-final-label rule — there
825 /// is no separate IP check, and there was a dead one until a mutation
826 /// showed this test green with it deleted. Deleting the label rule instead
827 /// fails every IP case here.
828 #[test]
829 fn did_web_hosts_obey_the_reserved_tld_and_ip_policy() {
830 for did in [
831 "did:web:169.254.169.254", // cloud metadata
832 "did:web:127.0.0.1",
833 "did:web:10.0.0.5",
834 "did:web:pds.internal",
835 "did:web:printer.local",
836 "did:web:something.localhost",
837 "did:web:site.onion",
838 ] {
839 assert!(!is_atproto_did(did), "accepted {did}");
840 }
841 // Non-canonical IP spellings resolve to loopback just as well, and
842 // `IpAddr::parse` accepts none of them — a numeric final label does.
843 for did in ["did:web:127.1", "did:web:0177.0.0.1", "did:web:0x7f.0.0.1"] {
844 assert!(!is_atproto_did(did), "accepted {did}");
845 }
846 // Length bounds: total and per-label, matching `normalize_handle`.
847 assert!(!is_atproto_did(&format!("did:web:{}.com", "a".repeat(300))));
848 assert!(
849 !is_atproto_did(&format!("did:web:{}.com", "a".repeat(64))),
850 "a 64-byte label exceeds the DNS limit"
851 );
852 assert!(is_atproto_did(&format!("did:web:{}.com", "a".repeat(63))));
853 // A digit-leading TLD is not a TLD.
854 assert!(!is_atproto_did("did:web:foo.1com"));
855 // Ordinary hosts still work.
856 assert!(is_atproto_did("did:web:pds.example.com"));
857 }
858
859 /// All three conditions must hold: id, type, and a parseable endpoint.
860 #[test]
861 fn a_service_failing_any_condition_is_not_the_pds() {
862 let cases = [
863 json!({"id": "#atproto_pds", "type": "SomethingElse", "serviceEndpoint": "https://a.example"}),
864 json!({"id": "#other", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://a.example"}),
865 json!({"id": "urn:evil#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://a.example"}),
866 json!({"id": "#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "not a url"}),
867 json!({"id": "#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": ["https://a.example"]}),
868 json!({"id": "#atproto_pds", "type": "AtprotoPersonalDataServer"}),
869 ];
870 for svc in cases {
871 let mut d = doc();
872 d["service"] = json!([svc.clone()]);
873 assert!(pds_endpoint(&d, DID).is_err(), "accepted {svc}");
874 }
875 }
876
877 #[test]
878 fn a_document_with_no_services_has_no_pds() {
879 let mut d = doc();
880 d["service"] = json!([]);
881 assert!(pds_endpoint(&d, DID).is_err());
882 d.as_object_mut().unwrap().remove("service");
883 assert!(pds_endpoint(&d, DID).is_err());
884 }
885
886 // ── bidirectional verification ───────────────────────────────────────────
887
888 #[test]
889 fn the_declared_handle_is_the_first_at_uri_entry_normalized() {
890 let mut d = doc();
891 d["alsoKnownAs"] = json!(["at://Alice.BSky.Social"]);
892 assert_eq!(declared_handle(&d).as_deref(), Some("alice.bsky.social"));
893 }
894
895 /// **The attack this closes.** An attacker's DID document can list the
896 /// victim's handle as a SECONDARY entry. Only the first `at://` entry is the
897 /// document's claim, so a membership test would accept the attacker's
898 /// document for the victim's handle.
899 #[test]
900 fn only_the_first_at_uri_entry_counts() {
901 // Note: not `.example` — that is a reserved TLD and would be rejected by
902 // `normalize_handle` before the ordering logic was ever reached.
903 let mut d = doc();
904 d["alsoKnownAs"] = json!(["at://attacker.com", "at://victim.com"]);
905 assert_eq!(declared_handle(&d).as_deref(), Some("attacker.com"));
906 assert!(verify_handle_claim(&d, "victim.com").is_err());
907 assert!(verify_handle_claim(&d, "attacker.com").is_ok());
908 }
909
910 /// Non-`at://` entries are skipped when looking for the first claim.
911 #[test]
912 fn non_at_uri_entries_are_skipped() {
913 let mut d = doc();
914 d["alsoKnownAs"] = json!(["https://alice.example", "at://alice.bsky.social"]);
915 assert_eq!(declared_handle(&d).as_deref(), Some("alice.bsky.social"));
916 }
917
918 #[test]
919 fn a_document_claiming_no_handle_fails_verification() {
920 let mut d = doc();
921 d["alsoKnownAs"] = json!([]);
922 assert!(verify_handle_claim(&d, "alice.bsky.social").is_err());
923 d.as_object_mut().unwrap().remove("alsoKnownAs");
924 assert!(verify_handle_claim(&d, "alice.bsky.social").is_err());
925 }
926
927 /// A malformed claim must not be usable as a wildcard.
928 #[test]
929 fn a_malformed_claimed_handle_fails_verification() {
930 for claim in ["at://", "at://not a handle", "at://alice.local"] {
931 let mut d = doc();
932 d["alsoKnownAs"] = json!([claim]);
933 assert!(
934 verify_handle_claim(&d, "alice.bsky.social").is_err(),
935 "accepted claim {claim}"
936 );
937 }
938 }
939
940 /// **A malformed FIRST claim must not fall through to the second.**
941 ///
942 /// The two neighbouring tests could not see this between them: this one's
943 /// sibling uses single-element arrays, so "reject" and "skip to the next"
944 /// look identical, and `only_the_first_at_uri_entry_counts` uses two VALID
945 /// entries, so nothing forces the first to be the one that fails.
946 ///
947 /// A mutation turning `find_map(strip).and_then(normalize)` into
948 /// `filter_map(strip).find_map(normalize)` — skip past an unusable claim —
949 /// passed the whole suite. Under it, an attacker document whose first entry
950 /// is junk verifies as whatever the SECOND entry says, which is precisely
951 /// what "only the first entry counts" exists to stop.
952 #[test]
953 fn a_malformed_first_claim_does_not_fall_through_to_the_second() {
954 for bad_first in ["at://", "at://not a handle", "at://alice.local"] {
955 let mut d = doc();
956 d["alsoKnownAs"] = json!([bad_first, "at://victim.com"]);
957 assert_eq!(
958 declared_handle(&d),
959 None,
960 "a malformed first claim ({bad_first}) was skipped and the second was taken"
961 );
962 assert!(
963 verify_handle_claim(&d, "victim.com").is_err(),
964 "({bad_first}) the second entry verified as the account's handle"
965 );
966 }
967 }
968
969 #[test]
970 fn verification_is_case_insensitive_on_both_sides() {
971 let mut d = doc();
972 d["alsoKnownAs"] = json!(["at://Alice.BSky.Social"]);
973 assert!(verify_handle_claim(&d, "ALICE.bsky.SOCIAL").is_ok());
974 }
975}