Skip to main content

feather_reader/oauth/
identity.rs

1//! atproto identity resolution: handle → DID → DID document → PDS.
2//!
3//! The security property that matters here is **bidirectional verification**.
4//! The atproto spec makes it mandatory: *"If starting with a handle, it is
5//! critical (mandatory) to bidirectionally verify the handle by checking that
6//! the DID document claims the handle."* A handle is a DNS name someone else
7//! controls; without the back-check, whoever controls `victim.example` can point
8//! it at any DID they like.
9//!
10//! The comparison is deliberately narrow — equality against the **first**
11//! `at://` entry in `alsoKnownAs`, not membership in the array. A "is the handle
12//! anywhere in the list" check reintroduces the attack, because an attacker's
13//! own DID document can list the victim's handle as a secondary entry.
14
15use anyhow::{bail, Context as _, Result};
16use serde_json::Value;
17use std::collections::HashSet;
18
19/// TLDs the handle spec disallows outright.
20///
21/// `.local` and `.internal` are the ones that matter operationally: a handle
22/// ending in either is an attempt to steer resolution at the internal network,
23/// and rejecting it here means the SSRF guard is a second line of defence rather
24/// than the only one.
25const RESERVED_TLDS: [&str; 8] = [
26    "alt",
27    "arpa",
28    "example",
29    "internal",
30    "invalid",
31    "local",
32    "localhost",
33    "onion",
34];
35
36// The `at://` prefix an `alsoKnownAs` handle claim carries — the crate's one
37// spelling of it.
38use crate::atproto::AT_URI_PREFIX;
39
40/// Normalize and validate a handle: lowercase, then check it against the
41/// handle grammar and the reserved-TLD list.
42///
43/// Normalizing BEFORE any comparison is what makes the bidirectional check in
44/// [`verify_handle_claim`] sound; comparing raw input against a raw claim would
45/// make `Alice.example` and `alice.example` different handles.
46pub fn normalize_handle(input: &str) -> Result<String> {
47    let handle = input.trim().to_ascii_lowercase();
48    if handle.is_empty() || handle.len() > 253 {
49        bail!("handle {input:?} has an invalid length");
50    }
51    let labels: Vec<&str> = handle.split('.').collect();
52    if labels.len() < 2 {
53        bail!("handle {input:?} must have at least two segments");
54    }
55    for label in &labels {
56        if label.is_empty() || label.len() > 63 {
57            bail!("handle {input:?} has an empty or over-long segment");
58        }
59        if !label
60            .bytes()
61            .all(|b| b.is_ascii_alphanumeric() || b == b'-')
62        {
63            bail!("handle {input:?} has a segment with illegal characters");
64        }
65        if label.starts_with('-') || label.ends_with('-') {
66            bail!("handle {input:?} has a segment starting or ending with a hyphen");
67        }
68    }
69    // Spec: "The last segment (the 'top level domain') can not start with a
70    // numeric digit." Rejecting only ALL-numeric TLDs would let `alice.1com`
71    // and `alice.4chan` through.
72    let tld = labels[labels.len() - 1];
73    if tld.starts_with(|c: char| c.is_ascii_digit()) {
74        bail!("handle {input:?} has a TLD starting with a digit");
75    }
76    if RESERVED_TLDS.contains(&tld) {
77        bail!("handle {input:?} uses the reserved TLD .{tld}");
78    }
79    Ok(handle)
80}
81
82/// Whether a `did:web` method-specific id is a bare, canonical hostname.
83///
84/// This is the gate that stops URL construction from being steerable. Two
85/// shapes matter and neither is obvious:
86///
87/// * `good.com@evil.com` builds `https://good.com@evil.com/…`, whose ACTUAL
88///   host is `evil.com` — the plausible-looking part is demoted to userinfo.
89/// * `evil.com/x` is a straight path injection into the well-known path.
90///
91/// `:` is the did:web path separator (a port is spelled `%3A`), and atproto
92/// permits neither, so any of them disqualifies the DID.
93fn is_bare_did_web_host(host: &str) -> bool {
94    if host.is_empty() || host != host.to_ascii_lowercase() {
95        return false;
96    }
97    // Explicitly, rather than relying on the URL parser to object: these are
98    // the characters that change what the host IS.
99    if host.bytes().any(|b| {
100        matches!(b, b':' | b'/' | b'@' | b'%' | b'?' | b'#' | b'\\') || b.is_ascii_whitespace()
101    }) {
102        return false;
103    }
104    if !host.contains('.') {
105        return false; // a bare token is not a resolvable host
106    }
107    // The SAME reserved-TLD policy handles are held to. `normalize_handle`
108    // applies it and this did not, so `did:web:printer.local` and
109    // `did:web:pds.internal` were accepted and fetched. The SSRF guard does stop
110    // them — but the rule this module states for handles is that rejecting here
111    // makes the guard a SECOND line of defence rather than the only one, and for
112    // `did:web` it was the only one.
113    if let Some(tld) = host.rsplit('.').next() {
114        if RESERVED_TLDS.contains(&tld) {
115            return false;
116        }
117    }
118    // **An IP literal is refused by the numeric-final-label rule**, not by a
119    // parse. An earlier version parsed the host as `IpAddr` first and refused
120    // on success; that branch was dead — `IpAddr` accepts only the canonical
121    // dotted-quad, whose final label is numeric and is refused below anyway,
122    // and IPv6 cannot reach here at all because `:` is refused above. A test
123    // named for the IP rule kept passing with it deleted, which is how a
124    // check that does nothing gets mistaken for one that does.
125    //
126    // The URL parser the fetch path uses accepts far more than `IpAddr`:
127    // `127.1`, `0177.0.0.1` and `0x7f.0.0.1` all resolve to 127.0.0.1. A real
128    // TLD is never entirely numeric, so refusing a numeric final label catches
129    // `did:web:169.254.169.254` (cloud metadata) and every non-canonical
130    // spelling of it, without reimplementing the URL parser's arithmetic. It
131    // is the same rule `normalize_handle` applies to handles.
132    if let Some(tld) = host.rsplit('.').next() {
133        if tld.starts_with(|c: char| c.is_ascii_digit()) {
134            return false;
135        }
136    }
137    // A DNS name is at most 253 bytes, and each label at most 63; anything
138    // longer cannot resolve, and an unbounded one is only useful for making us
139    // construct absurd URLs. Both bounds, to match what `normalize_handle`
140    // enforces — the earlier version checked only the total.
141    if host.len() > 253 || host.split('.').any(|label| label.len() > 63) {
142        return false;
143    }
144    host.split('.').all(|label| {
145        !label.is_empty()
146            && label
147                .bytes()
148                .all(|b| b.is_ascii_alphanumeric() || b == b'-')
149            && !label.starts_with('-')
150            && !label.ends_with('-')
151    })
152}
153
154/// Whether `did` is a DID this client can resolve: `did:plc:` or `did:web:`.
155///
156/// This is the ONLY validation applied to a DID arriving from a TXT record or a
157/// well-known document, so anything it waves through becomes the account's
158/// identity and, for `did:web`, part of a URL.
159pub fn is_atproto_did(did: &str) -> bool {
160    if let Some(ident) = did.strip_prefix("did:plc:") {
161        // base32-SORTABLE: `[a-z2-7]`, 24 characters. Not `[a-z0-9]` — `0`,
162        // `1`, `8` and `9` are not in that alphabet.
163        return ident.len() == 24
164            && ident
165                .bytes()
166                .all(|b| b.is_ascii_lowercase() || (b'2'..=b'7').contains(&b));
167    }
168    if let Some(host) = did.strip_prefix("did:web:") {
169        return is_bare_did_web_host(host);
170    }
171    false
172}
173
174/// Join one TXT record's character-strings into its value.
175///
176/// A DNS TXT record is a SEQUENCE of strings, each at most 255 bytes, and the
177/// record's value is their concatenation. A DID that crosses that boundary
178/// arrives as two chunks; passing them along as two separate records produces
179/// two truncated fragments, neither a valid DID, and the handle fails to resolve
180/// with nothing to show for it.
181///
182/// `dig +short` performs this join itself, which is precisely why a shell-based
183/// spike will never surface the bug.
184pub fn join_txt_chunks(chunks: &[&[u8]]) -> String {
185    let joined: Vec<u8> = chunks.iter().flat_map(|c| c.iter().copied()).collect();
186    String::from_utf8_lossy(&joined).into_owned()
187}
188
189/// Extract the DID from a handle's `_atproto` TXT records.
190///
191/// `Ok(None)` means no record — a normal outcome that falls through to the
192/// well-known lookup. An error means the records are present but unusable, which
193/// must NOT fall through.
194///
195/// **Two differing records fail rather than picking one.** Spec: *"If multiple
196/// valid records with different DIDs are present, resolution should fail."*
197/// Taking the first would let anyone able to add a TXT record to a zone hijack a
198/// handle that already resolves.
199///
200/// The value after `did=` is deliberately NOT trimmed, matching the spec and the
201/// reference: a padded record is invalid rather than silently repaired.
202pub fn did_from_txt_records(records: &[String]) -> Result<Option<String>> {
203    let mut candidates: Vec<&str> = records
204        .iter()
205        .filter_map(|r| r.strip_prefix("did="))
206        .collect();
207    // **Deduplicate before counting.** The spec fails resolution when multiple
208    // records name DIFFERENT DIDs; this counted RECORDS. A zone that serves the
209    // same `did=` value twice — routine with split-horizon or multi-provider DNS
210    // — was an unrecoverable error, and because it is an `Err` rather than
211    // `Ok(None)` it does not even fall through to the well-known lookup. That
212    // account simply could not log in.
213    candidates.sort_unstable();
214    candidates.dedup();
215
216    match candidates.len() {
217        0 => Ok(None),
218        1 => {
219            let did = candidates[0];
220            if !is_atproto_did(did) {
221                bail!("_atproto TXT record does not contain a usable DID: {did:?}");
222            }
223            Ok(Some(did.to_string()))
224        }
225        n => bail!("{n} `did=` TXT records present; resolution must fail rather than choose"),
226    }
227}
228
229/// Extract the DID from a `/.well-known/atproto-did` body: first line, trimmed.
230pub fn did_from_well_known(body: &str) -> Result<String> {
231    let did = body.lines().next().unwrap_or_default().trim();
232    if !is_atproto_did(did) {
233        bail!("/.well-known/atproto-did did not contain a usable DID");
234    }
235    Ok(did.to_string())
236}
237
238/// Where a DID's document lives.
239///
240/// atproto restricts `did:web` to a bare hostname — **no path components and no
241/// port** (localhost excepted, which this client has no use for). A `did:web`
242/// carrying extra segments would otherwise let a DID name an arbitrary path on a
243/// host, which is a needlessly large surface for something resolved from user
244/// input.
245pub fn did_document_url(did: &str, plc_directory: &str) -> Result<String> {
246    if let Some(ident) = did.strip_prefix("did:plc:") {
247        if !is_atproto_did(did) {
248            bail!("{did:?} is not a well-formed did:plc identifier ({ident:?})");
249        }
250        return Ok(format!("{}/{did}", plc_directory.trim_end_matches('/')));
251    }
252    if let Some(host) = did.strip_prefix("did:web:") {
253        if !is_bare_did_web_host(host) {
254            bail!(
255                "atproto did:web must be a bare, canonical hostname with no path, \
256                 port, credentials or escapes, got {host:?}"
257            );
258        }
259        return Ok(format!("https://{host}/.well-known/did.json"));
260    }
261    bail!("unsupported DID method in {did:?}; only did:plc and did:web are resolvable")
262}
263
264/// Validate a DID document against the DID that was requested.
265///
266/// The `id` check is the one that matters: without it, `plc.directory` — or
267/// whoever answers for a `did:web` host — can return a *different account's*
268/// document and every downstream decision is made about the wrong account.
269pub fn validate_did_document(document: &Value, expected_did: &str) -> Result<()> {
270    let id = document
271        .get("id")
272        .and_then(Value::as_str)
273        .context("DID document has no `id`")?;
274    if id != expected_did {
275        bail!("DID document id {id:?} does not match the requested DID {expected_did:?}");
276    }
277
278    // Duplicate service ids make "the first #atproto_pds" depend on array order.
279    //
280    // Ids are NORMALIZED to absolute form before comparing: `#atproto_pds` and
281    // `did:plc:xxx#atproto_pds` are two spellings of the SAME service, so a
282    // raw-string dedup sees two distinct ids and lets a document list the
283    // service twice with different endpoints.
284    if let Some(services) = document.get("service").and_then(Value::as_array) {
285        let mut seen = HashSet::new();
286        for service in services {
287            if let Some(sid) = service.get("id").and_then(Value::as_str) {
288                let absolute = if let Some(fragment) = sid.strip_prefix('#') {
289                    format!("{expected_did}#{fragment}")
290                } else {
291                    sid.to_string()
292                };
293                if !seen.insert(absolute.clone()) {
294                    bail!("DID document has duplicate service id {absolute:?}");
295                }
296            }
297        }
298    }
299    Ok(())
300}
301
302/// The PDS endpoint from a DID document.
303///
304/// Three conditions must all hold, per the reference's
305/// `isAtprotoPersonalDataServerService`: the id matches `#atproto_pds` in
306/// relative or absolute form, the type is exactly `AtprotoPersonalDataServer`,
307/// and `serviceEndpoint` is a **string** that parses as a URL.
308pub fn pds_endpoint(document: &Value, did: &str) -> Result<String> {
309    let services = document
310        .get("service")
311        .and_then(Value::as_array)
312        .context("DID document has no `service` array")?;
313    let absolute = format!("{did}#atproto_pds");
314
315    for service in services {
316        let id = service
317            .get("id")
318            .and_then(Value::as_str)
319            .unwrap_or_default();
320        // EXACTLY the relative or THIS DID's absolute spelling. An `ends_with`
321        // test would let a document list a decoy service -- `urn:evil#atproto_pds`,
322        // or another DID's `#atproto_pds` -- ahead of the real one and win,
323        // choosing the server for the entire session.
324        let matches_id = id == "#atproto_pds" || id == absolute;
325        let matches_type =
326            service.get("type").and_then(Value::as_str) == Some("AtprotoPersonalDataServer");
327        if !matches_id || !matches_type {
328            continue;
329        }
330        let endpoint = service
331            .get("serviceEndpoint")
332            .and_then(Value::as_str)
333            .context("#atproto_pds serviceEndpoint is not a string")?;
334        let parsed = url::Url::parse(endpoint)
335            .with_context(|| format!("#atproto_pds serviceEndpoint {endpoint:?} is not a URL"))?;
336        if parsed.scheme() != "https" {
337            // The `http`-on-loopback carve-out that used to live here was
338            // unreachable: every fetch against this endpoint goes through the
339            // SSRF guard, which rejects all loopback addresses unconditionally
340            // with no dev flag. It could never serve the local-dev case it
341            // named, and only widened what a hostile DID document could get
342            // past this function.
343            bail!("#atproto_pds serviceEndpoint must be https, got {endpoint:?}");
344        }
345        // Reject userinfo for the same reason `did:web` hosts reject `@`: the
346        // plausible-looking part becomes credentials and the REAL host is
347        // whatever follows. Every security decision downstream re-derives the
348        // host (`origin_of` and `htu` both strip userinfo), so this is not a
349        // trust bypass — but `aud` is what appears in logs and what reqwest
350        // would turn into a `Basic` credential on every XRPC call.
351        if !parsed.username().is_empty() || parsed.password().is_some() {
352            bail!("#atproto_pds serviceEndpoint must not carry credentials, got {endpoint:?}");
353        }
354        return Ok(endpoint.to_string());
355    }
356    bail!("DID document declares no #atproto_pds service for {did}")
357}
358
359/// The handle a DID document claims, normalized.
360///
361/// **The first `at://` entry only.** That entry is the document's claim; later
362/// entries are not. A membership test over the whole array would let an
363/// attacker's document list a victim's handle as a secondary entry and pass
364/// verification for it.
365pub fn declared_handle(document: &Value) -> Option<String> {
366    document
367        .get("alsoKnownAs")?
368        .as_array()?
369        .iter()
370        .filter_map(Value::as_str)
371        .find_map(|entry| entry.strip_prefix(AT_URI_PREFIX))
372        .and_then(|handle| normalize_handle(handle).ok())
373}
374
375/// Bidirectional verification: does this DID document claim `handle`?
376///
377/// Mandatory when a login starts from a handle. A handle is a DNS name under
378/// someone else's control, so without the back-check whoever controls
379/// `victim.example` can point it at any DID at all.
380pub fn verify_handle_claim(document: &Value, handle: &str) -> Result<()> {
381    let wanted = normalize_handle(handle)?;
382    let claimed = declared_handle(document)
383        .context("DID document claims no handle; cannot verify bidirectionally")?;
384    if claimed != wanted {
385        bail!("DID document claims handle {claimed:?}, not {wanted:?}");
386    }
387    Ok(())
388}
389
390#[cfg(test)]
391mod tests {
392    use super::*;
393    use serde_json::json;
394
395    const DID: &str = "did:plc:ewvi7nxzyoun6zhxrhs64oiz";
396
397    // ── handle normalization ─────────────────────────────────────────────────
398
399    #[test]
400    fn handles_are_lowercased() {
401        assert_eq!(
402            normalize_handle("Alice.BSky.Social").unwrap(),
403            "alice.bsky.social"
404        );
405        assert_eq!(
406            normalize_handle("  bob.example.com  ").unwrap(),
407            "bob.example.com"
408        );
409    }
410
411    /// **SSRF-adjacent.** `.local` / `.internal` handles are an attempt to steer
412    /// resolution at the internal network dressed up as a login. The spec
413    /// disallows these TLDs outright.
414    #[test]
415    fn reserved_tlds_are_rejected() {
416        for handle in [
417            "alice.local",
418            "alice.localhost",
419            "alice.internal",
420            "alice.arpa",
421            "alice.invalid",
422            "alice.example",
423            "alice.alt",
424            "alice.onion",
425            "deep.sub.local",
426        ] {
427            assert!(normalize_handle(handle).is_err(), "accepted {handle}");
428        }
429    }
430
431    #[test]
432    fn malformed_handles_are_rejected() {
433        for handle in [
434            "",
435            "alice",      // no TLD
436            "alice.",     // trailing dot
437            ".alice.com", // empty first label
438            "alice..com", // empty middle label
439            "-alice.com", // label starts with hyphen
440            "alice-.com", // label ends with hyphen
441            "alice.com-",
442            "alice_bob.com", // underscore not allowed
443            "alice.123",     // all-numeric TLD
444            // Spec: "The last segment (the 'top level domain') can not start
445            // with a numeric digit." Rejecting only ALL-numeric TLDs let these
446            // three through.
447            "alice.1com",
448            "alice.4chan",
449            "alice.0x",
450            "al ice.com",
451            "alice.com/path",
452            "alice.com:443",
453            "https://alice.com",
454        ] {
455            assert!(normalize_handle(handle).is_err(), "accepted {handle:?}");
456        }
457    }
458
459    #[test]
460    fn ordinary_handles_are_accepted() {
461        for handle in [
462            "alice.bsky.social",
463            "a.co",
464            "xn--80akhbyknj4f.com",
465            "very-long-label-with-hyphens.example.org",
466        ] {
467            assert!(normalize_handle(handle).is_ok(), "rejected {handle}");
468        }
469    }
470
471    // ── DNS TXT resolution ───────────────────────────────────────────────────
472
473    /// **A DNS TXT record is a sequence of character-strings**, each capped at
474    /// 255 bytes, and the record's value is their CONCATENATION. A DID that
475    /// crosses that boundary arrives as two chunks; treating them as separate
476    /// records yields two truncated fragments, neither a valid DID, and the
477    /// handle fails to resolve for no visible reason.
478    ///
479    /// `dig +short` hides this by joining for you, which is exactly why the
480    /// spike did not catch it.
481    #[test]
482    fn txt_chunks_are_joined_into_one_record_value() {
483        let long = format!("did={DID}");
484        let (head, tail) = long.split_at(20);
485        assert_eq!(
486            join_txt_chunks(&[head.as_bytes(), tail.as_bytes()]),
487            long,
488            "chunks were not concatenated"
489        );
490        assert_eq!(join_txt_chunks(&[long.as_bytes()]), long);
491        assert_eq!(join_txt_chunks(&[]), "");
492    }
493
494    /// Joined chunks must then resolve exactly as a single-chunk record would.
495    #[test]
496    fn a_did_split_across_txt_chunks_still_resolves() {
497        let long = format!("did={DID}");
498        let (head, tail) = long.split_at(20);
499        let joined = join_txt_chunks(&[head.as_bytes(), tail.as_bytes()]);
500        assert_eq!(
501            did_from_txt_records(&[joined]).unwrap().as_deref(),
502            Some(DID)
503        );
504    }
505
506    #[test]
507    fn a_single_did_record_resolves() {
508        let records = vec![format!("did={DID}")];
509        assert_eq!(
510            did_from_txt_records(&records).unwrap().as_deref(),
511            Some(DID)
512        );
513    }
514
515    #[test]
516    fn unrelated_txt_records_are_ignored() {
517        let records = vec![
518            "v=spf1 -all".to_string(),
519            format!("did={DID}"),
520            "google-site-verification=abc".to_string(),
521        ];
522        assert_eq!(
523            did_from_txt_records(&records).unwrap().as_deref(),
524            Some(DID)
525        );
526    }
527
528    /// Spec: *"If multiple valid records with different DIDs are present,
529    /// resolution should fail."* Picking the first would let anyone who can add
530    /// a TXT record to a zone hijack a handle that already has one.
531    #[test]
532    fn multiple_did_records_fail_rather_than_picking_one() {
533        let records = vec![
534            format!("did={DID}"),
535            "did=did:plc:aaaaaaaaaaaaaaaaaaaaaaaa".to_string(),
536        ];
537        assert!(did_from_txt_records(&records).is_err());
538    }
539
540    #[test]
541    fn no_did_record_is_absent_not_an_error() {
542        let records = vec!["v=spf1 -all".to_string()];
543        assert_eq!(did_from_txt_records(&records).unwrap(), None);
544    }
545
546    /// The value after `did=` is NOT trimmed — deliberately, to stay consistent
547    /// with the spec — so a padded record is invalid rather than silently fixed.
548    #[test]
549    fn a_padded_did_value_is_invalid() {
550        let records = vec![format!("did= {DID}")];
551        assert!(did_from_txt_records(&records).is_err());
552    }
553
554    #[test]
555    fn a_non_did_value_is_rejected() {
556        for value in ["did=notadid", "did=", "did=did:unknown:xyz"] {
557            assert!(
558                did_from_txt_records(&[value.to_string()]).is_err(),
559                "accepted {value}"
560            );
561        }
562    }
563
564    // ── .well-known/atproto-did ──────────────────────────────────────────────
565
566    #[test]
567    fn the_well_known_body_takes_the_first_line_trimmed() {
568        assert_eq!(did_from_well_known(&format!("{DID}\n")).unwrap(), DID);
569        assert_eq!(did_from_well_known(&format!("  {DID}  ")).unwrap(), DID);
570        assert_eq!(
571            did_from_well_known(&format!("{DID}\nignored")).unwrap(),
572            DID
573        );
574    }
575
576    #[test]
577    fn a_well_known_body_that_is_not_a_did_is_rejected() {
578        for body in ["", "\n", "not a did", "<html>", "did:unknown:x"] {
579            assert!(did_from_well_known(body).is_err(), "accepted {body:?}");
580        }
581    }
582
583    // ── DID → document URL ───────────────────────────────────────────────────
584
585    #[test]
586    fn plc_dids_map_to_the_directory() {
587        assert_eq!(
588            did_document_url(DID, "https://plc.directory").unwrap(),
589            format!("https://plc.directory/{DID}")
590        );
591    }
592
593    #[test]
594    fn did_web_maps_to_the_hosts_well_known() {
595        assert_eq!(
596            did_document_url("did:web:example.com", "https://plc.directory").unwrap(),
597            "https://example.com/.well-known/did.json"
598        );
599    }
600
601    /// atproto restricts `did:web` to a bare hostname: no path components, no
602    /// port. (`:` is the path separator in a did:web method-specific id; a port
603    /// is spelled `%3A`.)
604    #[test]
605    fn did_web_with_a_path_or_port_is_rejected() {
606        for did in [
607            "did:web:example.com:path",
608            "did:web:example.com:8080",
609            "did:web:example.com%3A8080",
610            "did:web:example.com:path:to:doc",
611        ] {
612            assert!(
613                did_document_url(did, "https://plc.directory").is_err(),
614                "accepted {did}"
615            );
616            assert!(!is_atproto_did(did), "is_atproto_did accepted {did}");
617        }
618    }
619
620    /// **Host confusion.** `did:web:good.com@evil.com` builds
621    /// `https://good.com@evil.com/…`, whose ACTUAL host is `evil.com` — the
622    /// plausible-looking part is demoted to userinfo. A `/` is a straight path
623    /// injection. Neither may survive as far as URL construction.
624    #[test]
625    fn did_web_host_confusion_and_path_injection_are_rejected() {
626        for did in [
627            "did:web:good.com@evil.com",
628            "did:web:evil.com/x",
629            "did:web:evil.com/.well-known/did.json#",
630            "did:web:%00",
631            "did:web:%2e%2e",
632            "did:web:ex ample.com",
633            "did:web:",
634            "did:web:.",
635            "did:web:-example.com",
636            "did:web:Example.com", // must be canonical lowercase
637        ] {
638            assert!(!is_atproto_did(did), "is_atproto_did accepted {did:?}");
639            assert!(
640                did_document_url(did, "https://plc.directory").is_err(),
641                "built a URL for {did:?}"
642            );
643        }
644    }
645
646    /// `did:plc` identifiers are base32-SORTABLE: `[a-z2-7]`. `0`, `1`, `8` and
647    /// `9` are not in that alphabet.
648    #[test]
649    fn did_plc_uses_the_base32_sortable_alphabet() {
650        assert!(is_atproto_did(DID));
651        for did in [
652            "did:plc:aaaaaaaaaaaaaaaaaaaaaa01",
653            "did:plc:aaaaaaaaaaaaaaaaaaaaaa89",
654            "did:plc:AAAAAAAAAAAAAAAAAAAAAAAA",
655            "did:plc:tooshort",
656            "did:plc:aaaaaaaaaaaaaaaaaaaaaaaaa",
657        ] {
658            assert!(!is_atproto_did(did), "accepted {did}");
659        }
660    }
661
662    #[test]
663    fn unsupported_did_methods_are_rejected() {
664        for did in ["did:key:z6Mk", "did:example:123", "notadid", "", "did:"] {
665            assert!(
666                did_document_url(did, "https://plc.directory").is_err(),
667                "accepted {did}"
668            );
669        }
670    }
671
672    // ── DID document validation ──────────────────────────────────────────────
673
674    fn doc() -> serde_json::Value {
675        json!({
676            "id": DID,
677            "alsoKnownAs": ["at://alice.bsky.social"],
678            "service": [{
679                "id": "#atproto_pds",
680                "type": "AtprotoPersonalDataServer",
681                "serviceEndpoint": "https://pds.example.com"
682            }]
683        })
684    }
685
686    /// Without this, `plc.directory` — or whoever answers for a `did:web` host —
687    /// can hand back a different account's document entirely.
688    #[test]
689    fn the_document_id_must_match_the_did_requested() {
690        let mut d = doc();
691        d["id"] = json!("did:plc:someoneelse00000000000");
692        assert!(validate_did_document(&d, DID).is_err());
693        assert!(validate_did_document(&doc(), DID).is_ok());
694    }
695
696    #[test]
697    fn a_document_without_an_id_is_rejected() {
698        let mut d = doc();
699        d.as_object_mut().unwrap().remove("id");
700        assert!(validate_did_document(&d, DID).is_err());
701    }
702
703    /// With duplicate ids, "the first `#atproto_pds`" depends on array order,
704    /// which is not a property anyone should rely on.
705    #[test]
706    fn duplicate_service_ids_are_rejected() {
707        let mut d = doc();
708        d["service"] = json!([
709            {"id": "#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://a.example"},
710            {"id": "#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://b.example"}
711        ]);
712        assert!(validate_did_document(&d, DID).is_err());
713    }
714
715    // ── PDS endpoint extraction ──────────────────────────────────────────────
716
717    #[test]
718    fn the_pds_endpoint_is_extracted() {
719        assert_eq!(
720            pds_endpoint(&doc(), DID).unwrap(),
721            "https://pds.example.com"
722        );
723    }
724
725    /// The id may be relative (`#atproto_pds`) or absolute
726    /// (`did:plc:xxx#atproto_pds`) — but ONLY those two spellings.
727    #[test]
728    fn an_absolute_service_id_is_accepted() {
729        let mut d = doc();
730        d["service"][0]["id"] = json!(format!("{DID}#atproto_pds"));
731        assert_eq!(pds_endpoint(&d, DID).unwrap(), "https://pds.example.com");
732    }
733
734    /// **PDS steering.** Matching on "ends with `#atproto_pds`" lets a DID
735    /// document put a DECOY service first whose id merely has that suffix, and
736    /// win. The PDS is what discovery runs against and what every later XRPC
737    /// call targets, so this chooses the server for the whole session.
738    ///
739    /// The decoy must be FIRST here — with a single service there is nothing to
740    /// beat, which is how the original tests missed this.
741    #[test]
742    fn a_foreign_service_id_ending_in_atproto_pds_does_not_win() {
743        let mut d = doc();
744        d["service"] = json!([
745            {"id": "urn:evil#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://attacker.example"},
746            {"id": "#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://real-pds.example"}
747        ]);
748        assert_eq!(
749            pds_endpoint(&d, DID).unwrap(),
750            "https://real-pds.example",
751            "a decoy service id steered the PDS"
752        );
753
754        // And an id belonging to a DIFFERENT DID is not ours either.
755        d["service"] = json!([
756            {"id": "did:plc:aaaaaaaaaaaaaaaaaaaaaaaa#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://attacker.example"},
757            {"id": "#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://real-pds.example"}
758        ]);
759        assert_eq!(pds_endpoint(&d, DID).unwrap(), "https://real-pds.example");
760    }
761
762    /// The relative and absolute spellings are the SAME service, so listing both
763    /// is a duplicate — which the raw-string dedup did not see.
764    #[test]
765    fn the_relative_and_absolute_spellings_count_as_one_service() {
766        let mut d = doc();
767        d["service"] = json!([
768            {"id": format!("{DID}#atproto_pds"), "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://attacker.example"},
769            {"id": "#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://real-pds.example"}
770        ]);
771        assert!(
772            validate_did_document(&d, DID).is_err(),
773            "two spellings of the same service id were not seen as duplicates"
774        );
775    }
776
777    /// A DID document is attacker-controlled in the `did:web` case, and tokens
778    /// go to this host, so plaintext is refused — **including on loopback**.
779    ///
780    /// This previously carved out `http` on loopback "for a local dev PDS". That
781    /// carve-out was unreachable: every fetch against this endpoint goes through
782    /// the SSRF guard, which rejects all loopback addresses unconditionally with
783    /// no dev flag and no config bypass. It could never serve the case it named,
784    /// and only widened what a hostile document could get past this function.
785    #[test]
786    fn a_plaintext_http_pds_is_rejected_including_on_loopback() {
787        let mut d = doc();
788        for endpoint in [
789            "http://pds.attacker.example",
790            "http://10.0.0.5",
791            "http://localhost:2583",
792            "http://127.0.0.1:2583",
793        ] {
794            d["service"][0]["serviceEndpoint"] = json!(endpoint);
795            assert!(pds_endpoint(&d, DID).is_err(), "accepted {endpoint}");
796        }
797    }
798
799    /// **Credentials in a `serviceEndpoint` are refused.**
800    ///
801    /// `https://good.example@attacker.example` has real host `attacker.example`
802    /// — the exact demotion this module already rejects for `did:web` hosts. The
803    /// downstream security decisions re-derive the host either way, so this is
804    /// not a trust bypass; but this value is `aud`, it is what appears in logs,
805    /// and reqwest would turn the userinfo into a `Basic` credential on every
806    /// XRPC call to the PDS.
807    #[test]
808    fn a_pds_endpoint_carrying_credentials_is_refused() {
809        let mut d = doc();
810        for endpoint in [
811            "https://good.example@attacker.example",
812            "https://u:p@attacker.example/x",
813        ] {
814            d["service"][0]["serviceEndpoint"] = json!(endpoint);
815            assert!(pds_endpoint(&d, DID).is_err(), "accepted {endpoint}");
816        }
817    }
818
819    /// `did:web` hosts are held to the SAME policy as handles: no reserved TLD,
820    /// no IP literal. The SSRF guard blocks these at fetch time, but this module
821    /// states that rejecting here is what makes the guard a second line of
822    /// defence rather than the only one.
823    ///
824    /// Every IP case below is refused by the numeric-final-label rule — there
825    /// is no separate IP check, and there was a dead one until a mutation
826    /// showed this test green with it deleted. Deleting the label rule instead
827    /// fails every IP case here.
828    #[test]
829    fn did_web_hosts_obey_the_reserved_tld_and_ip_policy() {
830        for did in [
831            "did:web:169.254.169.254", // cloud metadata
832            "did:web:127.0.0.1",
833            "did:web:10.0.0.5",
834            "did:web:pds.internal",
835            "did:web:printer.local",
836            "did:web:something.localhost",
837            "did:web:site.onion",
838        ] {
839            assert!(!is_atproto_did(did), "accepted {did}");
840        }
841        // Non-canonical IP spellings resolve to loopback just as well, and
842        // `IpAddr::parse` accepts none of them — a numeric final label does.
843        for did in ["did:web:127.1", "did:web:0177.0.0.1", "did:web:0x7f.0.0.1"] {
844            assert!(!is_atproto_did(did), "accepted {did}");
845        }
846        // Length bounds: total and per-label, matching `normalize_handle`.
847        assert!(!is_atproto_did(&format!("did:web:{}.com", "a".repeat(300))));
848        assert!(
849            !is_atproto_did(&format!("did:web:{}.com", "a".repeat(64))),
850            "a 64-byte label exceeds the DNS limit"
851        );
852        assert!(is_atproto_did(&format!("did:web:{}.com", "a".repeat(63))));
853        // A digit-leading TLD is not a TLD.
854        assert!(!is_atproto_did("did:web:foo.1com"));
855        // Ordinary hosts still work.
856        assert!(is_atproto_did("did:web:pds.example.com"));
857    }
858
859    /// All three conditions must hold: id, type, and a parseable endpoint.
860    #[test]
861    fn a_service_failing_any_condition_is_not_the_pds() {
862        let cases = [
863            json!({"id": "#atproto_pds", "type": "SomethingElse", "serviceEndpoint": "https://a.example"}),
864            json!({"id": "#other", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://a.example"}),
865            json!({"id": "urn:evil#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "https://a.example"}),
866            json!({"id": "#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": "not a url"}),
867            json!({"id": "#atproto_pds", "type": "AtprotoPersonalDataServer", "serviceEndpoint": ["https://a.example"]}),
868            json!({"id": "#atproto_pds", "type": "AtprotoPersonalDataServer"}),
869        ];
870        for svc in cases {
871            let mut d = doc();
872            d["service"] = json!([svc.clone()]);
873            assert!(pds_endpoint(&d, DID).is_err(), "accepted {svc}");
874        }
875    }
876
877    #[test]
878    fn a_document_with_no_services_has_no_pds() {
879        let mut d = doc();
880        d["service"] = json!([]);
881        assert!(pds_endpoint(&d, DID).is_err());
882        d.as_object_mut().unwrap().remove("service");
883        assert!(pds_endpoint(&d, DID).is_err());
884    }
885
886    // ── bidirectional verification ───────────────────────────────────────────
887
888    #[test]
889    fn the_declared_handle_is_the_first_at_uri_entry_normalized() {
890        let mut d = doc();
891        d["alsoKnownAs"] = json!(["at://Alice.BSky.Social"]);
892        assert_eq!(declared_handle(&d).as_deref(), Some("alice.bsky.social"));
893    }
894
895    /// **The attack this closes.** An attacker's DID document can list the
896    /// victim's handle as a SECONDARY entry. Only the first `at://` entry is the
897    /// document's claim, so a membership test would accept the attacker's
898    /// document for the victim's handle.
899    #[test]
900    fn only_the_first_at_uri_entry_counts() {
901        // Note: not `.example` — that is a reserved TLD and would be rejected by
902        // `normalize_handle` before the ordering logic was ever reached.
903        let mut d = doc();
904        d["alsoKnownAs"] = json!(["at://attacker.com", "at://victim.com"]);
905        assert_eq!(declared_handle(&d).as_deref(), Some("attacker.com"));
906        assert!(verify_handle_claim(&d, "victim.com").is_err());
907        assert!(verify_handle_claim(&d, "attacker.com").is_ok());
908    }
909
910    /// Non-`at://` entries are skipped when looking for the first claim.
911    #[test]
912    fn non_at_uri_entries_are_skipped() {
913        let mut d = doc();
914        d["alsoKnownAs"] = json!(["https://alice.example", "at://alice.bsky.social"]);
915        assert_eq!(declared_handle(&d).as_deref(), Some("alice.bsky.social"));
916    }
917
918    #[test]
919    fn a_document_claiming_no_handle_fails_verification() {
920        let mut d = doc();
921        d["alsoKnownAs"] = json!([]);
922        assert!(verify_handle_claim(&d, "alice.bsky.social").is_err());
923        d.as_object_mut().unwrap().remove("alsoKnownAs");
924        assert!(verify_handle_claim(&d, "alice.bsky.social").is_err());
925    }
926
927    /// A malformed claim must not be usable as a wildcard.
928    #[test]
929    fn a_malformed_claimed_handle_fails_verification() {
930        for claim in ["at://", "at://not a handle", "at://alice.local"] {
931            let mut d = doc();
932            d["alsoKnownAs"] = json!([claim]);
933            assert!(
934                verify_handle_claim(&d, "alice.bsky.social").is_err(),
935                "accepted claim {claim}"
936            );
937        }
938    }
939
940    /// **A malformed FIRST claim must not fall through to the second.**
941    ///
942    /// The two neighbouring tests could not see this between them: this one's
943    /// sibling uses single-element arrays, so "reject" and "skip to the next"
944    /// look identical, and `only_the_first_at_uri_entry_counts` uses two VALID
945    /// entries, so nothing forces the first to be the one that fails.
946    ///
947    /// A mutation turning `find_map(strip).and_then(normalize)` into
948    /// `filter_map(strip).find_map(normalize)` — skip past an unusable claim —
949    /// passed the whole suite. Under it, an attacker document whose first entry
950    /// is junk verifies as whatever the SECOND entry says, which is precisely
951    /// what "only the first entry counts" exists to stop.
952    #[test]
953    fn a_malformed_first_claim_does_not_fall_through_to_the_second() {
954        for bad_first in ["at://", "at://not a handle", "at://alice.local"] {
955            let mut d = doc();
956            d["alsoKnownAs"] = json!([bad_first, "at://victim.com"]);
957            assert_eq!(
958                declared_handle(&d),
959                None,
960                "a malformed first claim ({bad_first}) was skipped and the second was taken"
961            );
962            assert!(
963                verify_handle_claim(&d, "victim.com").is_err(),
964                "({bad_first}) the second entry verified as the account's handle"
965            );
966        }
967    }
968
969    #[test]
970    fn verification_is_case_insensitive_on_both_sides() {
971        let mut d = doc();
972        d["alsoKnownAs"] = json!(["at://Alice.BSky.Social"]);
973        assert!(verify_handle_claim(&d, "ALICE.bsky.SOCIAL").is_ok());
974    }
975}