Skip to main content

feather_reader/
atproto.rs

1//! The atproto identity + PDS record layer.
2//!
3//! FeatherReader's defining bet is that a user's feed
4//! subscriptions, folders, saved items, and batched read-state live as records
5//! in the user's **own** atproto PDS under the open `community.lexicon.rss.*`
6//! community lexicon — not in the app's database. This module is the client that
7//! reads and writes those records.
8//!
9//! It has three layers:
10//!
11//! 1. **Identity resolution** ([`resolve_handle`], [`resolve_did_to_pds`]) —
12//!    turn a handle (`alice.example.com`) into a DID (`did:plc:…`), then resolve
13//!    the DID document to the PDS service endpoint. Handles resolve via the
14//!    account's PDS `com.atproto.identity.resolveHandle` (or the well-known
15//!    `/.well-known/atproto-did`); DIDs resolve via the PLC directory
16//!    (`did:plc:*`) or the `did:web` well-known document.
17//! 2. **A lightweight [`PdsClient`]** — holds the resolved DID, the PDS base URL,
18//!    and an [`Auth`] token, and exposes typed calls over `com.atproto.repo.*`:
19//!    [`list_records`](PdsClient::list_records),
20//!    `create_record`,
21//!    `put_record`,
22//!    [`delete_record`](PdsClient::delete_record), and
23//!    `apply_writes` (the **batch** call the
24//!    read-state flusher uses to coalesce many per-feed cursor writes into one
25//!    round-trip).
26//! 3. **Typed convenience wrappers** wired to the [`crate::lexicon`] record
27//!    types (list/create [`Subscription`]/[`Folder`]/[`Saved`], put
28//!    [`ReadState`], batch-flush many `ReadState` cursors).
29//!
30//! ## Auth — the OAuth sidecar is the live path
31//!
32//! Auth is a **trait/enum boundary** so the mechanism can vary without touching
33//! call sites. There are three paths:
34//!
35//! * **The live path — the atproto OAuth confidential client, via [`SidecarClient`].**
36//!   atproto OAuth (DPoP, PAR, token refresh) is fiddly and is **not** hand-rolled
37//!   in Rust: it runs in a small, supported `@atproto/oauth-client-node` sidecar.
38//!   The Rust server never holds
39//!   PDS tokens — it POSTs every `com.atproto.repo.*` op to the sidecar's
40//!   `/internal/repo` endpoint (gated by a shared `X-Internal-Secret`), and the
41//!   sidecar restores the DID's OAuth session (transparent DPoP + token refresh)
42//!   and runs the matching XRPC call. [`SidecarClient`] is that client; the typed
43//!   convenience wrappers (list/create/put/delete subscriptions, batch-flush
44//!   read-state) live on it and map 1:1 to the old [`PdsClient`] surface.
45//! * **The interim path — [`Auth::Session`] (app password).** A session obtained
46//!   from `com.atproto.server.createSession`. Kept behind the [`Auth`] seam, but
47//!   it is **no longer the live path**: [`PdsClient`] and
48//!   [`login_with_app_password`] remain for tests, while [`SidecarClient`] is
49//!   what the web layer routes through.
50//!
51//!   ⚠️ **This is no longer a working "local runs without the sidecar" fallback,
52//!   and the docs used to claim otherwise.** Since v0.2.8 every [`PdsClient`]
53//!   request goes through the SSRF guard, which refuses loopback, RFC1918, ULA
54//!   and `100.64/10` (Tailscale). So pointing this at `http://localhost:2583`
55//!   or a tailnet PDS now fails with *"refusing to fetch forbidden (internal)
56//!   address"* rather than returning a session. That is the guard behaving
57//!   correctly — the target host is attacker-influenced in the cases that
58//!   matter, and a dev-only escape hatch is exactly the kind of flag that ends
59//!   up set in production — but it does mean a local-PDS workflow needs the
60//!   PDS reachable on a public address, or a deliberate change here.
61//! * **The public-read path — [`Auth::Anonymous`], via [`PdsClient::anonymous`].**
62//!   `com.atproto.repo.listRecords` is public on a standard PDS, so a stranger's
63//!   `community.lexicon.rss.*` records can be read with no credentials at all.
64//!   An anonymous client sends no `Authorization` header and is **read-only** —
65//!   every write fails closed on [`Auth::bearer`]. Because the target host is
66//!   then chosen by a stranger, the read is routed through
67//!   [`crate::net::guarded_get_no_privacy`] (per-hop SSRF re-validation +
68//!   connect-pinning) and capped by [`crate::net::read_capped`].
69//!
70//! ## Every PDS request goes through the SSRF guard
71//!
72//! A PDS host is *never* a host FeatherReader chose: it comes out of a DID
73//! document, which is attacker-controllable. So identity resolution, the record
74//! **reads**, and the record **writes** all route through [`crate::net`] —
75//! [`crate::net::guarded_get_no_privacy`] and
76//! [`crate::net::guarded_post_json`] — rather than the shared
77//! `reqwest::Client`. [`resolve_did_to_pds`] runs
78//! [`crate::net::assert_public_target`] on the `serviceEndpoint` it returns, but
79//! that check is a *separate DNS resolution* from the later request; only
80//! re-vetting and connect-pinning at request time closes the rebinding window.
81//! The writes matter most: they carry the session bearer, and
82//! [`login_with_app_password`] carries the app password in the request **body**,
83//! where reqwest's cross-origin header sanitisation offers no protection at all
84//! — which is why the guarded POST refuses redirects outright.
85//!
86//! All network I/O is `reqwest` (rustls, no OpenSSL); every fallible path returns
87//! [`anyhow::Result`] or the typed [`AtProtoError`] — nothing panics.
88
89use std::sync::Arc;
90
91use anyhow::{Context, Result};
92use reqwest::header::{HeaderName, HeaderValue, AUTHORIZATION};
93use reqwest::{Client, StatusCode};
94use serde::de::DeserializeOwned;
95use serde::{Deserialize, Serialize};
96use serde_json::{json, Value};
97
98use crate::lexicon::{self, Folder, ReadState, Saved, Subscription};
99
100/// The public PLC directory, used to resolve `did:plc:*` DIDs to their DID
101/// document (and thus their PDS service endpoint).
102pub const DEFAULT_PLC_DIRECTORY: &str = "https://plc.directory";
103
104/// The default appview/entryway used only as a bootstrap host for handle
105/// resolution when the caller has no PDS hint yet. Handle resolution ultimately
106/// works against any atproto host that implements
107/// `com.atproto.identity.resolveHandle`; `bsky.social` is a reliable default.
108pub const DEFAULT_RESOLVER_HOST: &str = "https://bsky.social";
109
110/// Hard cap on cursor pages any `list_all_records` walk will follow.
111///
112/// [`crate::net::read_capped`] bounds each individual response, but nothing
113/// bounded the *accumulation* across pages: a repo host that returns a full page
114/// and a fresh cursor forever walks memory until the (512 MB) box dies. At 100
115/// records per page this admits 20 000 records — far past any real
116/// `community.lexicon.rss.*` collection — while making the loop finite against a
117/// host we do not control. Mirrors [`crate::network::MAX_PAGES`], which bounds
118/// the relay walk for the same reason.
119const MAX_LIST_PAGES: usize = 200;
120
121/// Hard cap on the records a single `list_all_records` walk will accumulate.
122///
123/// [`MAX_LIST_PAGES`] bounds how many REQUESTS a walk makes. It bounds the
124/// accumulated memory only if the server honours `limit=100` — and a repo host
125/// we did not choose has no obligation to. Measured: an 8 MB page (the
126/// [`crate::net::read_capped`] ceiling) holds ~95 000 minimal records and
127/// retains ~23 MB as `Vec<RecordEntry>`, so the page cap alone admits gigabytes
128/// on a 512 MB box.
129///
130/// 20 000 is the number [`MAX_LIST_PAGES`]'s own comment already claimed — this
131/// makes the claim true rather than conditional on the server's cooperation.
132const MAX_LIST_RECORDS: usize = 20_000;
133
134/// The same cap for a collection whose records are **large**.
135///
136/// [`MAX_LIST_RECORDS`]'s figure was measured against *minimal* records
137/// (~1 KB). A `site.standard.document` carries the whole article — ~17 KB
138/// measured across 449 real ones — so 20 000 of them is ~340 MB retained on a
139/// 512 MB box. Sized to the record, not to the protocol.
140pub(crate) const MAX_LARGE_RECORDS: usize = 2_000;
141
142/// The at-URI scheme prefix, **the one Rust spelling**. Every Rust guard that
143/// asks "is this an at-URI" strips or compares this.
144///
145/// **SQL no longer holds a second opinion.** There used to be a matching string
146/// predicate in `store`, and this comment claimed a test pinned the two in
147/// agreement. Both are gone: the predicate was deleted when `feeds.kind` became
148/// a cache of [`crate::feed::FeedKind::of`], re-derived from the URL rather than
149/// re-described in SQL, and no such test survived it. Nothing outside Rust
150/// decides what an at-URI is, so there is nothing left to keep in step.
151pub(crate) const AT_URI_PREFIX: &str = "at://";
152
153/// Strip the at-URI scheme **case-insensitively**, returning the body.
154///
155/// Schemes are case-insensitive per RFC 3986 and `Url::parse` folds them, so
156/// `At://` names the same thing as `at://`. Recognition has to match that, or a
157/// mixed-case row is an at-URI to the fetcher (which refuses it) and an
158/// ordinary URL to every guard — polled forever, failing forever. Whether such
159/// a spelling may be STORED is a separate question, answered no.
160pub(crate) fn strip_at_prefix(url: &str) -> Option<&str> {
161    url.get(..AT_URI_PREFIX.len())
162        .filter(|p| p.eq_ignore_ascii_case(AT_URI_PREFIX))
163        .map(|p| &url[p.len()..])
164}
165
166/// atproto's record-key rules, all of them: charset `[A-Za-z0-9._:~-]`, length
167/// 1..=512, and not `.` or `..`. The repo's TID tests state the same rule; this
168/// is the one place it is enforced on a key that arrives from outside.
169pub(crate) fn is_valid_rkey(rkey: &str) -> bool {
170    !rkey.is_empty()
171        && rkey.len() <= 512
172        && rkey != "."
173        && rkey != ".."
174        && rkey
175            .chars()
176            .all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '_' | ':' | '~' | '-'))
177}
178
179/// Accumulate a page for a **reading** walk, keeping what fits and reporting
180/// whether anything was dropped.
181///
182/// **A truncation, never an error** — the opposite of [`extend_bounded`], and
183/// deliberately so. That function's refusal exists because its caller feeds
184/// `replace_sub_refs`, where a short list is revoked access. A walk that only
185/// ADDS entries has no such hazard, and refusing there is strictly worse: a
186/// publication with more documents than the cap would fail on every poll, so
187/// an ordinary long-running blog becomes permanently unreadable instead of
188/// partially read. The records kept are the ones the PDS returned first.
189pub(crate) fn extend_truncating(
190    out: &mut Vec<RecordEntry>,
191    page: Vec<RecordEntry>,
192    max: usize,
193) -> bool {
194    let room = max.saturating_sub(out.len());
195    // **Strictly greater.** `>=` called an exactly-full final page a
196    // truncation, so a collection holding exactly `max` records warned that it
197    // had dropped something on every poll.
198    let dropped = page.len() > room;
199    out.extend(page.into_iter().take(room));
200    dropped
201}
202
203/// The result of a bounded walk: what was read, and whether that is all of it.
204///
205/// **`complete` is a fact the caller cannot recover afterwards.** A short list
206/// from a truncating walk looks exactly like a short collection, and the
207/// difference is the one that matters: "this publication has nine articles" and
208/// "this reader gave up after nine" are the same `Vec` and very different
209/// answers.
210#[derive(Debug)]
211pub struct RecordWalk {
212    /// The records kept, in the order the PDS returned them.
213    pub records: Vec<RecordEntry>,
214    /// True when the collection ran out before any bound did.
215    pub complete: bool,
216    /// Records skipped because their envelope was malformed (#177). Only a
217    /// walk that SKIPS can report this; the walks that feed `replace_sub_refs`
218    /// refuse instead, with [`MalformedRecords`].
219    pub malformed: usize,
220    /// The walk stopped because its byte budget ran out — not at a record
221    /// cap, the page limit or a repeated cursor. Only this end is one a
222    /// caller can change by reading with a budget of its own (#229): the
223    /// others stop a read alone at the same place.
224    pub out_of_budget: bool,
225}
226
227impl RecordWalk {
228    fn complete(records: Vec<RecordEntry>) -> Self {
229        Self {
230            records,
231            complete: true,
232            malformed: 0,
233            out_of_budget: false,
234        }
235    }
236    fn partial(records: Vec<RecordEntry>) -> Self {
237        Self {
238            records,
239            complete: false,
240            malformed: 0,
241            out_of_budget: false,
242        }
243    }
244}
245
246/// The memory one walk may retain.
247///
248/// **A record cap bounds memory only if you know what a record costs.** The
249/// caps above are counts, chosen against a measured ~17 KB document, and
250/// `MAX_LIST_PAGES` bounds requests rather than bytes. A PDS whose records are
251/// not that shape satisfies every count and still exhausts the box.
252///
253/// **128 MiB of ACCUMULATION per read — which is not the same as 128 MiB of
254/// memory, and an earlier version of this comment said it was.**
255///
256/// Every charge here is taken after `serde_json` has already built the page, so
257/// the true peak is this ceiling plus one page's tree, and a page's tree is not
258/// small: measured, an 8 MiB response of `{"":0}` objects retains 824 MB, a
259/// wire-to-heap amplification of 98x. A bound consulted after the allocation
260/// cannot prevent that allocation. What it does prevent is the accumulation
261/// across pages and across the walks of one read, which is the part that scales
262/// with how long a walk runs rather than with one response.
263///
264/// Closing the single-page case needs a smaller wire cap for `listRecords` or a
265/// parser that counts as it goes. Neither belongs to this bound; both are filed
266/// as #197 rather than implied here.
267///
268/// A caller passes one [`ByteBudget`] into every walk it makes, so a publication
269/// read — which runs a second walk while still holding the first's records — is
270/// bounded once rather than twice. Two independent ceilings put roughly 384 MB of
271/// accumulation in flight: 128 for the publications, 128 for the documents, and a
272/// further 128 of transient page because the per-page check compared against the
273/// ceiling instead of what was left.
274///
275/// **Only one caller threads it today**: the publication reader, polled by the
276/// scheduler's publication loop since 0.4.0. Every other live read
277/// builds its own ceiling per walk, so the per-request total is still a multiple
278/// of this number — two walks on an OPML export, four on a login — and nothing
279/// bounds concurrent requests at all.
280/// An earlier version of this constant was also 128 MiB while
281/// [`approx_bytes`] charged serialized length — 42x optimistic on hostile
282/// shapes, so the bound was nearly fiction. It was then cut to 64 MiB to
283/// compensate. Charging nodes removed the reason for the cut: the charge is now
284/// at or above what the page really retains, so 128 MiB of budget is at most
285/// 128 MiB of memory, which a 512 MB box carries.
286///
287/// The cut had a cost, measured rather than assumed. Per walk:
288///
289/// - The subscription walks cap at 20 000 records (5 000 on the live one). A
290///   real five-field subscription charges **2 188 bytes** here, and one carrying
291///   a folder and a fetch hint charges **2 764** — so a full repo is 42 to 53 MB,
292///   which was 65 to 82 % of a 64 MiB budget. Their verdict is a hard refusal
293///   that drops the reader into the fail-closed branch, so an account near the
294///   record cap with slightly longer titles would have served a stale projection
295///   on every poll, permanently. At 128 MiB that is 41 % and the count still
296///   binds first. An earlier version of this comment claimed 1.5 KB and 30 MB;
297///   that is a three-field record, not a real one.
298/// - The publication walk caps at [`MAX_LARGE_RECORDS`] (2 000). At the measured
299///   ~17 KB document that is about 37 MB either way. Above roughly 66 KB per
300///   article the budget binds first and the walk truncates early, reporting
301///   `complete: false` as it already does for the record cap.
302///
303/// So the counts bind first on everything measured, and a publication of
304/// extremely long articles truncates sooner than the count would. It remains
305/// untrue that nothing truncates that did not truncate before.
306pub(crate) const MAX_LIST_BYTES: usize = 128 * 1024 * 1024;
307
308/// What one record retains once parsed.
309///
310/// **Nodes, not serialized text.** An earlier version of this charged the
311/// length of the JSON, which is the wrong quantity by up to 42x: a parsed value
312/// is a tree of 32-byte nodes held in vectors that over-allocate, so `[[],[]…]`
313/// costs three bytes on the wire and well over a hundred in memory. Measured
314/// against that estimate, a budget reporting 119 MiB held a process at 5.6 GiB.
315///
316/// Every arm therefore charges at least the node itself, and a container
317/// charges for the slack its backing allocation carries. The result
318/// over-estimates on every adversarial shape and costs honest traffic a couple
319/// of percent, which is the direction a bound has to err in.
320pub(crate) fn approx_bytes(entry: &RecordEntry) -> usize {
321    2 * std::mem::size_of::<RecordEntry>()
322        + entry.uri.len()
323        + entry.cid.as_ref().map_or(0, String::len)
324        + json_bytes(&entry.value)
325}
326
327/// What a parsed JSON value retains, without measuring the heap.
328fn json_bytes(v: &serde_json::Value) -> usize {
329    /// Every value, of every kind, occupies one of these wherever it sits.
330    const NODE: usize = std::mem::size_of::<serde_json::Value>();
331    /// Two nodes per value: the slot it occupies, and the slack the container
332    /// holding it carries — a `Vec` grows by doubling, so up to one spare slot
333    /// per live one.
334    const SLOT: usize = 2 * NODE;
335    /// A map entry is a tree node of its own, with links and a key beside the
336    /// value. Rounded up rather than derived, since the layout is not ours.
337    const MAP_ENTRY: usize = 104;
338    /// A map's backing node, allocated whole.
339    ///
340    /// **Empirical, and not derived from anything the compiler checks.** Unlike
341    /// [`NODE`], which is a `size_of`, this and `MAP_ENTRY` come from measuring
342    /// `std`'s `BTreeMap` layout — B = 6, so eleven pairs to a leaf — under the
343    /// `serde_json` in this lockfile. A toolchain that changes that layout, or a
344    /// `serde_json` that swaps the map type, moves the real cost without moving
345    /// these. The known-answer test below is the tripwire, and it is only as
346    /// good as the day its figures were taken.
347    ///
348    /// `serde_json::Map` is a `BTreeMap` here — no `preserve_order` in the
349    /// lock — and its leaf carries room for eleven pairs whether or not they
350    /// are used, measured at ~632 bytes. So a one-key object costs what an
351    /// eleven-key one does, and a chain of them costs that per level. Charging
352    /// a container's minimum the way an array does under-reports this by about
353    /// half, which is the same failure as the version this replaces, two orders
354    /// of magnitude smaller.
355    const MAP_NODE: usize = 512;
356    match v {
357        // The `4 * NODE` is the container's own minimum allocation; each child
358        // then charges for itself, recursively. Dropping that recursion is what
359        // made an array of empty arrays look free.
360        serde_json::Value::Array(a) => 4 * NODE + a.iter().map(json_bytes).sum::<usize>(),
361        serde_json::Value::Object(o) => {
362            MAP_NODE
363                + o.iter()
364                    .map(|(k, v)| MAP_ENTRY + k.len().max(NODE / 2) + SLOT + json_bytes(v))
365                    .sum::<usize>()
366        }
367        serde_json::Value::String(s) => SLOT + s.len(),
368        // Null, bool and number are all the node and nothing else.
369        _ => SLOT,
370    }
371}
372
373/// Running byte accounting for one walk.
374pub(crate) struct ByteBudget {
375    used: usize,
376    max: usize,
377}
378
379impl ByteBudget {
380    pub(crate) fn new(max: usize) -> Self {
381        Self { used: 0, max }
382    }
383
384    /// Charge a page. `false` when the walk must stop; a refused page is NOT
385    /// charged, so `used` always describes what the caller actually kept.
386    pub(crate) fn admit(&mut self, page: &[RecordEntry]) -> bool {
387        let cost: usize = page.iter().map(approx_bytes).sum();
388        match self.used.checked_add(cost) {
389            Some(total) if total <= self.max => {
390                self.used = total;
391                true
392            }
393            _ => false,
394        }
395    }
396
397    /// Charge `bytes` that no record accounts for. Same contract as
398    /// [`Self::admit`]: `false` means stop, and a refused charge is not taken.
399    pub(crate) fn charge(&mut self, bytes: usize) -> bool {
400        match self.used.checked_add(bytes) {
401            Some(total) if total <= self.max => {
402                self.used = total;
403                true
404            }
405            _ => false,
406        }
407    }
408
409    pub(crate) fn used(&self) -> usize {
410        self.used
411    }
412
413    /// The ceiling this budget was built with.
414    pub(crate) fn max(&self) -> usize {
415        self.max
416    }
417
418    /// What is left. A transient page has to fit in this, not in the ceiling —
419    /// otherwise a walk that has already retained most of its budget can still
420    /// hold a full budget's worth of page on top of it.
421    pub(crate) fn remaining(&self) -> usize {
422        self.max.saturating_sub(self.used)
423    }
424}
425
426/// Append a page, refusing to exceed `max`.
427///
428/// **An error, never a truncation.** The caller of the live walk is
429/// `web::resolve_subscriptions`, whose result reaches `store::replace_sub_refs`
430/// — a `DELETE` followed by reinserting exactly what it was handed. A short
431/// list there is not a short list, it is revoked access to whatever fell off
432/// the end. Returning `Err` lets `resolve_subscriptions` take its documented
433/// fail-closed branch and serve the last-known projection instead.
434///
435/// `out` is left untouched on refusal, so a partial page cannot survive.
436pub(crate) fn extend_bounded(
437    out: &mut Vec<RecordEntry>,
438    page: Vec<RecordEntry>,
439    max: usize,
440    collection: &str,
441) -> Result<()> {
442    if out.len() + page.len() > max {
443        return Err(ListingTooLarge::Records {
444            collection: collection.to_string(),
445            max,
446            held: out.len(),
447            offered: page.len(),
448        }
449        .into());
450    }
451    out.extend(page);
452    Ok(())
453}
454
455/// Errors from the atproto identity + PDS layer.
456///
457/// Wraps the transport, the atproto XRPC error envelope (`{"error","message"}`),
458/// and the identity-resolution failure modes so callers can distinguish "the
459/// network broke" from "the PDS said no" from "this handle doesn't resolve".
460#[derive(Debug, thiserror::Error)]
461pub enum AtProtoError {
462    /// The underlying HTTP transport failed (DNS, TLS, timeout, connect).
463    #[error("atproto transport error: {0}")]
464    Transport(#[from] reqwest::Error),
465
466    /// The XRPC endpoint returned a non-2xx status with an atproto error
467    /// envelope (or an opaque body). `error` is the atproto error name (e.g.
468    /// `RecordNotFound`, `AuthMissing`), `message` the human string.
469    #[error("atproto XRPC error {status}: {error}{}", .message.as_deref().map(|m| format!(" — {m}")).unwrap_or_default())]
470    Xrpc {
471        /// The HTTP status code.
472        status: StatusCode,
473        /// The atproto error name (the `error` field), or `"Unknown"`.
474        error: String,
475        /// The optional human-readable `message` field.
476        message: Option<String>,
477    },
478
479    /// A handle could not be resolved to a DID.
480    #[error("could not resolve handle {handle:?} to a DID")]
481    HandleResolution {
482        /// The handle that failed to resolve.
483        handle: String,
484    },
485
486    /// A DID document could not be resolved, or lacks a usable PDS service
487    /// endpoint (`#atproto_pds`).
488    #[error("could not resolve DID {did:?} to a PDS endpoint: {reason}")]
489    DidResolution {
490        /// The DID that failed to resolve.
491        did: String,
492        /// Why resolution failed.
493        reason: String,
494        /// The same, as a value a caller can branch on without reading `reason`.
495        cause: DidResolutionCause,
496    },
497}
498
499/// Why a DID did not resolve to a PDS, structured — so a poller can file a
500/// deleted account under "the server answered" rather than "the network broke".
501#[derive(Debug, Clone, Copy, PartialEq, Eq)]
502#[non_exhaustive]
503pub enum DidResolutionCause {
504    /// A DID method this reader does not resolve.
505    UnsupportedMethod,
506    /// The DID document fetch got an answer, and it was not a success (a
507    /// tombstoned or unknown DID is a 404 from the PLC directory).
508    Status,
509    /// The DID document has no `#atproto_pds` service.
510    NoPdsEndpoint,
511    /// The PDS endpoint it names is refused by the SSRF guard.
512    NotAPublicTarget,
513}
514
515/// Whether a write failed because its `swapRecord` no longer matched — the
516/// record moved between the caller's read and its write (#149).
517///
518/// **Matched on the structured rejection, by error name.** Every client
519/// surfaces a PDS refusal as [`AtProtoError::Xrpc`]: the direct client at the
520/// root, the Rust OAuth client under a context, the sidecar client with the
521/// PDS's name carried through `/internal/repo`. `InvalidSwap` is the reference
522/// PDS's own name for a compare-and-swap mismatch and is answered 400; the
523/// name is the signal rather than the status, because the name is what says
524/// "someone else wrote this" and a 400 alone says nothing of the kind. A
525/// transport failure, or a message that merely contains the word, is not one.
526///
527/// The cause of an [`ApplyWritesIncomplete`] is walked explicitly, as
528/// `readstate::may_be_existence_mismatch` does: that wrapper's `source()`
529/// continues from its cause's SOURCE, so on the sidecar client — whose cause
530/// IS the `AtProtoError` — `err.chain()` alone would step over it.
531pub fn is_invalid_swap(err: &anyhow::Error) -> bool {
532    let wrapped = ApplyWritesIncomplete::of(err).map(|p| p.cause().chain());
533    err.chain()
534        .chain(wrapped.into_iter().flatten())
535        .any(|cause| {
536            matches!(
537                cause.downcast_ref::<AtProtoError>(),
538                Some(AtProtoError::Xrpc { error, .. }) if error == "InvalidSwap"
539            )
540        })
541}
542
543impl AtProtoError {
544    /// True when the XRPC error is a "record not found" — handy for upsert paths
545    /// that treat a missing record as "create instead of update".
546    pub fn is_record_not_found(&self) -> bool {
547        matches!(
548            self,
549            AtProtoError::Xrpc { error, .. } if error == "RecordNotFound"
550        )
551    }
552}
553
554// ---------------------------------------------------------------------------
555// Auth — the direct-PDS path (dev / tests)
556// ---------------------------------------------------------------------------
557
558/// A source of atproto access tokens.
559///
560/// This trait abstracts over token acquisition for the direct [`PdsClient`]
561/// (used by local runs and tests). A [`PdsClient`] can hold a `dyn TokenSource`
562/// instead of a static [`Auth`] without any call-site change, so a token source
563/// that refreshes out of band can be dropped in later.
564///
565/// It is async + `Send + Sync` so a background refresh can live behind it.
566#[allow(async_fn_in_trait)]
567pub trait TokenSource: Send + Sync {
568    /// Return the current bearer access token to send as `Authorization`.
569    async fn access_token(&self) -> Result<String>;
570}
571
572/// The auth material a [`PdsClient`] carries.
573///
574/// A small enum rather than a bare string, so the match stays exhaustive if a
575/// second direct-auth mechanism is added alongside app-password sessions.
576#[derive(Clone)]
577pub enum Auth {
578    /// A bearer access token from a `com.atproto.server.createSession`
579    /// (app-password) session. This is the direct-PDS auth used by local runs
580    /// and tests; the live web path authenticates via the OAuth sidecar instead
581    /// (see [`SidecarClient`]).
582    Session(SessionAuth),
583
584    /// The atproto OAuth confidential-client path is handled entirely by the
585    /// `@atproto/oauth-client` sidecar ([`SidecarClient`]), which mints, DPoP-binds,
586    /// and refreshes tokens. The direct [`PdsClient`] does not carry OAuth tokens;
587    /// this variant is a placeholder so the `Auth` enum documents that the OAuth
588    /// path lives elsewhere.
589    Oauth(OauthPlaceholder),
590
591    /// **No credentials at all** — an unauthenticated public read of a repo the
592    /// caller does not own. `com.atproto.repo.listRecords` is public on a
593    /// standard PDS, so a stranger's `community.lexicon.rss.*` records can be
594    /// read with no session; this variant makes that expressible without
595    /// inventing a fake token.
596    ///
597    /// A client holding it is **read-only**: [`Auth::bearer`] returns an error,
598    /// so every write path (`create_record` / `put_record` / `delete_record` /
599    /// `apply_writes`, all of which go through
600    /// `authed_headers`) fails closed. Construct one
601    /// via [`PdsClient::anonymous`].
602    Anonymous,
603}
604
605impl Auth {
606    /// The bearer access token to present on `com.atproto.repo.*` calls.
607    ///
608    /// Only [`Auth::Session`] carries a token (the session's `accessJwt`).
609    /// [`Auth::Oauth`] carries none — the sidecar owns the OAuth path — so it
610    /// returns an error pointing callers at [`SidecarClient`]. [`Auth::Anonymous`]
611    /// carries none by construction, which is what makes an anonymous client
612    /// read-only.
613    pub fn bearer(&self) -> Result<&str> {
614        match self {
615            Auth::Session(s) => Ok(&s.access_jwt),
616            Auth::Oauth(_) => anyhow::bail!(
617                "the direct PdsClient does not carry OAuth tokens — atproto OAuth is \
618                 handled by the @atproto/oauth-client sidecar (SidecarClient); \
619                 use Auth::Session (app-password) for the direct-PDS path"
620            ),
621            Auth::Anonymous => anyhow::bail!(
622                "this PdsClient is anonymous (unauthenticated public read) and carries no \
623                 bearer token — authenticated repo writes require Auth::Session or the \
624                 SidecarClient"
625            ),
626        }
627    }
628}
629
630/// A session obtained from `com.atproto.server.createSession` (interim
631/// app-password auth). Holds the DID + tokens + handle the server returned.
632#[derive(Clone, Debug, Deserialize)]
633pub struct SessionAuth {
634    /// The account DID this session authenticates.
635    pub did: String,
636    /// The account handle at session-creation time.
637    #[serde(default)]
638    pub handle: Option<String>,
639    /// The bearer access token presented on authed XRPC calls.
640    #[serde(rename = "accessJwt")]
641    pub access_jwt: String,
642    /// The refresh token, exchanged via `com.atproto.server.refreshSession`.
643    /// The direct-PDS refresh flow is not implemented here; the live web path
644    /// refreshes via the OAuth sidecar instead.
645    #[serde(rename = "refreshJwt", default)]
646    pub refresh_jwt: Option<String>,
647}
648
649/// Placeholder for the OAuth variant of [`Auth`].
650///
651/// Intentionally empty: the OAuth session material (DPoP key handle, token
652/// references) is held entirely by the sidecar, not by the direct [`PdsClient`].
653/// This type exists only so [`Auth::Oauth`] is a real variant and the split is
654/// visible in the type system.
655#[derive(Clone, Debug, Default)]
656#[non_exhaustive]
657pub struct OauthPlaceholder {}
658
659// ---------------------------------------------------------------------------
660// Identity resolution
661// ---------------------------------------------------------------------------
662
663/// Resolve an atproto handle to its DID.
664///
665/// Uses `com.atproto.identity.resolveHandle` against `resolver_base` (any host
666/// that implements it; [`DEFAULT_RESOLVER_HOST`] is a safe bootstrap). A fuller
667/// implementation would also try the DNS `_atproto` TXT record and the
668/// `https://<handle>/.well-known/atproto-did` fallback; the XRPC path is the
669/// common case and the one implemented here.
670pub async fn resolve_handle(client: &Client, resolver_base: &str, handle: &str) -> Result<String> {
671    // Build the query manually rather than via reqwest's `.query()` so we don't
672    // depend on the optional `query`/`url` reqwest feature (the declared feature
673    // set is rustls + gzip + json only).
674    let url = format!(
675        "{}/xrpc/com.atproto.identity.resolveHandle?handle={}",
676        resolver_base.trim_end_matches('/'),
677        urlencode(handle)
678    );
679
680    #[derive(Deserialize)]
681    struct ResolveHandleOut {
682        did: String,
683    }
684
685    // Route through the SSRF guard: `resolver_base` can be a user-influenced PDS
686    // host (from a prior DID-doc resolution), so a hostile endpoint must not be
687    // able to target loopback / link-local / metadata. Feed-privacy is NOT
688    // applied here (this is a legitimate atproto XRPC call, not a feed fetch).
689    let resp = crate::net::guarded_get_no_privacy(client, &url, &[]).await?;
690    if !resp.status().is_success() {
691        // Surface the XRPC envelope but map the common "not found" to the typed
692        // handle-resolution error so callers get a clean signal.
693        let err = xrpc_error_from(resp).await;
694        if let AtProtoError::Xrpc { status, .. } = &err {
695            if *status == StatusCode::BAD_REQUEST || *status == StatusCode::NOT_FOUND {
696                return Err(AtProtoError::HandleResolution {
697                    handle: handle.to_string(),
698                }
699                .into());
700            }
701        }
702        return Err(err.into());
703    }
704
705    // Capped: `resolver_base` can be a user-influenced PDS host, as the comment
706    // above this function's guard already says.
707    let raw = crate::net::read_capped(resp).await?;
708    let out: ResolveHandleOut =
709        serde_json::from_slice(&raw).context("parsing resolveHandle response")?;
710    Ok(out.did)
711}
712
713/// Resolve a DID to its PDS service endpoint by fetching + parsing its DID
714/// document.
715///
716/// * `did:plc:*` → the PLC directory (`{plc_directory}/{did}`).
717/// * `did:web:host` → `https://host/.well-known/did.json`.
718///
719/// The PDS endpoint is the service in the DID doc whose `id` ends with
720/// `#atproto_pds` (type `AtprotoPersonalDataServer`); its `serviceEndpoint` is
721/// the base URL for all `com.atproto.repo.*` calls.
722pub async fn resolve_did_to_pds(client: &Client, plc_directory: &str, did: &str) -> Result<String> {
723    let doc_url = if let Some(rest) = did.strip_prefix("did:web:") {
724        // did:web host may itself be percent-encoded / contain a path; the
725        // common case is a bare host.
726        let host = rest.replace(':', "/");
727        format!("https://{host}/.well-known/did.json")
728    } else if did.starts_with("did:plc:") {
729        format!("{}/{}", plc_directory.trim_end_matches('/'), did)
730    } else {
731        return Err(AtProtoError::DidResolution {
732            did: did.to_string(),
733            reason: "unsupported DID method (only did:plc and did:web are handled)".to_string(),
734            cause: DidResolutionCause::UnsupportedMethod,
735        }
736        .into());
737    };
738
739    // SSRF guard: `doc_url` is attacker-controllable for `did:web:<host>` (the
740    // host comes straight from the DID) — a hostile `did:web:169.254.169.254`
741    // or `did:web:localhost` would otherwise make the server fetch an internal
742    // target and reflect its body. Route through the IP/scheme guard (no
743    // feed-privacy layer — this is a DID document, not a feed).
744    let resp = crate::net::guarded_get_no_privacy(client, &doc_url, &[]).await?;
745    if !resp.status().is_success() {
746        return Err(AtProtoError::DidResolution {
747            did: did.to_string(),
748            reason: format!("DID document fetch returned {}", resp.status()),
749            cause: DidResolutionCause::Status,
750        }
751        .into());
752    }
753
754    // **Capped, and this is the most remote-controlled body of the lot.** For a
755    // `did:web:` the host is taken straight out of the DID, so whoever supplies
756    // the DID chooses the server — and the SSRF guard only proves the address is
757    // public, not that the body is finite.
758    let raw = crate::net::read_capped(resp).await?;
759    let doc: DidDocument = serde_json::from_slice(&raw).context("parsing DID document")?;
760    let endpoint = doc
761        .pds_endpoint()
762        .ok_or_else(|| AtProtoError::DidResolution {
763            did: did.to_string(),
764            reason: "DID document has no #atproto_pds service endpoint".to_string(),
765            cause: DidResolutionCause::NoPdsEndpoint,
766        })?;
767
768    // SSRF guard on the RESOLVED endpoint: the `serviceEndpoint` is fully
769    // attacker-controlled (it's whatever the DID document says) and is handed to
770    // XRPC clients that fetch it directly. Reject a private/loopback/metadata
771    // target here so a hostile DID doc can't point the PDS at an internal host.
772    crate::net::assert_public_target(&endpoint)
773        .await
774        .map_err(|e| AtProtoError::DidResolution {
775            did: did.to_string(),
776            reason: format!("PDS serviceEndpoint is not a public target: {e}"),
777            cause: DidResolutionCause::NotAPublicTarget,
778        })?;
779    Ok(endpoint)
780}
781
782/// The subset of a DID document FeatherReader needs: its services, so it can
783/// find the `#atproto_pds` endpoint.
784#[derive(Debug, Clone, Deserialize)]
785pub struct DidDocument {
786    /// The document subject (the DID itself).
787    #[serde(default)]
788    pub id: String,
789    /// The declared services; the PDS is the one whose `id` ends `#atproto_pds`.
790    #[serde(default)]
791    pub service: Vec<DidService>,
792}
793
794/// One service entry in a [`DidDocument`].
795#[derive(Debug, Clone, Deserialize)]
796pub struct DidService {
797    /// The service id fragment (e.g. `#atproto_pds`).
798    pub id: String,
799    /// The service type (e.g. `AtprotoPersonalDataServer`).
800    #[serde(rename = "type", default)]
801    pub r#type: String,
802    /// The service base URL.
803    #[serde(rename = "serviceEndpoint")]
804    pub service_endpoint: String,
805}
806
807impl DidDocument {
808    /// The `#atproto_pds` service endpoint, if present.
809    pub fn pds_endpoint(&self) -> Option<String> {
810        self.service
811            .iter()
812            .find(|s| s.id.ends_with("#atproto_pds"))
813            .map(|s| s.service_endpoint.trim_end_matches('/').to_string())
814    }
815}
816
817// ---------------------------------------------------------------------------
818// Direct-PDS auth: app-password session
819// ---------------------------------------------------------------------------
820
821/// Create a session with an **app password** via
822/// `com.atproto.server.createSession`.
823///
824/// This is the direct-PDS path that makes [`PdsClient`] usable without the OAuth
825/// sidecar (local runs and tests). `pds_base` is the account's PDS (resolve it
826/// first with [`resolve_handle`] + [`resolve_did_to_pds`], or pass the entryway
827/// like `https://bsky.social`, which will service-proxy). `identifier` is a
828/// handle or DID; `app_password` is an app-password (never the main password).
829///
830/// The POST goes through [`crate::net::guarded_post_json`]. This is the single
831/// most credential-dense request in the crate — the app password travels in the
832/// JSON **body**, where reqwest's cross-origin header sanitisation cannot help
833/// it — so it gets the scheme/IP allow-list, the connect pin (no second DNS
834/// resolution to rebind), and a hard refusal to follow a redirect that would
835/// re-send that body to another host.
836pub async fn login_with_app_password(
837    client: &Client,
838    pds_base: &str,
839    identifier: &str,
840    app_password: &str,
841) -> Result<SessionAuth> {
842    let url = format!(
843        "{}/xrpc/com.atproto.server.createSession",
844        pds_base.trim_end_matches('/')
845    );
846    let body = serde_json::to_vec(&json!({ "identifier": identifier, "password": app_password }))
847        .context("serializing createSession request")?;
848    let resp = crate::net::guarded_post_json(client, &url, &[], body).await?;
849    if !resp.status().is_success() {
850        return Err(xrpc_error_from(resp).await.into());
851    }
852    let raw = crate::net::read_capped(resp).await?;
853    serde_json::from_slice(&raw).context("parsing createSession response")
854}
855
856// ---------------------------------------------------------------------------
857// The PDS client
858// ---------------------------------------------------------------------------
859
860/// A lightweight client for one user's PDS repo.
861///
862/// Holds the user's DID (the repo to read/write), the PDS base URL (resolved
863/// from the DID doc), the shared `reqwest::Client`, and the [`Auth`] token.
864/// All the `com.atproto.repo.*` methods below act on `self.did`'s repo.
865///
866/// The client may also be **anonymous** ([`PdsClient::anonymous`]), in which case
867/// it is read-only: it sends no `Authorization` header and every write path
868/// errors out of [`Auth::bearer`].
869///
870/// Cheap to clone (`Arc` internals); one is held per logged-in session.
871#[derive(Clone)]
872pub struct PdsClient {
873    http: Client,
874    /// The PDS base URL, e.g. `https://pds.example.com` (no trailing slash).
875    pds_base: Arc<str>,
876    /// The repo DID all calls target.
877    did: Arc<str>,
878    /// The auth material (an app-password session bearer for the direct path, or
879    /// [`Auth::Anonymous`] for a read-only public read of a stranger's repo).
880    auth: Auth,
881}
882
883/// A single record as returned in a `listRecords` / `getRecord` response.
884///
885/// `value` is the raw record body (with its `$type`); typed wrappers
886/// deserialize it into the matching [`crate::lexicon`] struct.
887#[derive(Debug, Clone, Deserialize)]
888pub struct RecordEntry {
889    /// The `at://did/collection/rkey` strong ref to this record.
890    pub uri: String,
891    /// The record CID (content hash).
892    #[serde(default)]
893    pub cid: Option<String>,
894    /// The raw record body.
895    pub value: Value,
896}
897
898impl RecordEntry {
899    /// The record key (the last `/`-segment of the `at://` URI).
900    pub fn rkey(&self) -> Option<&str> {
901        self.uri.rsplit('/').next()
902    }
903
904    /// Deserialize this record's `value` into a typed lexicon record.
905    pub fn parse<T: DeserializeOwned>(&self) -> Result<T> {
906        serde_json::from_value(self.value.clone())
907            .with_context(|| format!("deserializing record {}", self.uri))
908    }
909}
910
911/// The `com.atproto.repo.listRecords` response envelope.
912#[derive(Debug, Clone, Deserialize)]
913pub struct ListRecordsResponse {
914    /// The page of records.
915    #[serde(default)]
916    pub records: Vec<RecordEntry>,
917    /// The opaque pagination cursor for the next page, if any.
918    #[serde(default)]
919    pub cursor: Option<String>,
920    /// Records on this page whose envelope was malformed and were left out of
921    /// `records` (#177): one bad record no longer fails the page it is on.
922    #[serde(skip)]
923    pub malformed: usize,
924    /// The page's size on the wire, where the client knows it (0 otherwise).
925    /// A walk that SKIPS malformed records charges this to its budget, since
926    /// the skipped records are invisible to the per-record accounting.
927    #[serde(skip)]
928    pub wire_bytes: usize,
929}
930
931/// **A reader's own repo holds records this server cannot read** (#177).
932///
933/// Returned by every walk whose result is written through
934/// `store::replace_sub_refs`. Skipping a record there would silently drop it
935/// from the reader's subscriptions, so the walk refuses instead, and the web
936/// layer recognises this error by type to tell the reader why.
937#[derive(Debug, Clone, PartialEq, Eq)]
938pub struct MalformedRecords {
939    /// The collection being listed.
940    pub collection: String,
941    /// How many records on the refused page were malformed.
942    pub count: usize,
943}
944
945impl std::fmt::Display for MalformedRecords {
946    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
947        write!(
948            f,
949            "{} record(s) in {} have a malformed envelope; refusing the listing rather than \
950             dropping them",
951            self.count, self.collection
952        )
953    }
954}
955
956impl std::error::Error for MalformedRecords {}
957
958/// **A `listRecords` answer arrived, and it is not a page** (#227).
959///
960/// Typed so a poller can file it by what the PDS sent rather than as "the
961/// request never produced a response", which is what a bare string fell
962/// through to. The texts are the ones the bare strings carried.
963#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)]
964pub enum UnreadableListing {
965    /// The body was empty, or had no `records` field: what a proxy makes of an
966    /// empty or unexpected upstream body. Absent is not empty.
967    #[error("listRecords returned no records field (empty or unexpected body)")]
968    NoRecords,
969    /// A 2xx carrying an atproto error envelope: the PDS said no, in the body
970    /// instead of the status.
971    #[error("PDS answered 2xx with an error envelope: {error}{}", .message.as_deref().map(|m| format!(" — {m}")).unwrap_or_default())]
972    ErrorEnvelope {
973        /// The envelope's `error` name.
974        error: String,
975        /// Its `message`, already truncated for a log line.
976        message: Option<String>,
977    },
978}
979
980/// **A walk refused to read further because the listing is too big** (#227):
981/// more pages, bytes or records than it allows. Refused rather than truncated,
982/// for the reason `extend_bounded` gives.
983#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)]
984pub enum ListingTooLarge {
985    /// The page cap ran out with the PDS still offering a cursor.
986    #[error(
987        "listRecords for {collection} did not finish within {pages} pages \
988         ({held} held, and the PDS still offered more) — refusing a short list"
989    )]
990    Pages {
991        /// The collection being listed.
992        collection: String,
993        /// The walk's page cap.
994        pages: usize,
995        /// Records held when it ran out.
996        held: usize,
997    },
998    /// The walk's byte budget would be exceeded by the next page.
999    #[error(
1000        "listRecords for {collection} exceeded the {max_bytes}-byte cap \
1001         ({held} held, {charged} bytes charged) — refusing to accumulate further"
1002    )]
1003    Bytes {
1004        /// The collection being listed.
1005        collection: String,
1006        /// The budget's ceiling.
1007        max_bytes: usize,
1008        /// Records held when it was refused.
1009        held: usize,
1010        /// Bytes charged to the budget so far.
1011        charged: usize,
1012    },
1013    /// The byte budget ran out on pages that skipped malformed records.
1014    #[error(
1015        "listRecords for {collection} exceeded the {max_bytes}-byte cap on pages \
1016         of malformed records — refusing to read further"
1017    )]
1018    MalformedBytes {
1019        /// The collection being listed.
1020        collection: String,
1021        /// The budget's ceiling.
1022        max_bytes: usize,
1023    },
1024    /// The record cap would be exceeded by the next page.
1025    #[error(
1026        "listRecords for {collection} exceeded the {max}-record cap \
1027         ({held} held, {offered} more offered) — refusing to accumulate further"
1028    )]
1029    Records {
1030        /// The collection being listed.
1031        collection: String,
1032        /// The record cap.
1033        max: usize,
1034        /// Records held.
1035        held: usize,
1036        /// Records the refused page offered.
1037        offered: usize,
1038    },
1039}
1040
1041/// The `com.atproto.repo.createRecord` / `putRecord` response (a strong ref to
1042/// the written record).
1043#[derive(Debug, Clone, Deserialize)]
1044pub struct WriteResult {
1045    /// The `at://` URI of the written record.
1046    pub uri: String,
1047    /// The record CID after the write.
1048    #[serde(default)]
1049    pub cid: Option<String>,
1050}
1051
1052impl WriteResult {
1053    /// The record key — the last `/`-segment of the `at://` URI.
1054    ///
1055    /// The reader-facing `add_*` wrappers return this so the web layer can
1056    /// address the freshly-created record (delete/rename) without a re-list.
1057    pub fn rkey(&self) -> Option<&str> {
1058        self.uri.rsplit('/').next()
1059    }
1060
1061    /// The record key as an owned `String`, or the empty string if the URI is
1062    /// somehow segment-less (never in practice — a PDS always returns an
1063    /// `at://did/collection/rkey`). Convenience for the `-> rkey` wrappers.
1064    pub fn into_rkey(self) -> String {
1065        self.rkey().unwrap_or_default().to_string()
1066    }
1067}
1068
1069impl PdsClient {
1070    /// Construct a client against an already-resolved PDS base + DID + auth.
1071    pub fn new(
1072        http: Client,
1073        pds_base: impl Into<String>,
1074        did: impl Into<String>,
1075        auth: Auth,
1076    ) -> Self {
1077        Self {
1078            http,
1079            pds_base: Arc::from(pds_base.into().trim_end_matches('/')),
1080            did: Arc::from(did.into()),
1081            auth,
1082        }
1083    }
1084
1085    /// Construct a **read-only, unauthenticated** client for a public repo the
1086    /// caller does not own — `com.atproto.repo.listRecords` is public on a
1087    /// standard PDS, so a stranger's records need no credentials.
1088    ///
1089    /// Every `com.atproto.repo.*` **write** returns an error (there is no bearer;
1090    /// see [`Auth::Anonymous`]). Callers are expected to have obtained `pds_base`
1091    /// from [`resolve_did_to_pds`], which already runs
1092    /// [`crate::net::assert_public_target`] on the resolved `serviceEndpoint` —
1093    /// but that is not what makes the fetch safe: every read is **re-vetted at
1094    /// fetch time** by [`crate::net::guarded_get_no_privacy`], which closes the
1095    /// DNS-rebinding window between resolve and connect. This constructor is
1096    /// deliberately synchronous and does no validation of its own, so the
1097    /// authoritative check is not duplicated (or, worse, mistaken for sufficient).
1098    pub fn anonymous(http: Client, pds_base: impl Into<String>, did: impl Into<String>) -> Self {
1099        Self::new(http, pds_base, did, Auth::Anonymous)
1100    }
1101
1102    /// Resolve `handle` → DID → PDS, obtain an app-password session, and build a
1103    /// ready-to-use client. A convenience constructor for the direct-PDS path
1104    /// that exercises the whole stack end-to-end.
1105    ///
1106    /// `resolver_base` / `plc_directory` default to [`DEFAULT_RESOLVER_HOST`] /
1107    /// [`DEFAULT_PLC_DIRECTORY`] when passed `None`.
1108    pub async fn login(
1109        http: Client,
1110        handle: &str,
1111        app_password: &str,
1112        resolver_base: Option<&str>,
1113        plc_directory: Option<&str>,
1114    ) -> Result<Self> {
1115        let resolver = resolver_base.unwrap_or(DEFAULT_RESOLVER_HOST);
1116        let plc = plc_directory.unwrap_or(DEFAULT_PLC_DIRECTORY);
1117
1118        let did = resolve_handle(&http, resolver, handle).await?;
1119        let pds_base = resolve_did_to_pds(&http, plc, &did).await?;
1120        let session = login_with_app_password(&http, &pds_base, &did, app_password).await?;
1121
1122        Ok(Self::new(
1123            http,
1124            pds_base,
1125            session.did.clone(),
1126            Auth::Session(session),
1127        ))
1128    }
1129
1130    /// The repo DID this client targets.
1131    pub fn did(&self) -> &str {
1132        &self.did
1133    }
1134
1135    /// The PDS base URL this client talks to.
1136    pub fn pds_base(&self) -> &str {
1137        &self.pds_base
1138    }
1139
1140    /// Build the `Authorization: Bearer …` header pair for an authed write.
1141    ///
1142    /// A `Vec` of pairs rather than a [`HeaderMap`] because every write now goes
1143    /// through [`crate::net::guarded_post_json`], which takes header pairs and
1144    /// sets `Content-Type: application/json` itself. Fails closed on
1145    /// [`Auth::Anonymous`] (there is no bearer), which is what makes an anonymous
1146    /// client read-only.
1147    fn authed_headers(&self) -> Result<Vec<(HeaderName, HeaderValue)>> {
1148        let bearer = self.auth.bearer()?;
1149        let mut value = HeaderValue::from_str(&format!("Bearer {bearer}"))
1150            .context("building Authorization header")?;
1151        value.set_sensitive(true);
1152        Ok(vec![(AUTHORIZATION, value)])
1153    }
1154
1155    fn xrpc_url(&self, method: &str) -> String {
1156        format!("{}/xrpc/{}", self.pds_base, method)
1157    }
1158
1159    // -- com.atproto.repo.* --------------------------------------------------
1160
1161    /// `com.atproto.repo.listRecords` — one page of a collection's records.
1162    ///
1163    /// `cursor` continues a previous page; `limit` caps the page (atproto's max
1164    /// is 100). Use [`list_all_records`](Self::list_all_records) to page fully.
1165    ///
1166    /// The fetch is routed through [`crate::net::guarded_get_no_privacy`] — the
1167    /// same per-hop scheme/IP allow-list and connect-pinning the feed poller and
1168    /// the identity-resolution paths use. `pds_base` was vetted by
1169    /// [`crate::net::assert_public_target`] at resolve time, but that is a
1170    /// *separate* DNS resolution from this fetch; routing the request through the
1171    /// guard closes the rebinding window, which matters as soon as the repo (and
1172    /// therefore the host) is chosen by a stranger. The response body is read via
1173    /// [`crate::net::read_capped`] so a hostile PDS cannot stream an unbounded
1174    /// body at a 512 MB box.
1175    pub async fn list_records(
1176        &self,
1177        collection: &str,
1178        limit: Option<u32>,
1179        cursor: Option<&str>,
1180    ) -> Result<ListRecordsResponse> {
1181        // Build the query manually (see `resolve_handle`): no reqwest `query`
1182        // feature dependency.
1183        let mut url = format!(
1184            "{}?repo={}&collection={}",
1185            self.xrpc_url("com.atproto.repo.listRecords"),
1186            urlencode(&self.did),
1187            urlencode(collection),
1188        );
1189        if let Some(limit) = limit {
1190            url.push_str(&format!("&limit={limit}"));
1191        }
1192        if let Some(cursor) = cursor {
1193            url.push_str(&format!("&cursor={}", urlencode(cursor)));
1194        }
1195
1196        // listRecords is public/unauthenticated on most PDSes, but we send the
1197        // bearer when we have a session one so private repos work too. An
1198        // Auth::Oauth / Auth::Anonymous client sends no Authorization header at
1199        // all. The guard drops the header if a redirect leaves this PDS's origin.
1200        let mut headers: Vec<(reqwest::header::HeaderName, HeaderValue)> = Vec::new();
1201        if let Auth::Session(s) = &self.auth {
1202            let mut value = HeaderValue::from_str(&format!("Bearer {}", s.access_jwt))
1203                .context("building Authorization header")?;
1204            value.set_sensitive(true);
1205            headers.push((AUTHORIZATION, value));
1206        }
1207        let resp = crate::net::guarded_get_no_privacy(&self.http, &url, &headers).await?;
1208        if !resp.status().is_success() {
1209            return Err(xrpc_error_from(resp).await.into());
1210        }
1211        let body = crate::net::read_capped(resp).await?;
1212        parse_list_records(&body)
1213    }
1214
1215    /// Page through **all** records in a collection, following the cursor until
1216    /// exhausted. Convenience over [`list_records`](Self::list_records) for the
1217    /// login-time "load the whole follow-list" read.
1218    ///
1219    /// Bounded by `MAX_LIST_PAGES` and by cursor-repetition detection, because
1220    /// `pds_base` may be a host we did not choose (see [`PdsClient::anonymous`]).
1221    /// Both are refusals, not short answers: a repeated cursor on a non-empty
1222    /// page is an `Err` (#203), since the result may reach `replace_sub_refs`.
1223    pub async fn list_all_records(&self, collection: &str) -> Result<Vec<RecordEntry>> {
1224        self.list_all_records_within(collection, &mut ByteBudget::new(MAX_LIST_BYTES))
1225            .await
1226    }
1227
1228    /// [`list_all_records`](Self::list_all_records) against a caller's budget.
1229    ///
1230    /// **Shared, not per-walk.** Two walks that nest — a publication read runs a
1231    /// second walk while still holding the first's records — each had their own
1232    /// ceiling, so the process could hold twice it. Passing one budget in makes
1233    /// the bound a property of the caller's whole read, which is the thing that
1234    /// has to fit in the box, and the type enforces it where a comment would not.
1235    pub(crate) async fn list_all_records_within(
1236        &self,
1237        collection: &str,
1238        budget: &mut ByteBudget,
1239    ) -> Result<Vec<RecordEntry>> {
1240        self.walk_all_within(collection, budget, OnMalformed::Refuse)
1241            .await
1242            .map(|(records, _)| records)
1243    }
1244
1245    /// [`list_all_records_within`](Self::list_all_records_within) for a
1246    /// **stranger's** repo: a malformed record is skipped and counted instead of
1247    /// refusing the walk (#177). Never for a walk that reaches
1248    /// `replace_sub_refs`, where a skipped record is a dropped subscription.
1249    pub(crate) async fn list_all_records_skipping_within(
1250        &self,
1251        collection: &str,
1252        budget: &mut ByteBudget,
1253    ) -> Result<(Vec<RecordEntry>, usize)> {
1254        self.walk_all_within(collection, budget, OnMalformed::Skip)
1255            .await
1256    }
1257
1258    async fn walk_all_within(
1259        &self,
1260        collection: &str,
1261        budget: &mut ByteBudget,
1262        on_malformed: OnMalformed,
1263    ) -> Result<(Vec<RecordEntry>, usize)> {
1264        let mut out = Vec::new();
1265        let max_bytes = budget.max();
1266        let mut cursor: Option<String> = None;
1267        let mut more_offered = false;
1268        let mut malformed = 0usize;
1269        for _ in 0..MAX_LIST_PAGES {
1270            let page = self
1271                .list_records(collection, Some(100), cursor.as_deref())
1272                .await?;
1273            if on_malformed == OnMalformed::Refuse {
1274                refuse_malformed(&page, collection)?;
1275            }
1276            // **Skipped records still cost what they cost.** The per-record
1277            // accounting below never sees them, so without this a repo serving
1278            // pages of junk walks every page MAX_LIST_PAGES allows — gigabytes —
1279            // under a budget meant to stop it (found in review). The whole page
1280            // is charged: conservative, and only on pages that skipped something.
1281            // Charged ONCE, at the larger of what crossed the wire and what the
1282            // kept records retain: a parsed record can hold up to 42x its wire
1283            // size, so the wire alone under-charges (third review of #224).
1284            if page.malformed > 0
1285                && !budget.charge(
1286                    page.wire_bytes
1287                        .max(page.records.iter().map(approx_bytes).sum()),
1288                )
1289            {
1290                return Err(ListingTooLarge::MalformedBytes {
1291                    collection: collection.to_string(),
1292                    max_bytes,
1293                }
1294                .into());
1295            }
1296            malformed += page.malformed;
1297            // Skipped records count toward "the page had something", so a page
1298            // of nothing BUT malformed records does not end the walk early.
1299            let got = page.records.len() + page.malformed;
1300            // **Refused, not truncated**, for the reason `extend_bounded`
1301            // gives: this walk feeds `replace_sub_refs`, where a short list is
1302            // revoked access.
1303            // A page that skipped records was already charged its whole wire
1304            // size above, which covers its good records too; charging them again
1305            // failed walks that fit (found in review).
1306            if page.malformed == 0 && !budget.admit(&page.records) {
1307                return Err(ListingTooLarge::Bytes {
1308                    collection: collection.to_string(),
1309                    max_bytes,
1310                    held: out.len(),
1311                    charged: budget.used(),
1312                }
1313                .into());
1314            }
1315            extend_bounded(&mut out, page.records, MAX_LIST_RECORDS, collection)?;
1316            match page.cursor {
1317                // **A repeated cursor on a non-empty page is refused (#203).**
1318                // Following it loops forever; stopping with what we hold hands
1319                // `replace_sub_refs` a list the server itself said was
1320                // unfinished, which deletes everything past it.
1321                Some(next) if got > 0 && Some(&next) == cursor.as_ref() => {
1322                    anyhow::bail!(
1323                        "listRecords for {collection} returned a repeated cursor on a \
1324                         non-empty page ({} held) — refusing a list the server did not finish",
1325                        out.len(),
1326                    );
1327                }
1328                // A cursor on an EMPTY page ends the walk: a real PDS (this
1329                // project's own) returns one alongside its last page.
1330                Some(next) if got > 0 => {
1331                    cursor = Some(next);
1332                    more_offered = true;
1333                }
1334                _ => {
1335                    more_offered = false;
1336                    break;
1337                }
1338            }
1339        }
1340        // **Running out of pages is a refusal, not a short answer.** Falling out
1341        // of the loop used to return `Ok(out)`, so a repo bigger than the page
1342        // budget produced a truncated list indistinguishable from a complete
1343        // one — and `resolve_subscriptions` needs an `Err` for its fail-closed
1344        // branch. Given `Ok`, it hands the short list to `replace_sub_refs`,
1345        // which DELETEs the reader's whole `sub_ref` projection and reinserts
1346        // only what it was given. `extend_bounded` cannot catch this either:
1347        // `MAX_LIST_PAGES` x the 100 we request is `MAX_LIST_RECORDS`, so the
1348        // page budget runs out first.
1349        //
1350        // **The cap is on REQUESTS, so where it bites in RECORDS is the server's
1351        // choice and not ours.** We ask for 100 a page; a PDS MAY answer with
1352        // fewer, and only one that honours the limit puts the boundary anywhere
1353        // near `MAX_LIST_PAGES` x 100. Halve the page size and the same budget
1354        // reaches half as many records; a server that returns MORE than asked
1355        // trips `extend_bounded` first, which is the case the sentence above does
1356        // not cover. Said this way because an earlier version of this comment
1357        // named a fixed record window as though our own constants decided it.
1358        //
1359        // **And at the boundary the refusal is a FALSE one.** Terminating costs
1360        // one extra request, because a short page can still carry a cursor — this
1361        // project's own PDS does exactly that — so a walk that fills its last
1362        // allowed page is holding every record it was ever going to hold and
1363        // refuses anyway, on the strength of a cursor it never followed. With
1364        // `limit=100` honoured that window is a repo of roughly 19 901 to 20 000
1365        // records. The direction is safe and the alternative is deleting feeds,
1366        // but it is a false refusal and not a clean boundary.
1367        if more_offered {
1368            return Err(ListingTooLarge::Pages {
1369                collection: collection.to_string(),
1370                pages: MAX_LIST_PAGES,
1371                held: out.len(),
1372            }
1373            .into());
1374        }
1375        Ok((out, malformed))
1376    }
1377
1378    /// See [`RecordWalk`].
1379    /// The most recent records of a collection that the caller **keeps**,
1380    /// truncating rather than refusing.
1381    ///
1382    /// **The cap counts kept records, not walked ones.** Applying it to the
1383    /// raw collection starves a caller whose filter is selective: a quiet
1384    /// standard.site publication in a repo whose busy sibling fills the
1385    /// window returns nothing at all, permanently, and worse with every post
1386    /// the sibling makes. `MAX_LIST_PAGES` still bounds the request count, so
1387    /// a filter that matches nothing costs a fixed number of round trips.
1388    ///
1389    /// Truncating, not refusing, because this is an additive read: see
1390    /// `extend_truncating` for why the `extend_bounded` refusal would be
1391    /// strictly worse here.
1392    ///
1393    /// `page_size` is the caller's, because the right page depends on how big
1394    /// the records are: [`crate::net::read_capped`] bounds a response at 8 MB,
1395    /// so 100 long-form articles per page can exceed it and fail the whole
1396    /// walk.
1397    ///
1398    /// **Ordering is the PDS's**: `listRecords` is descending by *rkey*, which
1399    /// is newest-first only when rkeys are TIDs. For a publisher using slug
1400    /// rkeys the truncation keeps a lexicographic subset rather than a recent
1401    /// one — acceptable because the cap is now per-publication rather than
1402    /// per-repo, so reaching it at all means an archive larger than this
1403    /// reader stores.
1404    pub async fn list_recent_matching(
1405        &self,
1406        collection: &str,
1407        max_records: usize,
1408        page_size: u32,
1409        keep: impl FnMut(&RecordEntry) -> bool,
1410    ) -> Result<RecordWalk> {
1411        self.list_recent_matching_within(
1412            collection,
1413            max_records,
1414            &mut ByteBudget::new(MAX_LIST_BYTES),
1415            page_size,
1416            keep,
1417        )
1418        .await
1419    }
1420
1421    /// [`list_recent_matching`](Self::list_recent_matching) against a caller's
1422    /// budget. See [`list_all_records_within`](Self::list_all_records_within) for
1423    /// why it is the caller's and not the walk's.
1424    pub(crate) async fn list_recent_matching_within(
1425        &self,
1426        collection: &str,
1427        max_records: usize,
1428        budget: &mut ByteBudget,
1429        page_size: u32,
1430        mut keep: impl FnMut(&RecordEntry) -> bool,
1431    ) -> Result<RecordWalk> {
1432        let mut out = Vec::new();
1433        let mut cursor: Option<String> = None;
1434        let mut malformed = 0usize;
1435        // Every exit carries the skipped count, so none can forget it.
1436        let walk = |mut w: RecordWalk, malformed: usize| {
1437            w.malformed = malformed;
1438            w
1439        };
1440        for _ in 0..MAX_LIST_PAGES {
1441            let page = self
1442                .list_records(collection, Some(page_size), cursor.as_deref())
1443                .await?;
1444            // **Skipped, not refused**: this reads a stranger's collection, and
1445            // one bad record must not stall everything beside it (#177). Counted
1446            // in `got` too, so a page whose only records were malformed is not
1447            // mistaken for the end of the collection.
1448            // As in `walk_all_within`: skipped records are invisible to the
1449            // per-record charge, so their page pays for them here.
1450            let wire_charged = page.malformed > 0;
1451            malformed += page.malformed;
1452            let got = page.records.len() + page.malformed;
1453            // Is there a next page that is actually new? (A PDS may echo a
1454            // cursor with an empty page, or hand back the same one forever.)
1455            let more =
1456                matches!(&page.cursor, Some(next) if got > 0 && Some(next) != cursor.as_ref());
1457            // A repeated cursor on a non-empty page is not followed, but it is
1458            // not the end of the collection either: the server said it had more.
1459            // This walk keeps what it read (a stranger's repo, never
1460            // `replace_sub_refs`) and reports itself partial (#203).
1461            let cut_off =
1462                matches!(&page.cursor, Some(next) if got > 0 && Some(next) == cursor.as_ref());
1463            // **The transient page needs its own bound.** The running total
1464            // charges what is KEPT, because that is what the walk retains and a
1465            // page is dropped after the filter. But transient is not free, and
1466            // `net::read_capped`'s 8 MB bounds the WIRE — the whole point of
1467            // this budget is that wire size and retained size are not the same
1468            // number. A page of records the filter rejects entirely charges
1469            // nothing against the total and can still hold hundreds of
1470            // megabytes, so no single page may exceed the walk's budget alone.
1471            // On a page that skipped records, the transient is the larger of
1472            // its wire size and its parsed records: the skipped ones are
1473            // invisible to the per-record sum.
1474            let mut page_cost: usize = page.records.iter().map(approx_bytes).sum();
1475            if wire_charged {
1476                page_cost = page_cost.max(page.wire_bytes);
1477            }
1478            if page_cost > budget.remaining() {
1479                let mut w = RecordWalk::partial(out);
1480                w.out_of_budget = true;
1481                return Ok(walk(w, malformed));
1482            }
1483            let kept: Vec<RecordEntry> = page.records.into_iter().filter(|r| keep(r)).collect();
1484            // Charged on what is KEPT, which is what this walk retains. **This
1485            // cannot refuse**, and saying so matters: the page check above
1486            // already proved the whole page fits in the remainder, and `kept` is
1487            // a subset of it. Written as `if !admit(…) { return }` it reads as a
1488            // second stopping rule, and a reader would look for the case that
1489            // trips it. There isn't one — the call is the bookkeeping.
1490            // A page that skipped records also pays for the skipped bytes,
1491            // once: the larger of its wire size and what it keeps.
1492            let charged = if wire_charged {
1493                budget.charge(page.wire_bytes.max(kept.iter().map(approx_bytes).sum()))
1494            } else {
1495                budget.admit(&kept)
1496            };
1497            debug_assert!(charged, "the page charge already proved this fits");
1498            if extend_truncating(&mut out, kept, max_records) {
1499                return Ok(walk(RecordWalk::partial(out), malformed));
1500            }
1501            if out.len() >= max_records {
1502                // Landing exactly on the cap is only a truncation if the
1503                // collection had more to give — `extend_truncating` cannot see
1504                // that, so the caller's "incomplete" signal is decided here.
1505                return Ok(RecordWalk {
1506                    complete: !more && !cut_off,
1507                    records: out,
1508                    malformed,
1509                    out_of_budget: false,
1510                });
1511            }
1512            if cut_off {
1513                return Ok(walk(RecordWalk::partial(out), malformed));
1514            }
1515            if !more {
1516                return Ok(walk(RecordWalk::complete(out), malformed));
1517            }
1518            cursor = page.cursor;
1519        }
1520        // **The page budget ran out with the collection still going.** Silence
1521        // here reintroduces the starvation this function exists to prevent, one
1522        // order of magnitude further out: a quiet publication in a repo whose
1523        // busy sibling has more records than MAX_LIST_PAGES × page_size can
1524        // reach returns nothing at all, forever, having spent every round trip
1525        // to find out. The caller is told so it can say which feed.
1526        Ok(walk(RecordWalk::partial(out), malformed))
1527    }
1528
1529    /// `com.atproto.repo.createRecord` — create a new record (server assigns the
1530    /// rkey, `key: tid`). Returns the written record's strong ref.
1531    /// **Private, not `pub` — and not `pub(crate)`.** This is generic over
1532    /// `T: Serialize`, so it will happily write a raw `lexicon::Subscription`:
1533    /// the general case of the hole `create_subscriptions_batch` was one
1534    /// instance of. The vetted wrappers in this `impl` are the sanctioned entry
1535    /// points. `pub(crate)` was tried first and stops nothing that matters — a
1536    /// handler in `web.rs` is in this crate. Private is what makes the wrappers
1537    /// a fact rather than a convention, and it costs nothing: nothing outside
1538    /// this module ever called it.
1539    async fn create_record<T: Serialize>(
1540        &self,
1541        collection: &str,
1542        record: &T,
1543    ) -> Result<WriteResult> {
1544        let body = json!({
1545            "repo": self.did.as_ref(),
1546            "collection": collection,
1547            "record": record,
1548        });
1549        self.repo_write("com.atproto.repo.createRecord", body).await
1550    }
1551
1552    /// `com.atproto.repo.putRecord` — upsert a record at a **known** rkey
1553    /// (`key: any`). This is the `readState` upsert primitive: a feed-derived
1554    /// rkey makes the write idempotent (one record per feed).
1555    /// **Private, not `pub` — and not `pub(crate)`.** This is generic over
1556    /// `T: Serialize`, so it will happily write a raw `lexicon::Subscription`:
1557    /// the general case of the hole `create_subscriptions_batch` was one
1558    /// instance of. The vetted wrappers in this `impl` are the sanctioned entry
1559    /// points. `pub(crate)` was tried first and stops nothing that matters — a
1560    /// handler in `web.rs` is in this crate. Private is what makes the wrappers
1561    /// a fact rather than a convention, and it costs nothing: nothing outside
1562    /// this module ever called it.
1563    ///
1564    /// `swap_record` is the CID the caller read the record at, sent as
1565    /// `swapRecord`: the PDS then refuses the write with `InvalidSwap` (see
1566    /// [`is_invalid_swap`]) if the record has moved since, rather than
1567    /// silently overwriting another client's change (#149). `None` omits the
1568    /// field — an unconditional write, which is what a write that read nothing
1569    /// means.
1570    async fn put_record<T: Serialize>(
1571        &self,
1572        collection: &str,
1573        rkey: &str,
1574        record: &T,
1575        swap_record: Option<&str>,
1576    ) -> Result<WriteResult> {
1577        let mut body = json!({
1578            "repo": self.did.as_ref(),
1579            "collection": collection,
1580            "rkey": rkey,
1581            "record": record,
1582        });
1583        if let Some(cid) = swap_record {
1584            body["swapRecord"] = json!(cid);
1585        }
1586        self.repo_write("com.atproto.repo.putRecord", body).await
1587    }
1588
1589    /// The single outbound path for every authenticated `com.atproto.repo.*`
1590    /// **write**, routed through [`crate::net::guarded_post_json`].
1591    ///
1592    /// Reads were hardened first (see [`list_records`](Self::list_records)), but
1593    /// the argument applies with more force here: `pds_base` is vetted by
1594    /// [`crate::net::assert_public_target`] at *resolve* time, and the write is a
1595    /// *separate* DNS resolution — the rebinding window `net.rs` exists to close.
1596    /// A write also carries the session bearer and, in
1597    /// [`login_with_app_password`], the app password itself, so the guard's
1598    /// refusal to follow redirects (a `307` re-sends the body verbatim to the new
1599    /// host) is doing real work and not just symmetry.
1600    async fn guarded_post(&self, url: &str, body: &Value) -> Result<reqwest::Response> {
1601        let headers = self.authed_headers()?;
1602        let payload = serde_json::to_vec(body).context("serializing XRPC request body")?;
1603        crate::net::guarded_post_json(&self.http, url, &headers, payload).await
1604    }
1605
1606    /// `com.atproto.repo.deleteRecord` — delete a record by collection + rkey
1607    /// (e.g. unsubscribe → delete the subscription record).
1608    pub async fn delete_record(&self, collection: &str, rkey: &str) -> Result<()> {
1609        let url = self.xrpc_url("com.atproto.repo.deleteRecord");
1610        let body = json!({
1611            "repo": self.did.as_ref(),
1612            "collection": collection,
1613            "rkey": rkey,
1614        });
1615        let resp = self.guarded_post(&url, &body).await?;
1616        if !resp.status().is_success() {
1617            return Err(xrpc_error_from(resp).await.into());
1618        }
1619        Ok(())
1620    }
1621
1622    /// `com.atproto.repo.applyWrites` — a **batch** of create/update/delete
1623    /// operations in one atomic-per-repo round-trip.
1624    ///
1625    /// This is the read-state flusher's workhorse: dozens of dirty per-feed
1626    /// [`ReadState`] cursors coalesce into one call rather than one `putRecord`
1627    /// each. See [`flush_read_states`](Self::flush_read_states).
1628    /// **Private, not `pub` — and not `pub(crate)`.** This is generic over
1629    /// `T: Serialize`, so it will happily write a raw `lexicon::Subscription`:
1630    /// the general case of the hole `create_subscriptions_batch` was one
1631    /// instance of. The vetted wrappers in this `impl` are the sanctioned entry
1632    /// points. `pub(crate)` was tried first and stops nothing that matters — a
1633    /// handler in `web.rs` is in this crate. Private is what makes the wrappers
1634    /// a fact rather than a convention, and it costs nothing: nothing outside
1635    /// this module ever called it.
1636    ///
1637    /// Sent in chunks within the PDS's limits — see [`apply_writes_chunked`]
1638    /// for what a failure part-way means. This used to send any batch as one
1639    /// call, including an empty one; an empty batch now sends nothing, as the
1640    /// other two clients already did.
1641    async fn apply_writes(&self, writes: &[WriteOp]) -> Result<()> {
1642        apply_writes_chunked(writes, |chunk| self.apply_writes_once(chunk)).await
1643    }
1644
1645    /// One `applyWrites` call, unchunked. Reached only through
1646    /// [`apply_writes`](Self::apply_writes).
1647    async fn apply_writes_once(&self, writes: &[WriteOp]) -> Result<()> {
1648        let url = self.xrpc_url("com.atproto.repo.applyWrites");
1649        let ops: Vec<Value> = writes.iter().map(WriteOp::to_json).collect();
1650        let body = json!({
1651            "repo": self.did.as_ref(),
1652            "writes": ops,
1653        });
1654        let resp = self.guarded_post(&url, &body).await?;
1655        if !resp.status().is_success() {
1656            return Err(xrpc_error_from(resp).await.into());
1657        }
1658        Ok(())
1659    }
1660
1661    /// Shared create/put path (both return a `{uri,cid}` strong ref).
1662    async fn repo_write(&self, method: &str, body: Value) -> Result<WriteResult> {
1663        let url = self.xrpc_url(method);
1664        let resp = self.guarded_post(&url, &body).await?;
1665        if !resp.status().is_success() {
1666            return Err(xrpc_error_from(resp).await.into());
1667        }
1668        // `read_capped` rather than `resp.json()`: a hostile PDS must not be able
1669        // to stream an unbounded body at a 512 MB box (same rule as the reads).
1670        let raw = crate::net::read_capped(resp).await?;
1671        serde_json::from_slice(&raw).with_context(|| format!("parsing {method} response"))
1672    }
1673
1674    // -- typed lexicon wrappers ---------------------------------------------
1675
1676    /// List every [`Subscription`] record in the user's repo (paged fully). The
1677    /// login-time "what does this user follow?" read.
1678    pub async fn list_subscriptions(&self) -> Result<Vec<(String, Subscription)>> {
1679        self.list_typed(lexicon::nsid::SUBSCRIPTION).await
1680    }
1681
1682    /// Every [`Subscription`] with the CID it was listed at (#149).
1683    pub async fn list_subscriptions_with_cids(
1684        &self,
1685    ) -> Result<Vec<(String, Option<String>, Subscription)>> {
1686        self.list_typed_with_cids(lexicon::nsid::SUBSCRIPTION).await
1687    }
1688
1689    /// Create a [`Subscription`] record (subscribe to a feed).
1690    pub async fn create_subscription(
1691        &self,
1692        sub: &crate::vetted::VettedSubscription,
1693    ) -> Result<WriteResult> {
1694        self.create_record(lexicon::nsid::SUBSCRIPTION, sub).await
1695    }
1696
1697    /// List every [`Folder`] record in the user's repo.
1698    pub async fn list_folders(&self) -> Result<Vec<(String, Folder)>> {
1699        self.list_typed(lexicon::nsid::FOLDER).await
1700    }
1701
1702    /// Every [`Folder`] with the CID it was listed at (#268).
1703    pub async fn list_folders_with_cids(&self) -> Result<Vec<(String, Option<String>, Folder)>> {
1704        self.list_typed_with_cids(lexicon::nsid::FOLDER).await
1705    }
1706
1707    /// Create a [`Folder`] record.
1708    pub async fn create_folder(&self, folder: &Folder) -> Result<WriteResult> {
1709        self.create_record(lexicon::nsid::FOLDER, folder).await
1710    }
1711
1712    /// List every [`Saved`] (starred) record in the user's repo.
1713    pub async fn list_saved(&self) -> Result<Vec<(String, Saved)>> {
1714        self.list_typed(lexicon::nsid::SAVED).await
1715    }
1716
1717    /// Create a [`Saved`] record (star an article).
1718    pub async fn create_saved(&self, saved: &crate::vetted::VettedSaved) -> Result<WriteResult> {
1719        self.create_record(lexicon::nsid::SAVED, saved).await
1720    }
1721
1722    /// List every [`ReadState`] cursor in the user's repo (the read side a
1723    /// login-time read-state merge would consume).
1724    pub async fn list_read_states(&self) -> Result<Vec<(String, ReadState)>> {
1725        self.list_typed(lexicon::nsid::READ_STATE).await
1726    }
1727
1728    /// Upsert a single [`ReadState`] cursor at its feed-derived rkey. For a
1729    /// batch of dirty cursors prefer [`flush_read_states`](Self::flush_read_states).
1730    pub async fn put_read_state(&self, rkey: &str, state: &ReadState) -> Result<WriteResult> {
1731        self.put_record(lexicon::nsid::READ_STATE, rkey, state, None)
1732            .await
1733    }
1734
1735    /// Batch-flush many dirty [`ReadState`] cursors via `applyWrites` (chunked) —
1736    /// the debounced read-state flusher's coalesced write.
1737    ///
1738    /// Each `(rkey, state, pds_created)` becomes a `create` op at the feed-derived
1739    /// rkey when the record does not yet exist, and an `update` when it does — so a
1740    /// feed's FIRST flush succeeds (an `#update` on a missing record errors, and
1741    /// `applyWrites` is atomic per-repo). Both kinds ride the same batch.
1742    pub async fn flush_read_states(&self, cursors: &[(String, ReadState, bool)]) -> Result<()> {
1743        if cursors.is_empty() {
1744            return Ok(());
1745        }
1746        let writes = read_state_write_ops(cursors)?;
1747        self.apply_writes(&writes).await
1748    }
1749
1750    /// List a collection and parse each record's value into `T`, pairing it with
1751    /// its rkey. Records that fail to deserialize are skipped with a warning
1752    /// (forward-compat: a future writer's extra fields shouldn't break login).
1753    async fn list_typed<T: DeserializeOwned>(&self, collection: &str) -> Result<Vec<(String, T)>> {
1754        Ok(self
1755            .list_typed_with_cids(collection)
1756            .await?
1757            .into_iter()
1758            .map(|(rkey, _cid, value)| (rkey, value))
1759            .collect())
1760    }
1761
1762    /// [`list_typed`](Self::list_typed), keeping each record's CID (#149).
1763    async fn list_typed_with_cids<T: DeserializeOwned>(
1764        &self,
1765        collection: &str,
1766    ) -> Result<Vec<(String, Option<String>, T)>> {
1767        let records = self.list_all_records(collection).await?;
1768        let mut out = Vec::with_capacity(records.len());
1769        for rec in records {
1770            let rkey = rec.rkey().unwrap_or_default().to_string();
1771            match rec.parse::<T>() {
1772                Ok(value) => out.push((rkey, rec.cid, value)),
1773                Err(e) => tracing::warn!(
1774                    collection,
1775                    uri = %rec.uri,
1776                    error = %e,
1777                    "skipping unparseable record in collection"
1778                ),
1779            }
1780        }
1781        Ok(out)
1782    }
1783}
1784
1785// ---------------------------------------------------------------------------
1786// The OAuth sidecar client — the LIVE com.atproto.repo.* path
1787// ---------------------------------------------------------------------------
1788
1789/// A client for the atproto OAuth sidecar's **internal** API.
1790///
1791/// This is the live path for every authed repo operation. Rather than the Rust
1792/// server holding PDS tokens, it POSTs `{did, action, …}` to the sidecar's
1793/// `/internal/repo` endpoint (gated by the shared `X-Internal-Secret`); the
1794/// sidecar `restore(did)`s the OAuth session — transparent DPoP + token refresh —
1795/// and runs the matching XRPC call via `@atproto/api`. The `did` (plus the shared
1796/// secret) is what authorizes the call; there is no bearer token on the Rust side.
1797///
1798/// It also fronts `/internal/session/:id`, the one-shot handoff the Rust callback
1799/// uses to turn a `session_id` (from the sidecar's browser redirect) into the
1800/// `{did, handle}` it keys its own signed cookie by.
1801///
1802/// Cheap to clone (shared `reqwest::Client` + `Arc`'d config).
1803#[derive(Clone)]
1804pub struct SidecarClient {
1805    http: Client,
1806    public_url: Arc<str>,
1807    internal_url: Arc<str>,
1808    internal_secret: Arc<str>,
1809}
1810
1811/// The `{did, handle}` a session-id resolves to (the sidecar's
1812/// `/internal/session/:id` body).
1813#[derive(Debug, Clone, Deserialize)]
1814pub struct SidecarSession {
1815    /// The account DID that logged in.
1816    pub did: String,
1817    /// The account handle at login time.
1818    #[serde(default)]
1819    pub handle: Option<String>,
1820}
1821
1822/// The sidecar's `/internal/revoke` response body:
1823/// `{ ok:true, did, revoked, hadSession }`.
1824#[derive(Debug, Clone, Deserialize)]
1825pub struct RevokeResult {
1826    /// The DID that was revoked.
1827    #[serde(default)]
1828    pub did: String,
1829    /// Whether the OAuth token revocation at the PDS succeeded. `false` means
1830    /// the local rows were still purged (best-effort), but the PDS-side tokens
1831    /// may not have been invalidated (network failure).
1832    #[serde(default)]
1833    pub revoked: bool,
1834    /// Whether the sidecar actually had a stored session for the DID.
1835    #[serde(default, rename = "hadSession")]
1836    pub had_session: bool,
1837}
1838
1839/// The action verbs the sidecar's `/internal/repo` endpoint dispatches on.
1840#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1841pub enum RepoAction {
1842    /// `com.atproto.repo.listRecords`.
1843    List,
1844    /// `com.atproto.repo.createRecord`.
1845    Create,
1846    /// `com.atproto.repo.putRecord`.
1847    Put,
1848    /// `com.atproto.repo.deleteRecord`.
1849    Delete,
1850    /// `com.atproto.repo.applyWrites` (batch).
1851    ApplyWrites,
1852}
1853
1854impl RepoAction {
1855    fn as_str(self) -> &'static str {
1856        match self {
1857            RepoAction::List => "list",
1858            RepoAction::Create => "create",
1859            RepoAction::Put => "put",
1860            RepoAction::Delete => "delete",
1861            RepoAction::ApplyWrites => "applyWrites",
1862        }
1863    }
1864}
1865
1866/// The `/internal/repo` success envelope: `{ ok:true, data:<raw XRPC JSON> }`.
1867#[derive(Debug, Deserialize)]
1868struct RepoOk {
1869    #[serde(default)]
1870    data: Value,
1871}
1872
1873/// `/internal/repo`'s ok envelope for a **listing**, typed all the way down.
1874///
1875/// `data` absent is not `data` empty, the same distinction `records` carries: it
1876/// is what a proxy makes of an unexpected upstream body.
1877#[derive(Debug, Deserialize)]
1878struct RepoOkList {
1879    /// The sidecar's own `ok`, which it can set false on a 200.
1880    #[serde(default)]
1881    ok: Option<bool>,
1882    /// And its own `error` — a different envelope from the PDS's, one layer out.
1883    #[serde(default)]
1884    error: Option<Value>,
1885    #[serde(default)]
1886    data: Option<ListRecordsBody>,
1887}
1888
1889/// The `/internal/repo` error envelope: `{ ok:false, error, message, status? }`.
1890#[derive(Debug, Deserialize)]
1891struct RepoErr {
1892    #[serde(default)]
1893    error: Option<String>,
1894    #[serde(default)]
1895    message: Option<String>,
1896    #[serde(default)]
1897    status: Option<u16>,
1898}
1899
1900impl SidecarClient {
1901    /// Build a sidecar client from the shared `reqwest::Client` and the
1902    /// resolved public + internal base URLs + internal secret (from
1903    /// [`crate::config::SidecarConfig`]). `public_url` anchors the browser
1904    /// `/login` redirect; `internal_url` is the loopback base for the `/internal/*`
1905    /// API (they collapse to the same value in single-URL local dev).
1906    pub fn new(
1907        http: Client,
1908        public_url: impl Into<String>,
1909        internal_url: impl Into<String>,
1910        internal_secret: impl Into<String>,
1911    ) -> Self {
1912        Self {
1913            http,
1914            public_url: Arc::from(public_url.into().trim_end_matches('/')),
1915            internal_url: Arc::from(internal_url.into().trim_end_matches('/')),
1916            internal_secret: Arc::from(internal_secret.into()),
1917        }
1918    }
1919
1920    /// The sidecar's public `/login` URL for a handle, round-tripping an opaque
1921    /// `return` value through OAuth state (used to bounce the browser back to a
1922    /// specific place after login). The browser is redirected here.
1923    pub fn login_url(&self, handle: &str, return_to: Option<&str>) -> String {
1924        let mut url = format!("{}/login?handle={}", self.public_url, urlencode(handle));
1925        if let Some(r) = return_to {
1926            url.push_str(&format!("&return={}", urlencode(r)));
1927        }
1928        url
1929    }
1930
1931    /// Resolve a one-shot `session_id` (from the sidecar's post-OAuth redirect)
1932    /// to the `{did, handle}` that logged in. `Ok(None)` on `404 SessionNotFound`.
1933    pub async fn resolve_session(&self, session_id: &str) -> Result<Option<SidecarSession>> {
1934        let url = format!(
1935            "{}/internal/session/{}",
1936            self.internal_url,
1937            urlencode(session_id)
1938        );
1939        let resp = self
1940            .http
1941            .get(&url)
1942            .header("X-Internal-Secret", self.internal_secret.as_ref())
1943            .send()
1944            .await?;
1945        if resp.status() == StatusCode::NOT_FOUND {
1946            return Ok(None);
1947        }
1948        if !resp.status().is_success() {
1949            return Err(xrpc_error_from(resp).await.into());
1950        }
1951        let raw = crate::net::read_capped(resp).await?;
1952        let session: SidecarSession =
1953            serde_json::from_slice(&raw).context("parsing /internal/session response")?;
1954        Ok(Some(session))
1955    }
1956
1957    /// Revoke a DID's OAuth session at the sidecar: `POST /internal/revoke`.
1958    ///
1959    /// This revokes the refresh + access tokens at the PDS **and** purges the
1960    /// sidecar's stored `oauth_session` + `app_session` rows for the DID. It is
1961    /// idempotent — revoking a DID with no live session returns
1962    /// `had_session: false`. Called on `/logout` (so the cookie clear isn't the
1963    /// only thing that ends the session) and on `/account/delete`.
1964    pub async fn revoke_session(&self, did: &str) -> Result<RevokeResult> {
1965        let url = format!("{}/internal/revoke", self.internal_url);
1966        let resp = self
1967            .http
1968            .post(&url)
1969            .header("X-Internal-Secret", self.internal_secret.as_ref())
1970            .json(&json!({ "did": did }))
1971            .send()
1972            .await?;
1973        if !resp.status().is_success() {
1974            return Err(xrpc_error_from(resp).await.into());
1975        }
1976        let raw = crate::net::read_capped(resp).await?;
1977        let result: RevokeResult =
1978            serde_json::from_slice(&raw).context("parsing /internal/revoke response")?;
1979        Ok(result)
1980    }
1981
1982    /// POST one op to `/internal/repo` and return the raw XRPC `data` payload.
1983    ///
1984    /// `body` must already carry `did` + `action` + the action's required fields
1985    /// (the typed wrappers below build these). Maps the sidecar's error envelope
1986    /// to [`AtProtoError`]: `404 SessionNotFound` → `Xrpc{error:"SessionNotFound"}`
1987    /// so callers can treat it as "re-login required".
1988    async fn repo(&self, body: Value) -> Result<Value> {
1989        let raw = self.repo_bytes(body).await?;
1990        // **The same guard as the listing path, twelve lines below.** This is
1991        // the response path for every write — create, put, delete, applyWrites —
1992        // and `RepoOk.data` is an unbounded `Value`. Measured without it: the
1993        // identical 8 MB attack retains 786 MB. The listing had the guard and
1994        // this did not, which is the drift a shared helper exists to prevent.
1995        refuse_a_structure_explosion(&raw, "the /internal/repo body")?;
1996        let ok: RepoOk = serde_json::from_slice(&raw).context("parsing /internal/repo ok body")?;
1997        Ok(ok.data)
1998    }
1999
2000    /// [`repo`](Self::repo) without the `Value`.
2001    ///
2002    /// The listing path needs the bytes: a `serde_json::Map` resolves a repeated
2003    /// key last-wins, so a body carrying a second, empty `records` array read as
2004    /// a successful empty page — and an empty page here is `replace_sub_refs`
2005    /// deleting every `sub_ref` the reader has. serde refuses a duplicated field
2006    /// outright, but only if it sees the bytes.
2007    async fn repo_bytes(&self, body: Value) -> Result<Vec<u8>> {
2008        let url = format!("{}/internal/repo", self.internal_url);
2009        let resp = self
2010            .http
2011            .post(&url)
2012            .header("X-Internal-Secret", self.internal_secret.as_ref())
2013            .json(&body)
2014            .send()
2015            .await?;
2016        let status = resp.status();
2017        // **Capped, like every other body this codebase reads.** `resp.json()`
2018        // buffers whatever arrives; `/internal/repo` proxies the account's PDS,
2019        // so that length is chosen by a host the reader picked and we did not.
2020        // The 8 MB ceiling that bounds the direct client did not exist here,
2021        // and the sidecar is the default backend — so the one path with no byte
2022        // bound at all was the one most deployments run.
2023        let raw = crate::net::read_capped(resp).await;
2024        if status.is_success() {
2025            return raw;
2026        }
2027        // Error path: parse the sidecar's `{ok:false,error,message,status}` shape.
2028        //
2029        // **A body we could not read must not cost us the status.** Reading
2030        // before the branch was the obvious shape and it swallowed the HTTP
2031        // status on an over-cap or truncated error body, turning a `404
2032        // SessionNotFound` into a bare "body exceeded the cap". `xrpc_error_from`
2033        // already makes the opposite choice deliberately, for the same reason.
2034        let err: RepoErr = raw
2035            .ok()
2036            .and_then(|body| serde_json::from_slice(&body).ok())
2037            .unwrap_or(RepoErr {
2038                error: None,
2039                message: None,
2040                status: None,
2041            });
2042        let mapped = err
2043            .status
2044            .and_then(|s| StatusCode::from_u16(s).ok())
2045            .unwrap_or(status);
2046        Err(AtProtoError::Xrpc {
2047            status: mapped,
2048            error: err.error.unwrap_or_else(|| "Unknown".to_string()),
2049            message: err.message,
2050        }
2051        .into())
2052    }
2053
2054    // -- raw com.atproto.repo.* over the sidecar -----------------------------
2055
2056    /// `list` — one page of a collection's records for `did`.
2057    pub async fn list_records(
2058        &self,
2059        did: &str,
2060        collection: &str,
2061        limit: Option<u32>,
2062        cursor: Option<&str>,
2063    ) -> Result<ListRecordsResponse> {
2064        let mut body = json!({
2065            "did": did,
2066            "action": RepoAction::List.as_str(),
2067            "collection": collection,
2068        });
2069        if let Some(limit) = limit {
2070            body["limit"] = json!(limit);
2071        }
2072        if let Some(cursor) = cursor {
2073            body["cursor"] = json!(cursor);
2074        }
2075        // Straight into the shared wire struct, one parse, no `Value` between —
2076        // so this client gets the same guards as the other two, including the
2077        // duplicated-key refusal that only serde can make.
2078        let raw = self.repo_bytes(body).await?;
2079        refuse_a_structure_explosion(&raw, "the sidecar listRecords body")?;
2080        let envelope: RepoOkList =
2081            serde_json::from_slice(&raw).context("parsing sidecar listRecords data")?;
2082        // **Two envelope layers here, not one.** `page_from_body` guards the
2083        // PDS's, which arrives inside `data`; this is the sidecar's own, and it
2084        // can say `{"ok":false,"error":"ExpiredToken"}` on a 200 while still
2085        // carrying a `data` that reads as a perfectly good empty page.
2086        if envelope.ok == Some(false) {
2087            let name = envelope
2088                .error
2089                .as_ref()
2090                .and_then(envelope_error_name)
2091                .unwrap_or_else(|| "unspecified".to_string());
2092            anyhow::bail!("the sidecar answered 2xx with ok:false ({name})");
2093        }
2094        if let Some(name) = envelope.error.as_ref().and_then(envelope_error_name) {
2095            anyhow::bail!("the sidecar answered 2xx with an error envelope: {name}");
2096        }
2097        let Some(data) = envelope.data else {
2098            return Err(UnreadableListing::NoRecords.into());
2099        };
2100        page_from_body(data).context("parsing sidecar listRecords data")
2101    }
2102
2103    /// Page through **all** records in a collection for `did`.
2104    ///
2105    /// Bounded by `MAX_LIST_PAGES` and cursor-repetition detection, same as
2106    /// [`PdsClient::list_all_records`] — the sidecar proxies to the account's
2107    /// PDS, so the page count is ultimately remote-controlled here too. Both
2108    /// refuse rather than return a short list (#203).
2109    pub async fn list_all_records(&self, did: &str, collection: &str) -> Result<Vec<RecordEntry>> {
2110        self.list_all_records_within(did, collection, &mut ByteBudget::new(MAX_LIST_BYTES))
2111            .await
2112    }
2113
2114    /// [`list_all_records`](Self::list_all_records) against a caller's budget.
2115    pub(crate) async fn list_all_records_within(
2116        &self,
2117        did: &str,
2118        collection: &str,
2119        budget: &mut ByteBudget,
2120    ) -> Result<Vec<RecordEntry>> {
2121        let mut out = Vec::new();
2122        let max_bytes = budget.max();
2123        let mut cursor: Option<String> = None;
2124        let mut more_offered = false;
2125        for _ in 0..MAX_LIST_PAGES {
2126            let page = self
2127                .list_records(did, collection, Some(100), cursor.as_deref())
2128                .await?;
2129            refuse_malformed(&page, collection)?;
2130            let got = page.records.len();
2131            // The sidecar proxies the account's PDS, so this walk's size is as
2132            // remote-controlled as the direct client's. It carried no budget at
2133            // all until a review noticed it was the default backend.
2134            if !budget.admit(&page.records) {
2135                return Err(ListingTooLarge::Bytes {
2136                    collection: collection.to_string(),
2137                    max_bytes,
2138                    held: out.len(),
2139                    charged: budget.used(),
2140                }
2141                .into());
2142            }
2143            extend_bounded(&mut out, page.records, MAX_LIST_RECORDS, collection)?;
2144            match page.cursor {
2145                // Same two exits as `walk_all_within`: a repeated cursor on a
2146                // non-empty page is refused (#203); a cursor on an empty page
2147                // ends the walk, as a real PDS's last page does.
2148                Some(next) if got > 0 && Some(&next) == cursor.as_ref() => {
2149                    anyhow::bail!(
2150                        "listRecords for {collection} returned a repeated cursor on a \
2151                         non-empty page ({} held) — refusing a list the server did not finish",
2152                        out.len(),
2153                    );
2154                }
2155                Some(next) if got > 0 => {
2156                    cursor = Some(next);
2157                    more_offered = true;
2158                }
2159                _ => {
2160                    more_offered = false;
2161                    break;
2162                }
2163            }
2164        }
2165        // **Running out of pages is a refusal, not a short answer.** Falling out
2166        // of the loop used to return `Ok(out)`, so a repo bigger than the page
2167        // budget produced a truncated list indistinguishable from a complete
2168        // one — and `resolve_subscriptions` needs an `Err` for its fail-closed
2169        // branch. Given `Ok`, it hands the short list to `replace_sub_refs`,
2170        // which DELETEs the reader's whole `sub_ref` projection and reinserts
2171        // only what it was given. `extend_bounded` cannot catch this either:
2172        // `MAX_LIST_PAGES` x the 100 we request is `MAX_LIST_RECORDS`, so the
2173        // page budget runs out first.
2174        //
2175        // **The cap is on REQUESTS, so where it bites in RECORDS is the server's
2176        // choice and not ours.** We ask for 100 a page; a PDS MAY answer with
2177        // fewer, and only one that honours the limit puts the boundary anywhere
2178        // near `MAX_LIST_PAGES` x 100. Halve the page size and the same budget
2179        // reaches half as many records; a server that returns MORE than asked
2180        // trips `extend_bounded` first, which is the case the sentence above does
2181        // not cover. Said this way because an earlier version of this comment
2182        // named a fixed record window as though our own constants decided it.
2183        //
2184        // **And at the boundary the refusal is a FALSE one.** Terminating costs
2185        // one extra request, because a short page can still carry a cursor — this
2186        // project's own PDS does exactly that — so a walk that fills its last
2187        // allowed page is holding every record it was ever going to hold and
2188        // refuses anyway, on the strength of a cursor it never followed. With
2189        // `limit=100` honoured that window is a repo of roughly 19 901 to 20 000
2190        // records. The direction is safe and the alternative is deleting feeds,
2191        // but it is a false refusal and not a clean boundary.
2192        if more_offered {
2193            return Err(ListingTooLarge::Pages {
2194                collection: collection.to_string(),
2195                pages: MAX_LIST_PAGES,
2196                held: out.len(),
2197            }
2198            .into());
2199        }
2200        Ok(out)
2201    }
2202
2203    /// `create` — create a record (server-assigned rkey). Returns its strong ref.
2204    /// **Private, not `pub` — and not `pub(crate)`.** This is generic over
2205    /// `T: Serialize`, so it will happily write a raw `lexicon::Subscription`:
2206    /// the general case of the hole `create_subscriptions_batch` was one
2207    /// instance of. The vetted wrappers in this `impl` are the sanctioned entry
2208    /// points. `pub(crate)` was tried first and stops nothing that matters — a
2209    /// handler in `web.rs` is in this crate. Private is what makes the wrappers
2210    /// a fact rather than a convention, and it costs nothing: nothing outside
2211    /// this module ever called it.
2212    async fn create_record<T: Serialize>(
2213        &self,
2214        did: &str,
2215        collection: &str,
2216        record: &T,
2217    ) -> Result<WriteResult> {
2218        let body = json!({
2219            "did": did,
2220            "action": RepoAction::Create.as_str(),
2221            "collection": collection,
2222            "record": record,
2223        });
2224        let data = self.repo(body).await?;
2225        serde_json::from_value(data).context("parsing sidecar createRecord data")
2226    }
2227
2228    /// `put` — upsert a record at a known rkey. Returns its strong ref.
2229    /// **Private, not `pub` — and not `pub(crate)`.** This is generic over
2230    /// `T: Serialize`, so it will happily write a raw `lexicon::Subscription`:
2231    /// the general case of the hole `create_subscriptions_batch` was one
2232    /// instance of. The vetted wrappers in this `impl` are the sanctioned entry
2233    /// points. `pub(crate)` was tried first and stops nothing that matters — a
2234    /// handler in `web.rs` is in this crate. Private is what makes the wrappers
2235    /// a fact rather than a convention, and it costs nothing: nothing outside
2236    /// this module ever called it.
2237    async fn put_record<T: Serialize>(
2238        &self,
2239        did: &str,
2240        collection: &str,
2241        rkey: &str,
2242        record: &T,
2243        swap_record: Option<&str>,
2244    ) -> Result<WriteResult> {
2245        let mut body = json!({
2246            "did": did,
2247            "action": RepoAction::Put.as_str(),
2248            "collection": collection,
2249            "rkey": rkey,
2250            "record": record,
2251        });
2252        // The sidecar validates it as a CID and passes it to `putRecord`; its
2253        // `InvalidSwap` comes back through `repo`'s error envelope with the
2254        // PDS's status and name intact (#149).
2255        if let Some(cid) = swap_record {
2256            body["swapRecord"] = json!(cid);
2257        }
2258        let data = self.repo(body).await?;
2259        serde_json::from_value(data).context("parsing sidecar putRecord data")
2260    }
2261
2262    /// `delete` — delete a record by collection + rkey.
2263    pub async fn delete_record(&self, did: &str, collection: &str, rkey: &str) -> Result<()> {
2264        let body = json!({
2265            "did": did,
2266            "action": RepoAction::Delete.as_str(),
2267            "collection": collection,
2268            "rkey": rkey,
2269        });
2270        // A 200 carrying an error envelope is not a delete: this reported
2271        // success while the record stayed in the reader's repo, and the UI
2272        // showed them unsubscribed from a feed they still had.
2273        self.repo(body)
2274            .await
2275            .and_then(|data| reject_error_envelope(&data))?;
2276        Ok(())
2277    }
2278
2279    /// `applyWrites` — a batch of create/update/delete ops in one round-trip.
2280    /// **Private, not `pub` — and not `pub(crate)`.** This is generic over
2281    /// `T: Serialize`, so it will happily write a raw `lexicon::Subscription`:
2282    /// the general case of the hole `create_subscriptions_batch` was one
2283    /// instance of. The vetted wrappers in this `impl` are the sanctioned entry
2284    /// points. `pub(crate)` was tried first and stops nothing that matters — a
2285    /// handler in `web.rs` is in this crate. Private is what makes the wrappers
2286    /// a fact rather than a convention, and it costs nothing: nothing outside
2287    /// this module ever called it.
2288    ///
2289    /// Sent in chunks within the PDS's limits — see [`apply_writes_chunked`]
2290    /// for what a failure part-way means. **Chunked here, not in the sidecar**,
2291    /// although the sidecar is what calls the PDS: it answers one
2292    /// `/internal/repo` request with one result, which has no way to say that
2293    /// half a batch landed, and this client is its only caller. Chunking here
2294    /// also keeps the hop itself under the sidecar's 1 MiB Fastify body limit.
2295    async fn apply_writes(&self, did: &str, writes: &[WriteOp]) -> Result<()> {
2296        apply_writes_chunked(writes, |chunk| self.apply_writes_once(did, chunk)).await
2297    }
2298
2299    /// One `/internal/repo` `applyWrites`, unchunked. Reached only through
2300    /// [`apply_writes`](Self::apply_writes).
2301    async fn apply_writes_once(&self, did: &str, writes: &[WriteOp]) -> Result<()> {
2302        let ops: Vec<Value> = writes.iter().map(WriteOp::to_sidecar_json).collect();
2303        let body = json!({
2304            "did": did,
2305            "action": RepoAction::ApplyWrites.as_str(),
2306            "writes": ops,
2307        });
2308        self.repo(body)
2309            .await
2310            .and_then(|data| reject_error_envelope(&data))?;
2311        Ok(())
2312    }
2313
2314    // -- typed lexicon wrappers (mirror the old PdsClient surface) ------------
2315
2316    /// List every [`Subscription`] record in `did`'s repo (paged fully).
2317    pub async fn list_subscriptions(&self, did: &str) -> Result<Vec<(String, Subscription)>> {
2318        self.list_typed(did, lexicon::nsid::SUBSCRIPTION).await
2319    }
2320
2321    /// Every [`Subscription`] with the CID it was listed at, unsorted — the
2322    /// read half of a read-modify-write that puts with `swapRecord` (#149).
2323    pub async fn list_subscriptions_with_cids(
2324        &self,
2325        did: &str,
2326    ) -> Result<Vec<(String, Option<String>, Subscription)>> {
2327        self.list_typed_with_cids(did, lexicon::nsid::SUBSCRIPTION)
2328            .await
2329    }
2330
2331    /// Create a [`Subscription`] record (subscribe to a feed).
2332    pub async fn create_subscription(
2333        &self,
2334        did: &str,
2335        sub: &crate::vetted::VettedSubscription,
2336    ) -> Result<WriteResult> {
2337        self.create_record(did, lexicon::nsid::SUBSCRIPTION, sub)
2338            .await
2339    }
2340
2341    /// Delete a [`Subscription`] record by rkey (unsubscribe).
2342    pub async fn delete_subscription(&self, did: &str, rkey: &str) -> Result<()> {
2343        self.delete_record(did, lexicon::nsid::SUBSCRIPTION, rkey)
2344            .await
2345    }
2346
2347    /// List every [`Folder`] record in `did`'s repo.
2348    pub async fn list_folders(&self, did: &str) -> Result<Vec<(String, Folder)>> {
2349        self.list_typed(did, lexicon::nsid::FOLDER).await
2350    }
2351
2352    /// Every [`Folder`] with the CID it was listed at, unsorted (#268).
2353    pub async fn list_folders_with_cids(
2354        &self,
2355        did: &str,
2356    ) -> Result<Vec<(String, Option<String>, Folder)>> {
2357        self.list_typed_with_cids(did, lexicon::nsid::FOLDER).await
2358    }
2359
2360    /// List every [`Saved`] record in `did`'s repo.
2361    pub async fn list_saved(&self, did: &str) -> Result<Vec<(String, Saved)>> {
2362        self.list_typed(did, lexicon::nsid::SAVED).await
2363    }
2364
2365    /// List every [`ReadState`] cursor in `did`'s repo (the read side a
2366    /// login-time read-state merge would consume).
2367    pub async fn list_read_states(&self, did: &str) -> Result<Vec<(String, ReadState)>> {
2368        self.list_typed(did, lexicon::nsid::READ_STATE).await
2369    }
2370
2371    /// Upsert a single [`ReadState`] cursor at its feed-derived rkey.
2372    pub async fn put_read_state(
2373        &self,
2374        did: &str,
2375        rkey: &str,
2376        state: &ReadState,
2377    ) -> Result<WriteResult> {
2378        self.put_record(did, lexicon::nsid::READ_STATE, rkey, state, None)
2379            .await
2380    }
2381
2382    /// Batch-flush many dirty [`ReadState`] cursors via `applyWrites` (chunked).
2383    ///
2384    /// Each `(rkey, state, pds_created)` becomes a `create` op at the feed-derived
2385    /// rkey when the record does NOT yet exist (`pds_created == false`), and an
2386    /// `update` op when it does. This is what makes the FIRST flush of a feed
2387    /// succeed: `applyWrites#update` errors on a record that does not pre-exist,
2388    /// and `applyWrites` is atomic per-repo, so a single not-yet-created cursor
2389    /// would otherwise drop the whole DID batch. Both kinds ride the SAME
2390    /// `applyWrites` batch so batching is preserved.
2391    pub async fn flush_read_states(
2392        &self,
2393        did: &str,
2394        cursors: &[(String, ReadState, bool)],
2395    ) -> Result<()> {
2396        if cursors.is_empty() {
2397            return Ok(());
2398        }
2399        let writes = read_state_write_ops(cursors)?;
2400        self.apply_writes(did, &writes).await
2401    }
2402
2403    // -- reader-facing record CRUD (the surface the web layer calls) ----------
2404    //
2405    // These are the typed convenience methods `web.rs` uses to manage a user's
2406    // feeds/folders/saved items *as records in their PDS*. They mirror the
2407    // create/list surface above but use the reader vocabulary
2408    // (add/remove/rename) and, for the `add_*` verbs, return the server-assigned
2409    // rkey so the caller can address the new record without a re-list. Ordering
2410    // is made deterministic where it matters (see [`list_subscriptions_sorted`]
2411    // etc.) so the server-rendered HTML is stable between reads.
2412
2413    // -- subscriptions -------------------------------------------------------
2414
2415    /// Add a subscription (subscribe to a feed) — `createRecord`, server-assigned
2416    /// `tid` rkey. Returns the new record's **rkey** so the web layer can offer
2417    /// unsubscribe/rename immediately.
2418    pub async fn add_subscription(
2419        &self,
2420        did: &str,
2421        sub: &crate::vetted::VettedSubscription,
2422    ) -> Result<String> {
2423        Ok(self.create_subscription(did, sub).await?.into_rkey())
2424    }
2425
2426    /// Remove a subscription (unsubscribe) by rkey — `deleteRecord`. Alias of
2427    /// [`delete_subscription`](Self::delete_subscription) in the reader vocabulary.
2428    pub async fn remove_subscription(&self, did: &str, rkey: &str) -> Result<()> {
2429        self.delete_subscription(did, rkey).await
2430    }
2431
2432    /// Update / rename a subscription in place at a known rkey — `putRecord`.
2433    ///
2434    /// The whole record is replaced (retitle, move to a folder, change the
2435    /// fetch hint …). Upsert semantics: it also creates the record if the rkey
2436    /// is somehow absent, so it is safe as a general "write this exact record".
2437    pub async fn update_subscription(
2438        &self,
2439        did: &str,
2440        rkey: &str,
2441        sub: &crate::vetted::VettedSubscription,
2442        swap_record: Option<&str>,
2443    ) -> Result<WriteResult> {
2444        self.put_record(did, lexicon::nsid::SUBSCRIPTION, rkey, sub, swap_record)
2445            .await
2446    }
2447
2448    /// List every subscription, **sorted deterministically** — by display title
2449    /// (case-insensitive), then feed URL, then rkey as the final tiebreaker — so
2450    /// the rendered feed list is stable across reads regardless of PDS return
2451    /// order. Untitled feeds sort by their URL.
2452    pub async fn list_subscriptions_sorted(
2453        &self,
2454        did: &str,
2455    ) -> Result<Vec<(String, Subscription)>> {
2456        let mut subs = self.list_subscriptions(did).await?;
2457        // The comparator is SHARED with the Rust-native client so the two
2458        // cannot order the list differently across the cutover.
2459        subs.sort_by(lexicon::sort::subscriptions);
2460        Ok(subs)
2461    }
2462
2463    /// Batch-add many subscriptions via `applyWrites` (chunked) — the OPML-import path.
2464    ///
2465    /// Each feed becomes one `create` op. Client-side monotonic `tid`
2466    /// rkeys are assigned so the batch is deterministic and the imported feeds
2467    /// keep OPML order (server-assigned tids would also be monotonic, but pinning
2468    /// them here makes the whole import reproducible and testable offline).
2469    /// Returns the assigned rkeys in input order.
2470    ///
2471    /// More than [`APPLY_WRITES_MAX_OPS`] feeds is more than one call, so the
2472    /// import can part-land: on an error, [`ApplyWritesIncomplete::of`] gives
2473    /// `landed`, and the first `landed` of `subs` are in the repo.
2474    pub async fn add_subscriptions_bulk(
2475        &self,
2476        did: &str,
2477        subs: &[crate::vetted::VettedSubscription],
2478    ) -> Result<Vec<String>> {
2479        let mut gen = TidGenerator::new();
2480        let mut rkeys = Vec::with_capacity(subs.len());
2481        let mut writes = Vec::with_capacity(subs.len());
2482        for sub in subs {
2483            let rkey = gen.next();
2484            writes.push(WriteOp::Create {
2485                collection: lexicon::nsid::SUBSCRIPTION.to_string(),
2486                rkey: Some(rkey.clone()),
2487                value: serde_json::to_value(sub)?,
2488            });
2489            rkeys.push(rkey);
2490        }
2491        self.apply_writes(did, &writes).await?;
2492        Ok(rkeys)
2493    }
2494
2495    // -- folders -------------------------------------------------------------
2496
2497    /// Add a folder — `createRecord`, server-assigned `tid` rkey. Returns the
2498    /// new folder's rkey (subscriptions reference it by its `at://` URI).
2499    pub async fn add_folder(&self, did: &str, folder: &Folder) -> Result<String> {
2500        Ok(self
2501            .create_record(did, lexicon::nsid::FOLDER, folder)
2502            .await?
2503            .into_rkey())
2504    }
2505
2506    /// Remove a folder by rkey — `deleteRecord`. (Subscriptions referencing it
2507    /// are left untouched; a dangling `folder` ref reads as "unfiled".)
2508    pub async fn remove_folder(&self, did: &str, rkey: &str) -> Result<()> {
2509        self.delete_record(did, lexicon::nsid::FOLDER, rkey).await
2510    }
2511
2512    /// Rename / update a folder in place at a known rkey — `putRecord`
2513    /// (rename, or change its `position` sort hint).
2514    ///
2515    /// Replaces the WHOLE record, so `folder` must be the record as read with
2516    /// only the intended change. `swap_record` is the CID it was read at: the
2517    /// PDS refuses the write with `InvalidSwap` if the record has moved since
2518    /// (#268). `None` writes unconditionally.
2519    pub async fn rename_folder(
2520        &self,
2521        did: &str,
2522        rkey: &str,
2523        folder: &Folder,
2524        swap_record: Option<&str>,
2525    ) -> Result<WriteResult> {
2526        self.put_record(did, lexicon::nsid::FOLDER, rkey, folder, swap_record)
2527            .await
2528    }
2529
2530    /// List every folder, **sorted deterministically** — by `position` (the
2531    /// lexicon's sort hint; unset sorts last), then name (case-insensitive),
2532    /// then rkey — so the sidebar order is stable.
2533    pub async fn list_folders_sorted(&self, did: &str) -> Result<Vec<(String, Folder)>> {
2534        let mut folders = self.list_folders(did).await?;
2535        folders.sort_by(lexicon::sort::folders);
2536        Ok(folders)
2537    }
2538
2539    // -- saved / starred -----------------------------------------------------
2540
2541    /// Add a saved (starred / save-for-later) entry — `createRecord`,
2542    /// server-assigned `tid` rkey. Returns the new record's rkey.
2543    pub async fn add_saved(&self, did: &str, saved: &crate::vetted::VettedSaved) -> Result<String> {
2544        Ok(self
2545            .create_record(did, lexicon::nsid::SAVED, saved)
2546            .await?
2547            .into_rkey())
2548    }
2549
2550    /// Remove a saved entry by rkey — `deleteRecord` (un-star).
2551    pub async fn remove_saved(&self, did: &str, rkey: &str) -> Result<()> {
2552        self.delete_record(did, lexicon::nsid::SAVED, rkey).await
2553    }
2554
2555    /// List every saved entry, **sorted deterministically** — newest first by
2556    /// `createdAt` (RFC-3339 sorts lexicographically), then rkey — so the
2557    /// "saved for later" list reads most-recent-first and is stable.
2558    pub async fn list_saved_sorted(&self, did: &str) -> Result<Vec<(String, Saved)>> {
2559        let mut saved = self.list_saved(did).await?;
2560        saved.sort_by(lexicon::sort::saved);
2561        Ok(saved)
2562    }
2563
2564    /// List a collection for `did` and parse each record's value into `T`,
2565    /// pairing it with its rkey. Unparseable records are skipped with a warning
2566    /// (forward-compat).
2567    async fn list_typed<T: DeserializeOwned>(
2568        &self,
2569        did: &str,
2570        collection: &str,
2571    ) -> Result<Vec<(String, T)>> {
2572        Ok(self
2573            .list_typed_with_cids(did, collection)
2574            .await?
2575            .into_iter()
2576            .map(|(rkey, _cid, value)| (rkey, value))
2577            .collect())
2578    }
2579
2580    /// [`list_typed`](Self::list_typed), keeping each record's CID (#149).
2581    async fn list_typed_with_cids<T: DeserializeOwned>(
2582        &self,
2583        did: &str,
2584        collection: &str,
2585    ) -> Result<Vec<(String, Option<String>, T)>> {
2586        let records = self.list_all_records(did, collection).await?;
2587        let mut out = Vec::with_capacity(records.len());
2588        for rec in records {
2589            let rkey = rec.rkey().unwrap_or_default().to_string();
2590            match rec.parse::<T>() {
2591                Ok(value) => out.push((rkey, rec.cid, value)),
2592                Err(e) => tracing::warn!(
2593                    collection,
2594                    uri = %rec.uri,
2595                    error = %e,
2596                    "skipping unparseable record in collection"
2597                ),
2598            }
2599        }
2600        Ok(out)
2601    }
2602}
2603
2604// ---------------------------------------------------------------------------
2605// applyWrites operations
2606// ---------------------------------------------------------------------------
2607
2608/// Build the `applyWrites` ops for a batch of dirty read-state cursors.
2609///
2610/// Each `(rkey, state, pds_created)` becomes a `#create` op (at the stable
2611/// feed-derived rkey) when the PDS record does NOT yet exist, and a `#update`
2612/// when it does. This is the crux of the first-flush fix: an `#update` on a
2613/// missing record errors, and `applyWrites` is atomic per-repo, so a single
2614/// not-yet-created cursor in the batch would drop the whole DID's flush. Emitting
2615/// a `create` for those makes a feed's first flush succeed while keeping every
2616/// op in ONE batch. Shared by both the sidecar and direct-PDS flush paths.
2617pub(crate) fn read_state_write_ops(cursors: &[(String, ReadState, bool)]) -> Result<Vec<WriteOp>> {
2618    cursors
2619        .iter()
2620        .map(|(rkey, state, pds_created)| {
2621            let value = serde_json::to_value(state)?;
2622            Ok(if *pds_created {
2623                WriteOp::Update {
2624                    collection: lexicon::nsid::READ_STATE.to_string(),
2625                    rkey: rkey.clone(),
2626                    value,
2627                }
2628            } else {
2629                WriteOp::Create {
2630                    collection: lexicon::nsid::READ_STATE.to_string(),
2631                    rkey: Some(rkey.clone()),
2632                    value,
2633                }
2634            })
2635        })
2636        .collect()
2637}
2638
2639/// Most writes one `com.atproto.repo.applyWrites` call may carry.
2640///
2641/// **The limit is the reference PDS's, not the lexicon's.** The lexicon's
2642/// `writes` array has no `maxLength` (checked against
2643/// `lexicons/com/atproto/repo/applyWrites.json` on bluesky-social/atproto
2644/// `main`, and against its history back to 2024-02); the cap is enforced by the
2645/// handler, `packages/pds/src/api/com/atproto/repo/applyWrites.ts`
2646/// (`if (writes.length > 200) throw new InvalidRequestError('Too many writes.
2647/// Max: 200')`, unchanged at a0c49d9). vlpds documents the same figure for its
2648/// commit coalescing. A PDS that allows more loses nothing by being sent 200.
2649pub const APPLY_WRITES_MAX_OPS: usize = 200;
2650
2651/// Most bytes of serialized writes one `applyWrites` call may carry.
2652///
2653/// **Sized to the smallest limit a deployed PDS is known to apply, not the
2654/// largest.** The reference PDS took `applyWrites` bodies up to its server-wide
2655/// `jsonLimit` of 150 KiB (`150 * 1024` in `packages/pds/src/index.ts`) until
2656/// atproto#4989 (2026-05-21) raised the record methods to `1_000_000` bytes.
2657/// Self-hosted PDSes run older releases for months, so the 150 KiB figure is
2658/// the live one for some readers. The other limits on this path are all
2659/// larger: the newer reference PDS's 1,000,000 bytes, the sidecar hop's Fastify
2660/// default `bodyLimit` of 1 MiB, and vlpds's coalesced commit of "up to 200
2661/// operations or 1 MB of record bytes".
2662///
2663/// 128 KiB leaves 22 KiB under 150 KiB for what this does not count — the
2664/// `{"repo": …, "writes": [ ]}` envelope, around a hundred bytes — and is
2665/// measured on the PDS-shaped op (`WriteOp::to_json`), which is also what the
2666/// sidecar forwards and is larger than the sidecar's own request shape.
2667///
2668/// What it costs, measured: 200 subscriptions with a title and site URL each
2669/// are ~76 KB, so an ordinary OPML import is still one call per 200 feeds. A
2670/// read-state cursor with a full 1,000-id set is ~10 KB with the store's
2671/// integer entry ids (~20 KB with both sets full), so a flush crosses the
2672/// bound at around a dozen full cursors — rare, since a cursor is compacted at
2673/// half the cap — and a refused body is worse than an extra round trip.
2674///
2675/// A single op larger than this is still sent, alone: it cannot be split, and
2676/// the PDS is the one to judge it.
2677pub const APPLY_WRITES_MAX_BYTES: usize = 128 * 1024;
2678
2679/// Split a batch into the consecutive ranges [`apply_writes_chunked`] sends,
2680/// each within [`APPLY_WRITES_MAX_OPS`] and [`APPLY_WRITES_MAX_BYTES`].
2681///
2682/// Ranges rather than slices so a caller can say WHICH writes a chunk held.
2683/// Order is preserved, every op is in exactly one range, and no range is empty;
2684/// an empty batch yields no ranges.
2685pub(crate) fn chunk_writes(writes: &[WriteOp]) -> Vec<std::ops::Range<usize>> {
2686    let mut chunks = Vec::new();
2687    let mut start = 0;
2688    let mut bytes = 0;
2689    for (i, op) in writes.iter().enumerate() {
2690        // +1 for the comma between array elements.
2691        let size = op.to_json().to_string().len() + 1;
2692        let held = i - start;
2693        if held > 0 && (held == APPLY_WRITES_MAX_OPS || bytes + size > APPLY_WRITES_MAX_BYTES) {
2694            chunks.push(start..i);
2695            start = i;
2696            bytes = 0;
2697        }
2698        bytes += size;
2699    }
2700    if start < writes.len() {
2701        chunks.push(start..writes.len());
2702    }
2703    chunks
2704}
2705
2706/// Send `writes` as consecutive `applyWrites` calls within the PDS's limits,
2707/// stopping at the first that fails.
2708///
2709/// **Every client's `apply_writes` goes through this**, so no caller can send
2710/// an oversized call: the OAuth client ([`crate::oauth::xrpc::Repo`]), the
2711/// sidecar client ([`SidecarClient`]) and the direct client ([`PdsClient`]).
2712/// `send` is that client's single-call primitive.
2713///
2714/// **The batch is no longer atomic.** `applyWrites` is atomic per CALL, so a
2715/// split batch can half-land. The contract a caller gets instead:
2716///
2717/// * chunks go in input order, one at a time, and a failure stops the run —
2718///   so what landed is always a PREFIX of `writes`;
2719/// * on failure the error carries an [`ApplyWritesIncomplete`] saying how long
2720///   that prefix is, how many writes after it are in doubt (the failed chunk:
2721///   atomic, so all or none, but a timeout cannot say which), and that the
2722///   rest were never sent.
2723///
2724/// Stopping rather than carrying on is what keeps that a prefix: sending chunk
2725/// 3 after chunk 2 failed would leave a gap no caller could describe in one
2726/// number, and a gap in an OPML import is feeds silently missing from the
2727/// middle of the list.
2728pub(crate) async fn apply_writes_chunked<'a, F, Fut>(
2729    writes: &'a [WriteOp],
2730    mut send: F,
2731) -> Result<()>
2732where
2733    F: FnMut(&'a [WriteOp]) -> Fut,
2734    Fut: std::future::Future<Output = Result<()>>,
2735{
2736    let chunks = chunk_writes(writes);
2737    let count = chunks.len();
2738    for (i, range) in chunks.into_iter().enumerate() {
2739        let (landed, in_doubt) = (range.start, range.len());
2740        if let Err(cause) = send(&writes[range]).await {
2741            return Err(ApplyWritesIncomplete {
2742                landed,
2743                in_doubt,
2744                total: writes.len(),
2745                chunk: i + 1,
2746                chunks: count,
2747                cause,
2748            }
2749            .into());
2750        }
2751    }
2752    Ok(())
2753}
2754
2755/// How far a chunked `applyWrites` got before a chunk failed — carried by the
2756/// error from every client's `apply_writes`, and so from `flush_read_states`
2757/// and `add_subscriptions_bulk`.
2758///
2759/// Read it with [`ApplyWritesIncomplete::of`]. In terms of the caller's own
2760/// input, in order:
2761///
2762/// * `writes[..landed]` were committed (each chunk was acknowledged);
2763/// * `writes[landed..landed + in_doubt]` were the failed call — usually not
2764///   committed, but a timeout or a lost response cannot rule it out;
2765/// * everything after was never sent.
2766///
2767/// Display is the underlying failure's message, with the progress appended
2768/// when the batch had more than one chunk, so a log line still names the
2769/// PDS's reason. The failure's own causes stay reachable through
2770/// [`anyhow::Error::chain`].
2771#[derive(Debug)]
2772pub struct ApplyWritesIncomplete {
2773    /// Writes committed, counted from the start of the input.
2774    pub landed: usize,
2775    /// Writes in the failed call, starting at `landed`.
2776    pub in_doubt: usize,
2777    /// Writes in the whole batch.
2778    pub total: usize,
2779    chunk: usize,
2780    chunks: usize,
2781    cause: anyhow::Error,
2782}
2783
2784impl ApplyWritesIncomplete {
2785    /// The progress record an `apply_writes` error carries, if it has one.
2786    pub fn of(err: &anyhow::Error) -> Option<&Self> {
2787        err.chain().find_map(|e| e.downcast_ref::<Self>())
2788    }
2789
2790    /// The failure that stopped the run, as the client reported it.
2791    pub fn cause(&self) -> &anyhow::Error {
2792        &self.cause
2793    }
2794}
2795
2796impl std::fmt::Display for ApplyWritesIncomplete {
2797    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
2798        write!(f, "{}", self.cause)?;
2799        if self.chunks > 1 {
2800            write!(
2801                f,
2802                " (applyWrites call {} of {}; {} of {} writes had landed)",
2803                self.chunk, self.chunks, self.landed, self.total
2804            )?;
2805        }
2806        Ok(())
2807    }
2808}
2809
2810impl std::error::Error for ApplyWritesIncomplete {
2811    // The cause's own Display is already in ours, so the chain continues from
2812    // ITS source — `{:#}` would otherwise print the PDS's message twice.
2813    fn source(&self) -> Option<&(dyn std::error::Error + 'static)> {
2814        self.cause.chain().nth(1)
2815    }
2816}
2817
2818/// One operation in a `PdsClient::apply_writes` batch.
2819///
2820/// Maps to the `com.atproto.repo.applyWrites` union of
2821/// `#create` / `#update` / `#delete`.
2822#[derive(Debug, Clone)]
2823pub enum WriteOp {
2824    /// Create a record (server-assigned rkey unless `rkey` is given).
2825    Create {
2826        /// The collection NSID.
2827        collection: String,
2828        /// Optional explicit rkey (`None` → server assigns a tid).
2829        rkey: Option<String>,
2830        /// The record body.
2831        value: Value,
2832    },
2833    /// Upsert a record at a known rkey (the read-state cursor case).
2834    Update {
2835        /// The collection NSID.
2836        collection: String,
2837        /// The rkey to write at.
2838        rkey: String,
2839        /// The record body.
2840        value: Value,
2841    },
2842    /// Delete a record by collection + rkey.
2843    Delete {
2844        /// The collection NSID.
2845        collection: String,
2846        /// The rkey to delete.
2847        rkey: String,
2848    },
2849}
2850
2851impl WriteOp {
2852    /// Render this op as the tagged JSON `com.atproto.repo.applyWrites` expects.
2853    ///
2854    /// `pub(crate)` so [`crate::oauth::xrpc`] can build the same batch body.
2855    /// Sharing the rendering rather than reimplementing it is what keeps the two
2856    /// clients wire-identical across the cutover.
2857    pub(crate) fn to_json(&self) -> Value {
2858        match self {
2859            WriteOp::Create {
2860                collection,
2861                rkey,
2862                value,
2863            } => {
2864                let mut op = json!({
2865                    "$type": "com.atproto.repo.applyWrites#create",
2866                    "collection": collection,
2867                    "value": value,
2868                });
2869                if let Some(rkey) = rkey {
2870                    op["rkey"] = json!(rkey);
2871                }
2872                op
2873            }
2874            WriteOp::Update {
2875                collection,
2876                rkey,
2877                value,
2878            } => json!({
2879                "$type": "com.atproto.repo.applyWrites#update",
2880                "collection": collection,
2881                "rkey": rkey,
2882                "value": value,
2883            }),
2884            WriteOp::Delete { collection, rkey } => json!({
2885                "$type": "com.atproto.repo.applyWrites#delete",
2886                "collection": collection,
2887                "rkey": rkey,
2888            }),
2889        }
2890    }
2891
2892    /// Render this op in the shape the OAuth sidecar's `/internal/repo`
2893    /// `applyWrites` expects: `{action, collection, rkey?, value?}` (the sidecar
2894    /// maps `action` → the `com.atproto.repo.applyWrites#<kind>` union member).
2895    fn to_sidecar_json(&self) -> Value {
2896        match self {
2897            WriteOp::Create {
2898                collection,
2899                rkey,
2900                value,
2901            } => {
2902                let mut op = json!({
2903                    "action": "create",
2904                    "collection": collection,
2905                    "value": value,
2906                });
2907                if let Some(rkey) = rkey {
2908                    op["rkey"] = json!(rkey);
2909                }
2910                op
2911            }
2912            WriteOp::Update {
2913                collection,
2914                rkey,
2915                value,
2916            } => json!({
2917                "action": "update",
2918                "collection": collection,
2919                "rkey": rkey,
2920                "value": value,
2921            }),
2922            WriteOp::Delete { collection, rkey } => json!({
2923                "action": "delete",
2924                "collection": collection,
2925                "rkey": rkey,
2926            }),
2927        }
2928    }
2929}
2930
2931// ---------------------------------------------------------------------------
2932// TID rkeys (client-assigned, sortable, deterministic within a batch)
2933// ---------------------------------------------------------------------------
2934
2935/// The atproto base32-sortable alphabet (`s32`) — the digits/letters, minus the
2936/// ambiguous set, in **ascending** order so a bytewise string compare of two
2937/// TIDs matches their timestamp order.
2938const S32_ALPHABET: &[u8; 32] = b"234567abcdefghijklmnopqrstuvwxyz";
2939
2940/// A monotonic generator of atproto **TID** record keys.
2941///
2942/// A TID is a 13-char `s32`-encoded 64-bit integer: a 53-bit microsecond
2943/// timestamp in the high bits and a 10-bit "clock id" in the low bits (the top
2944/// bit is always 0). Encoded in the ascending `s32` alphabet, TIDs sort
2945/// lexicographically in creation order — which is exactly what we want for a
2946/// batched OPML import: assigning the rkeys ourselves keeps the imported feeds
2947/// in input order and makes [`add_subscriptions_bulk`](SidecarClient::add_subscriptions_bulk)
2948/// fully reproducible/testable without a live PDS.
2949///
2950/// Monotonicity within one generator is guaranteed by tracking the last value
2951/// and bumping to `last + 1` if the clock hasn't advanced — so a burst of
2952/// same-microsecond calls still yields strictly increasing, ordered rkeys.
2953pub(crate) struct TidGenerator {
2954    /// The last raw 64-bit TID value emitted (0 = none yet).
2955    last: u64,
2956    /// The low-10-bit clock id, randomized once per generator to avoid
2957    /// cross-instance collisions on the same microsecond.
2958    clock_id: u64,
2959}
2960
2961impl TidGenerator {
2962    /// A fresh generator with a per-instance clock id derived from the current
2963    /// nanosecond clock (no extra deps; uniqueness only needs to hold within a
2964    /// single import batch, and the timestamp bits carry the ordering).
2965    pub(crate) fn new() -> Self {
2966        let nanos = std::time::SystemTime::now()
2967            .duration_since(std::time::UNIX_EPOCH)
2968            .map(|d| d.subsec_nanos() as u64)
2969            .unwrap_or(0);
2970        Self {
2971            last: 0,
2972            clock_id: nanos & 0x3ff,
2973        }
2974    }
2975
2976    /// The next monotonic TID rkey (13 `s32` chars).
2977    pub(crate) fn next(&mut self) -> String {
2978        let micros = std::time::SystemTime::now()
2979            .duration_since(std::time::UNIX_EPOCH)
2980            .map(|d| d.as_micros() as u64)
2981            .unwrap_or(0);
2982        // Timestamp in bits 63..10 (top bit stays 0), clock id in bits 9..0.
2983        let mut raw = ((micros & 0x001f_ffff_ffff_ffff) << 10) | self.clock_id;
2984        if raw <= self.last {
2985            raw = self.last + 1;
2986        }
2987        self.last = raw;
2988        encode_s32_tid(raw)
2989    }
2990}
2991
2992/// Encode a 64-bit TID value as a 13-char big-endian `s32` string.
2993fn encode_s32_tid(mut v: u64) -> String {
2994    let mut buf = [0u8; 13];
2995    for slot in buf.iter_mut().rev() {
2996        *slot = S32_ALPHABET[(v & 0x1f) as usize];
2997        v >>= 5;
2998    }
2999    // 13 * 5 = 65 bits cover the 64-bit value; the leading char carries bits
3000    // 64..60, and bit 64 does not exist in a `u64` while bit 63 is always 0 in
3001    // a real TID, so the leading char is always one of the alphabet's first
3002    // eight symbols. Between 2005-09-05 and 2041-05-10 it is the second one,
3003    // which is why real TIDs all begin with `3`.
3004    String::from_utf8(buf.to_vec()).unwrap_or_default()
3005}
3006
3007/// The earliest instant a real TID can encode: 2020-01-01T00:00:00Z, in
3008/// microseconds.
3009///
3010/// atproto did not exist before this, so a "TID" decoding to earlier is a record
3011/// key that merely *looks* like one.
3012///
3013/// **This bound catches only the slugs that fall outside the window, and that
3014/// is a minority of them.** 13 lowercase alphanumerics is an ordinary slug
3015/// shape and also a valid `s32` value, and one beginning `3` decodes into the
3016/// last few years as readily as a real record key does: `3hoursinparis` reads
3017/// as 2020-11-24, `3ideasforjune` as 2021-08-12. Nothing in the string
3018/// distinguishes them — telling a slug from a TID would mean asking the PDS
3019/// when the record was written, which the listing does not report.
3020///
3021/// What the window does buy is that a mis-read date is always an ordinary past
3022/// instant rather than an unsweepable future one. That is worth having and it
3023/// is *not* harmless: a slug reading as 2020 is older than any realistic
3024/// retention window, so the row is swept, re-listed on the next poll, and
3025/// arrives unread again — the cycle this dating work narrows but does not
3026/// close. Refusing to insert what is already past the floor is what closes it,
3027/// for a mis-read slug and a genuine archive alike, and that belongs with the
3028/// retention floor rather than here.
3029const TID_FLOOR_MICROS: i64 = 1_577_836_800_000_000;
3030
3031/// How far ahead of our own clock a timestamp someone else authored may be and
3032/// still be believed.
3033///
3034/// A PDS a second or two fast would otherwise leave a brand-new document
3035/// undated until the following poll, and an undated row is the least visible
3036/// one in the reading list. Well under any interval that matters to retention
3037/// or the per-feed cap.
3038///
3039/// **Both date sources use it.** It began as a TID-only allowance, which left a
3040/// stated `publishedAt` judged against a bare `now` while the record key two
3041/// lines below got five minutes — the same clock, two different answers, for no
3042/// reason either comment could give.
3043pub(crate) const CLOCK_SKEW_GRACE_SECS: i64 = 300;
3044
3045/// Decode a 13-char `s32` TID rkey back to its raw 64-bit value.
3046///
3047/// The exact inverse of [`encode_s32_tid`] over the values a TID can hold.
3048///
3049/// `None` for anything that is not a 13-character `s32` value: wrong length, a
3050/// character outside the alphabet, or a value whose top bit is set. That last
3051/// rejection is stricter than the TID syntax regex, which admits leading `c`
3052/// through `j`; the spec's separate rule that the high bit is always 0 is the
3053/// one enforced here, and it keeps every decoded value inside the range
3054/// [`tid_timestamp`] can shift without loss.
3055///
3056/// **This does not decide whether the string is a TID**, only whether it is a
3057/// number. Thirteen lowercase alphanumerics is also an ordinary slug, and a
3058/// slug decodes as readily as a record key does. Refusing an implausible
3059/// instant is [`tid_timestamp`]'s job, and it is where that case is caught.
3060pub(crate) fn decode_s32_tid(rkey: &str) -> Option<u64> {
3061    if rkey.len() != 13 {
3062        return None;
3063    }
3064    let mut v: u64 = 0;
3065    for b in rkey.bytes() {
3066        let digit = S32_ALPHABET.iter().position(|c| *c == b)? as u64;
3067        // `checked_*` rather than shifting: 13 chars carry 65 bits, so the
3068        // largest 13-char string overflows a `u64` and must read as "not a
3069        // TID" instead of wrapping to a plausible-looking value.
3070        v = v.checked_mul(32)?.checked_add(digit)?;
3071    }
3072    (v >> 63 == 0).then_some(v)
3073}
3074
3075/// The instant a TID rkey encodes, or `None` if the rkey is not a plausible
3076/// TID.
3077///
3078/// **Bounded at both ends on purpose.** A TID's timestamp is minted from the
3079/// writer's clock, so one decoding far into the future is either a broken clock
3080/// or a slug that happens to be 13 `s32` characters; one decoding to before
3081/// [`TID_FLOOR_MICROS`] predates atproto. Neither is a date worth trusting, and
3082/// the caller's fallback for "no date" is safer than a wrong one.
3083///
3084/// The bounds are not a slug detector — see [`TID_FLOOR_MICROS`] for why they
3085/// cannot be, and for what they do guarantee instead.
3086pub(crate) fn tid_timestamp(rkey: &str) -> Option<chrono::DateTime<chrono::Utc>> {
3087    // The low 10 bits are the clock id; the rest is microseconds since the
3088    // epoch, and clearing bit 63 above bounds it well inside `i64`.
3089    let micros = i64::try_from(decode_s32_tid(rkey)? >> 10).ok()?;
3090    if micros < TID_FLOOR_MICROS {
3091        return None;
3092    }
3093    let at = chrono::DateTime::from_timestamp_micros(micros)?;
3094    let ceiling = chrono::Utc::now() + chrono::Duration::seconds(CLOCK_SKEW_GRACE_SECS);
3095    (at <= ceiling).then_some(at)
3096}
3097
3098// ---------------------------------------------------------------------------
3099// XRPC error helper
3100// ---------------------------------------------------------------------------
3101
3102/// Minimal percent-encoding for a query-string component.
3103///
3104/// Encodes everything outside the RFC 3986 unreserved set, which covers the
3105/// values FeatherReader passes (DIDs like `did:plc:…`, NSIDs, opaque cursors,
3106/// handles) without pulling in the optional reqwest `url`/`query` feature.
3107///
3108/// `pub(crate)` so [`crate::network`] builds its relay query strings the same
3109/// way rather than keeping a second copy of the escape table.
3110pub(crate) fn urlencode(s: &str) -> String {
3111    let mut out = String::with_capacity(s.len());
3112    for b in s.bytes() {
3113        match b {
3114            b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => {
3115                out.push(b as char)
3116            }
3117            _ => out.push_str(&format!("%{b:02X}")),
3118        }
3119    }
3120    out
3121}
3122
3123/// The most nodes a `listRecords` body may ask us to build.
3124///
3125/// **A bound on the parse, checked before the parse.** Every other limit here is
3126/// consulted after `serde_json` has already materialised the page, which cannot
3127/// prevent the allocation it exists to prevent: one 8 MB response of `{"":0}`
3128/// objects was measured retaining 824 MB, a 98x wire-to-heap amplification, on a
3129/// 512 MB box. A record cap does not see it — the page holds one record. A page
3130/// cap does not see it — there is one request. A byte budget does not see it
3131/// until the memory is already spent.
3132///
3133/// **640 000, measured — not two million, which this file's own arithmetic
3134/// already contradicted.** An earlier version reasoned "32 bytes a node plus
3135/// slack, so two million is about 128 MB". That is the model `json_bytes` two
3136/// hundred lines above explicitly rejects: it charges `MAP_NODE` + `MAP_ENTRY` +
3137/// `SLOT` = 680 bytes for a single-entry object, which costs three counted
3138/// characters. Measured against a counting allocator, the worst shape reaches
3139/// **210 bytes per counted character**, so two million admitted **400 MB** — a
3140/// bound that let through more than the attack it was written to stop, and that
3141/// the walk's own byte budget then refused a step later.
3142///
3143/// 640 000 x 210 B is about 128 MiB, which is the figure the walk budget uses
3144/// and the one this claims.
3145///
3146/// **Re-measured when a review put the worst shape at 221 B; it does not
3147/// reproduce.** Sweeping nesting depths 10, 50, 100 and 120 against the same
3148/// counting allocator, the worst is 210.6 B per counted character (at depth 120)
3149/// and peak equals retained — `serde_json` overshoots by nothing measurable
3150/// while it builds. Depth cannot be pushed further to raise the ratio, either:
3151/// `serde_json`'s own recursion limit of 128 refuses a deeper body outright,
3152/// before this guard would even matter.
3153///
3154/// **The floor is real traffic, not comfort.** The densest legitimate page is a
3155/// full `readState` listing — 100 records each carrying two arrays of
3156/// [`crate::lexicon::ReadState::MAX_IDS`] ids — which counts 403 003. So the cap
3157/// sits above the densest page the lexicons permit, and a test holds it there.
3158///
3159/// Ordinary traffic is nowhere near either number: a page of 100 real-sized
3160/// standard.site documents (seven fields, a 15 kB `textContent`, 1.5 MB on the
3161/// wire) counts **4 003**. An earlier version of this line said 40 000, which was
3162/// wrong by an order of magnitude in the direction that makes the cap look tighter
3163/// than it is; the test that was supposed to hold it served a single record and
3164/// would have passed with the cap set to 1 000.
3165pub(crate) const MAX_LIST_STRUCTURAL_CHARS: usize = 640_000;
3166
3167/// How many nodes `body` would parse into, to within one, without parsing it.
3168///
3169/// **A lower bound, despite what an earlier name said.** `[1,2,3]` counts three
3170/// — one `[` and two commas — and builds four values. The deficit is never more
3171/// than one (verified exhaustively over every body of length 1-5 from a JSON
3172/// alphabet), because every node but the outermost is introduced by one of the
3173/// characters counted here. That is the direction a guard needs: it can
3174/// under-count by one and still refuse everything it must.
3175///
3176/// Counts the structural characters that introduce a value — `{`, `[`, `,`, `:`
3177/// — **outside strings**, which is what makes this sound: a node cannot appear
3178/// without one, and a string's contents cannot invent one. Skipping strings is
3179/// the whole difficulty; counting naively would refuse a legitimate article that
3180/// happens to contain a million commas.
3181pub(crate) fn count_structural_chars(body: &[u8]) -> usize {
3182    let mut nodes = 0usize;
3183    let mut in_string = false;
3184    let mut escaped = false;
3185    for &b in body {
3186        if in_string {
3187            // `\"` stays inside the string; `\\` does not escape the quote that
3188            // follows it. Getting this pair wrong makes the scan count a whole
3189            // document as structure, or none of it.
3190            if escaped {
3191                escaped = false;
3192            } else if b == b'\\' {
3193                escaped = true;
3194            } else if b == b'"' {
3195                in_string = false;
3196            }
3197            continue;
3198        }
3199        match b {
3200            // A string is a node, and everything inside it is not.
3201            b'"' => {
3202                in_string = true;
3203                nodes += 1;
3204            }
3205            b'{' | b'[' | b',' | b':' => nodes += 1,
3206            _ => {}
3207        }
3208    }
3209    nodes
3210}
3211
3212/// Refuse a body carrying more JSON structure than
3213/// [`MAX_LIST_STRUCTURAL_CHARS`].
3214pub(crate) fn refuse_a_structure_explosion(body: &[u8], what: &str) -> Result<()> {
3215    let counted = count_structural_chars(body);
3216    anyhow::ensure!(
3217        counted <= MAX_LIST_STRUCTURAL_CHARS,
3218        "{what} counts at least {counted} structural characters, over the \
3219         {MAX_LIST_STRUCTURAL_CHARS} cap — refusing before parsing it"
3220    );
3221    Ok(())
3222}
3223
3224/// Parse a `listRecords` body, refusing an error envelope that arrived on a 2xx.
3225///
3226/// Some PDS implementations answer 200 for application failures, and the status
3227/// check in the caller cannot see those. Without the guard, `{"error","message"}`
3228/// deserialises as a page with no records — so a walk over a stranger's
3229/// collection returns a healthy, empty result in place of an error, and for the
3230/// walk that feeds `replace_sub_refs` that is revoked access rather than an empty
3231/// repo. The guard itself lives in [`page_from_body`], which every client shares.
3232pub(crate) fn parse_list_records(body: &[u8]) -> Result<ListRecordsResponse> {
3233    // **An empty body is the "unexpected body" case, not a parse error.** Reading
3234    // bytes reaches it as "EOF while parsing", where the OAuth client used to
3235    // reach it as "no records field" (its `send` mapped an empty 2xx to
3236    // `Value::Null`) and the direct client reached it as "EOF" too. Refused
3237    // either way, so this is a unification rather than a preservation — nothing
3238    // outside the tests matches on the text, and `resolve_subscriptions` fails
3239    // closed on any `Err`. It is for whoever reads the log.
3240    if body.is_empty() {
3241        return Err(UnreadableListing::NoRecords.into());
3242    }
3243    refuse_a_structure_explosion(body, "the listRecords body")?;
3244    let parsed: ListRecordsBody =
3245        serde_json::from_slice(body).context("parsing listRecords response")?;
3246    let mut page = page_from_body(parsed)?;
3247    page.wire_bytes = body.len();
3248    Ok(page)
3249}
3250
3251/// Apply both invariants to an already-deserialised body.
3252///
3253/// **The one place the guards live, for all three clients.** They were added a
3254/// client at a time twice over, which is the whole reason a shared function
3255/// exists; splitting the sidecar onto a different route would have started that
3256/// again, so it deserialises into this same struct.
3257fn page_from_body(parsed: ListRecordsBody) -> Result<ListRecordsResponse> {
3258    if let Some(error) = parsed.error.as_ref().and_then(envelope_error_name) {
3259        let message = parsed
3260            .message
3261            .as_ref()
3262            .and_then(Value::as_str)
3263            .map(truncate_for_message);
3264        return Err(UnreadableListing::ErrorEnvelope { error, message }.into());
3265    }
3266    let entries = parsed.records.ok_or(UnreadableListing::NoRecords)?;
3267    let mut records = Vec::with_capacity(entries.len());
3268    let mut malformed = 0;
3269    for entry in entries {
3270        match entry {
3271            MaybeRecord::Record(r) => records.push(r),
3272            MaybeRecord::Malformed(_) => malformed += 1,
3273        }
3274    }
3275    Ok(ListRecordsResponse {
3276        records,
3277        cursor: parsed.cursor,
3278        malformed,
3279        wire_bytes: 0,
3280    })
3281}
3282
3283/// The wire shape of a `listRecords` body, read in **one** pass.
3284///
3285/// **Parsing to `Value` and then into the struct materialises the page twice.**
3286/// `serde_json::from_value` rebuilds rather than moves, so an 8 MB response was
3287/// measured holding both copies at once — a peak of roughly double the retained
3288/// size, reached before any accounting the caller does, which is why no budget
3289/// charged after the parse can cover it.
3290///
3291/// The two invariants that used to live on a `Value` are
3292/// expressed here as fields instead of lookups, and mean exactly what they did:
3293/// an `error` present on a 2xx is a failure, not an empty page, and `records`
3294/// ABSENT is not `records` empty.
3295#[derive(Debug, Default)]
3296struct ListRecordsBody {
3297    /// A `Value`, not a `String`. Typing it as a string made
3298    /// `{"error":404,"records":[]}` fail as "invalid type: integer" rather than
3299    /// as an envelope — the wrong reason for the exact shape the guard exists
3300    /// for, and the guard's whole point is that this distinction is load-bearing.
3301    error: Option<Value>,
3302    /// Likewise, and for a duller reason: `message` carries no security role,
3303    /// and typing it as a string made a PDS that stamps a non-string one onto an
3304    /// otherwise good page of a thousand records fail the entire listing.
3305    message: Option<Value>,
3306    /// `None` means the field was absent — what a proxy makes of an empty or
3307    /// unexpected upstream body. `Some(vec![])` is a genuine empty page.
3308    records: Option<Vec<MaybeRecord>>,
3309    cursor: Option<String>,
3310}
3311
3312/// One element of a `listRecords` page: a record, or something that is not one.
3313///
3314/// **Parsed per record, so one malformed envelope costs that record and not
3315/// the page** (#177). `RecordEntry.uri` is required, and the page used to be
3316/// parsed in one `from_value`, so a single `{"cid":…,"value":{}}` failed every
3317/// record beside it. Whoever reads the page decides what a skipped record
3318/// means: a stranger's publication skips it, a reader's own repo refuses.
3319#[derive(Debug, Deserialize)]
3320#[serde(untagged)]
3321enum MaybeRecord {
3322    Record(RecordEntry),
3323    Malformed(serde::de::IgnoredAny),
3324}
3325
3326/// Refuse a page that skipped records, for a walk that must not drop any.
3327///
3328/// Every walk whose result reaches `store::replace_sub_refs` calls this: a
3329/// record left out there is a subscription silently removed.
3330/// What a walk does with a record whose envelope is malformed (#177).
3331#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3332enum OnMalformed {
3333    /// Refuse the walk with [`MalformedRecords`]: the result reaches
3334    /// `replace_sub_refs`, where a skipped record is a dropped subscription.
3335    Refuse,
3336    /// Skip and count it: a stranger's repo, where one bad record must not
3337    /// stall everything beside it.
3338    Skip,
3339}
3340
3341pub(crate) fn refuse_malformed(page: &ListRecordsResponse, collection: &str) -> Result<()> {
3342    if page.malformed > 0 {
3343        return Err(MalformedRecords {
3344            collection: collection.to_string(),
3345            count: page.malformed,
3346        }
3347        .into());
3348    }
3349    Ok(())
3350}
3351
3352/// **Hand-written, because the derive accepts a listing that is not an object.**
3353///
3354/// serde's derived `Deserialize` takes a struct in POSITIONAL form as well as
3355/// map form, so with every field defaulted the fourteen bytes `[null,null,[]]`
3356/// bound `records` to an empty vector and read as a healthy page — on all three
3357/// clients, and the `Value` route this replaced refused it, because
3358/// `Value::Array::get("records")` is always `None`. Neither the envelope guard
3359/// nor the duplicated-key refusal can fire on a body with no keys at all, so one
3360/// short array defeated every protection here at once and reached
3361/// `replace_sub_refs`, which deletes the reader's whole subscription projection.
3362///
3363/// **Unknown fields are read, not skipped.** `IgnoredAny` does not validate what
3364/// it skips, so `{"records":[],"x":"<invalid utf-8>"}` — not valid JSON at all —
3365/// also read as a healthy empty page where the `Value` route refused it. Reading
3366/// the value into a `Value` and dropping it costs an allocation on a field nobody
3367/// wants, and buys back the validation.
3368impl<'de> Deserialize<'de> for ListRecordsBody {
3369    fn deserialize<D: serde::Deserializer<'de>>(d: D) -> std::result::Result<Self, D::Error> {
3370        struct AsMap;
3371        impl<'de> serde::de::Visitor<'de> for AsMap {
3372            type Value = ListRecordsBody;
3373            fn expecting(&self, f: &mut std::fmt::Formatter) -> std::fmt::Result {
3374                f.write_str("a listRecords object")
3375            }
3376            fn visit_map<M: serde::de::MapAccess<'de>>(
3377                self,
3378                mut map: M,
3379            ) -> std::result::Result<ListRecordsBody, M::Error> {
3380                use serde::de::Error;
3381                let mut out = ListRecordsBody::default();
3382                let (mut error, mut message, mut records, mut cursor) =
3383                    (false, false, false, false);
3384                while let Some(key) = map.next_key::<String>()? {
3385                    let seen = match key.as_str() {
3386                        "error" => std::mem::replace(&mut error, true),
3387                        "message" => std::mem::replace(&mut message, true),
3388                        "records" => std::mem::replace(&mut records, true),
3389                        "cursor" => std::mem::replace(&mut cursor, true),
3390                        _ => false,
3391                    };
3392                    if seen {
3393                        // A repeated key is last-wins in a `Value`, which is how
3394                        // a smuggled second, empty `records` array read as a
3395                        // successful page. Refused here.
3396                        return Err(M::Error::duplicate_field(match key.as_str() {
3397                            "error" => "error",
3398                            "message" => "message",
3399                            "records" => "records",
3400                            _ => "cursor",
3401                        }));
3402                    }
3403                    match key.as_str() {
3404                        "error" => out.error = Some(map.next_value()?),
3405                        "message" => out.message = Some(map.next_value()?),
3406                        "records" => out.records = Some(map.next_value()?),
3407                        "cursor" => out.cursor = map.next_value()?,
3408                        _ => {
3409                            let _validated: Value = map.next_value()?;
3410                        }
3411                    }
3412                }
3413                Ok(out)
3414            }
3415        }
3416        d.deserialize_map(AsMap)
3417    }
3418}
3419
3420/// Refuse an atproto error envelope that arrived on a 2xx.
3421///
3422/// **No listing reaches this any more — [`page_from_body`] is the one every
3423/// client shares.** It survives for the WRITE paths, where the response is still
3424/// a `Value`: `deleteRecord` and `applyWrites` on both live clients.
3425///
3426/// The history is worth keeping, because it is why a shared function exists at
3427/// all. Each client used to take `records` off the JSON its own way — the live
3428/// one with `unwrap_or(Array([]))`, the sidecar through a defaulted `Value` — and
3429/// each turned `200 {"error": …}` into `Ok(empty)`. That is not the fail-closed
3430/// branch in `web::resolve_subscriptions`: `sync_sub_refs` wrote the empty set
3431/// and `replace_sub_refs` DELETEd the DID's entire `sub_ref` projection. One bad
3432/// response revoked a reader's access to every feed they had. The guard was added
3433/// to one client at a time, twice, which is the drift a single function prevents
3434/// — and why the listing guard now lives in exactly one place rather than here.
3435pub(crate) fn reject_error_envelope(value: &Value) -> Result<()> {
3436    let Some(error) = value.get("error").and_then(envelope_error_name) else {
3437        return Ok(());
3438    };
3439    let message = value
3440        .get("message")
3441        .and_then(Value::as_str)
3442        .map(truncate_for_message);
3443    Err(UnreadableListing::ErrorEnvelope { error, message }.into())
3444}
3445
3446/// The name in an `error` field, or `None` when the field does not denote one.
3447///
3448/// **Whatever its type.** Keying on `as_str` meant a PDS answering
3449/// `{"error":404,"records":[]}` — or `{}`, or `[]` — passed the guard and read as
3450/// a healthy empty page, the shape that makes `replace_sub_refs` delete every
3451/// `sub_ref` a reader has. A non-string `error` is not a well-formed envelope,
3452/// but it is certainly not a successful listing either.
3453///
3454/// **Except the four spellings of "no error".** Absent and `null` are what an
3455/// ordinary listing carries; `false` and `0` are a convention proxies use, and
3456/// treating those as envelopes turns a good page of a thousand records into a
3457/// hard refusal, which on these walks means the reader's sidebar degrades to a
3458/// stale projection on every request.
3459///
3460/// **Bounded.** The name reaches a `warn!` that also logs the DID, and the value
3461/// is attacker-chosen: a PDS answering with hundreds of kilobytes under `error`
3462/// would otherwise put all of it in the log and allocate another copy, in code
3463/// whose purpose is cutting peak allocation.
3464fn envelope_error_name(error: &Value) -> Option<String> {
3465    match error {
3466        Value::Null | Value::Bool(false) => None,
3467        // Integer zero only. `as_f64() == Some(0.0)` also matched `-0`, `0.0`
3468        // and anything that underflows, so `1e-400` was "no error".
3469        Value::Number(n) if n.as_i64() == Some(0) || n.as_u64() == Some(0) => None,
3470
3471        Value::String(s) => Some(truncate_for_message(s)),
3472        // **The type, not the value.** `to_string()` would serialise the whole
3473        // attacker-chosen subtree before truncating it, allocating a full extra
3474        // copy of up to the body cap — in code whose purpose is cutting peak
3475        // allocation. A non-string `error` is malformed, so its contents tell a
3476        // reader nothing its shape does not.
3477        Value::Bool(_) => Some("<non-string error: bool>".to_string()),
3478        Value::Number(_) => Some("<non-string error: number>".to_string()),
3479        Value::Array(_) => Some("<non-string error: array>".to_string()),
3480        Value::Object(_) => Some("<non-string error: object>".to_string()),
3481    }
3482}
3483
3484/// Cap a string destined for an error message at a readable length.
3485fn truncate_for_message(s: &str) -> String {
3486    /// **Bytes, not characters.** A log line is bytes, and counting characters
3487    /// let astral-plane code points render four times the intended bound.
3488    const MAX_BYTES: usize = 120;
3489    if s.len() <= MAX_BYTES {
3490        return s.to_string();
3491    }
3492    let cut = s
3493        .char_indices()
3494        .map(|(i, _)| i)
3495        .take_while(|i| *i <= MAX_BYTES)
3496        .last()
3497        .unwrap_or(0);
3498    format!("{}… ({} bytes)", &s[..cut], s.len())
3499}
3500
3501/// The atproto XRPC error envelope body: `{"error": "...", "message": "..."}`.
3502#[derive(Debug, Deserialize)]
3503struct XrpcErrorBody {
3504    #[serde(default)]
3505    error: Option<String>,
3506    #[serde(default)]
3507    message: Option<String>,
3508}
3509
3510/// Consume a non-2xx response into a typed [`AtProtoError::Xrpc`], parsing the
3511/// atproto error envelope when present (falling back to `"Unknown"`).
3512///
3513/// The body is read through [`crate::net::read_capped`], **not** `resp.json()`.
3514/// Every guarded call caps its success body; routing the error body through
3515/// `resp.json()` would have left a hole exactly where the hostile-PDS threat
3516/// model points — reqwest decompresses gzip before deserialising, so a `400`
3517/// carrying a decompression bomb was an unbounded allocation on a 512 MB box.
3518/// A body we cannot read (over-cap, transport error) degrades to `"Unknown"`,
3519/// which is the same fallback an unparseable envelope already took.
3520async fn xrpc_error_from(resp: reqwest::Response) -> AtProtoError {
3521    let status = resp.status();
3522    let (error, message) = match crate::net::read_capped(resp).await {
3523        Ok(raw) => match serde_json::from_slice::<XrpcErrorBody>(&raw) {
3524            Ok(body) => (
3525                body.error.unwrap_or_else(|| "Unknown".to_string()),
3526                body.message,
3527            ),
3528            Err(_) => ("Unknown".to_string(), None),
3529        },
3530        Err(_) => ("Unknown".to_string(), None),
3531    };
3532    AtProtoError::Xrpc {
3533        status,
3534        error,
3535        message,
3536    }
3537}
3538
3539// ---------------------------------------------------------------------------
3540// Tests — record (de)serialization against a repo listRecords response shape.
3541// No network.
3542// ---------------------------------------------------------------------------
3543
3544#[cfg(test)]
3545pub(crate) mod tests {
3546    use super::*;
3547
3548    /// **Regression (v0.2.8 review).** Every guarded call caps its *success*
3549    /// body via `read_capped`, but the non-2xx branch went through
3550    /// `resp.json::<XrpcErrorBody>()` — unbounded, and with reqwest's gzip
3551    /// decompression in front of it. That left a hole precisely where the
3552    /// module's own threat model points: a hostile or DNS-rebound PDS answers
3553    /// `400` with a decompression bomb and gets an unbounded allocation on a
3554    /// 512 MB box. Both this PR's review passes checked the success path and
3555    /// walked past the error path, so the cap is asserted here explicitly.
3556    ///
3557    /// Fetched directly rather than through the guard, which rightly refuses
3558    /// loopback — the same reason `net::tests::read_capped_rejects_over_cap_body`
3559    /// bypasses it. The stub answers 200; `xrpc_error_from` reads the status only
3560    /// to record it, so the body handling under test is identical.
3561    #[tokio::test]
3562    async fn xrpc_error_body_is_capped() {
3563        // A syntactically VALID envelope, one byte past the cap. If the body were
3564        // parsed unbounded this would deserialize and yield "TooBig"; capped, it
3565        // is refused unread and degrades to the "Unknown" fallback.
3566        let filler = "x".repeat(crate::net::MAX_BODY_BYTES);
3567        let big = format!(r#"{{"error":"TooBig","message":"{filler}"}}"#).into_bytes();
3568        assert!(big.len() > crate::net::MAX_BODY_BYTES);
3569
3570        let base = crate::net::tests::serve_body(big).await;
3571        let resp = reqwest::Client::builder()
3572            .build()
3573            .unwrap()
3574            .get(&base)
3575            .send()
3576            .await
3577            .unwrap();
3578
3579        match xrpc_error_from(resp).await {
3580            AtProtoError::Xrpc { error, message, .. } => {
3581                assert_eq!(error, "Unknown", "an over-cap error body must not parse");
3582                assert!(message.is_none());
3583            }
3584            other => panic!("expected Xrpc, got {other:?}"),
3585        }
3586    }
3587
3588    /// The other half: a normal-sized envelope still parses, so capping the
3589    /// error path did not cost the diagnostics it exists to provide.
3590    #[tokio::test]
3591    async fn xrpc_error_body_within_the_cap_still_parses() {
3592        let base = crate::net::tests::serve_body(
3593            br#"{"error":"InvalidRequest","message":"bad rkey"}"#.to_vec(),
3594        )
3595        .await;
3596        let resp = reqwest::Client::builder()
3597            .build()
3598            .unwrap()
3599            .get(&base)
3600            .send()
3601            .await
3602            .unwrap();
3603
3604        match xrpc_error_from(resp).await {
3605            AtProtoError::Xrpc { error, message, .. } => {
3606                assert_eq!(error, "InvalidRequest");
3607                assert_eq!(message.as_deref(), Some("bad rkey"));
3608            }
3609            other => panic!("expected Xrpc, got {other:?}"),
3610        }
3611    }
3612
3613    /// A realistic `com.atproto.repo.listRecords` response for the subscription
3614    /// collection, as a PDS returns it — the envelope wraps each record in
3615    /// `{uri, cid, value}` and the record `value` carries its `$type`.
3616    fn subscription_list_json() -> Value {
3617        json!({
3618            "records": [
3619                {
3620                    "uri": "at://did:plc:abc123/community.lexicon.rss.subscription/3ksub0001",
3621                    "cid": "bafyreisubone",
3622                    "value": {
3623                        "$type": "community.lexicon.rss.subscription",
3624                        "url": "https://example.com/feed.xml",
3625                        "title": "Example Blog",
3626                        "siteUrl": "https://example.com/",
3627                        "fetchHint": "hourly",
3628                        "createdAt": "2026-07-12T00:00:00.000Z"
3629                    }
3630                },
3631                {
3632                    "uri": "at://did:plc:abc123/community.lexicon.rss.subscription/3ksub0002",
3633                    "cid": "bafyreisubtwo",
3634                    "value": {
3635                        "$type": "community.lexicon.rss.subscription",
3636                        "url": "https://blog.example.org/atom.xml",
3637                        "createdAt": "2026-07-11T12:00:00.000Z"
3638                    }
3639                }
3640            ],
3641            "cursor": "3ksub0002"
3642        })
3643    }
3644
3645    /// **A big archive is truncated, not refused.** `extend_bounded` bails on
3646    /// its cap, which is right for the `sub_ref` walk (a short list there is
3647    /// revoked access) and wrong for an additive read: a publication with more
3648    /// documents than the cap would return `Err` on every poll — permanently
3649    /// unreadable rather than partially read. 2 000 posts is an ordinary
3650    /// figure for a long-running blog.
3651    #[test]
3652    fn a_reading_walk_truncates_where_the_sub_ref_walk_refuses() {
3653        let page = |n: usize| -> Vec<RecordEntry> {
3654            (0..n)
3655                .map(|i| RecordEntry {
3656                    uri: format!("at://did:plc:x/c/{i}"),
3657                    cid: None,
3658                    value: Value::Null,
3659                })
3660                .collect()
3661        };
3662        let mut out = Vec::new();
3663        assert!(!extend_truncating(&mut out, page(2), 3), "not full yet");
3664        assert_eq!(out.len(), 2);
3665        // The page that overshoots contributes what fits, and says "stop".
3666        assert!(extend_truncating(&mut out, page(5), 3), "must report full");
3667        assert_eq!(out.len(), 3, "a reading walk must keep what fits");
3668        // The same overshoot is a hard error on the fail-closed path.
3669        let mut refused = Vec::new();
3670        assert!(extend_bounded(&mut refused, page(5), 3, "c").is_err());
3671        assert!(refused.is_empty(), "a refusal must leave nothing behind");
3672    }
3673
3674    /// **The cap counts the records the caller KEEPS, not the ones the repo
3675    /// holds.** A repo-wide cap applied before the caller's filter starves a
3676    /// quiet publication whose busy sibling fills the window: poll it, walk
3677    /// the newest 2 000 documents, discard all of them as the sibling's,
3678    /// return nothing — permanently, and worse with every post the sibling
3679    /// makes. The walk pages on until it has `max` MATCHING records (still
3680    /// bounded by `MAX_LIST_PAGES` requests).
3681    #[tokio::test]
3682    async fn the_cap_counts_matching_records_not_walked_ones() {
3683        // Every page: 4 records, only the last of which the caller wants.
3684        let records: Vec<Value> = (0..4)
3685            .map(|i| {
3686                serde_json::json!({
3687                    "uri": format!("at://did:plc:x/c/{i}"),
3688                    "value": {"mine": i == 3}
3689                })
3690            })
3691            .collect();
3692        let body = serde_json::json!({ "records": records, "cursor": serde_json::Value::Null })
3693            .to_string();
3694        let base = crate::net::tests::serve_body(body.into_bytes()).await;
3695        let port: u16 = base
3696            .trim_end_matches('/')
3697            .rsplit(':')
3698            .next()
3699            .unwrap()
3700            .parse()
3701            .unwrap();
3702        crate::net::test_host_override(
3703            "matching-pds.test",
3704            std::net::SocketAddr::from(([127, 0, 0, 1], port)),
3705        );
3706        let client = PdsClient::anonymous(
3707            ssrf_test_client(),
3708            format!("http://matching-pds.test:{port}"),
3709            "did:plc:x",
3710        );
3711
3712        let kept = client
3713            .list_recent_matching("site.standard.document", 3, 100, |r| {
3714                r.value
3715                    .get("mine")
3716                    .and_then(Value::as_bool)
3717                    .unwrap_or(false)
3718            })
3719            .await
3720            .expect("walk failed")
3721            .records;
3722        // One page, no cursor: one match survives. The point is that the three
3723        // non-matching records did NOT consume the cap.
3724        assert_eq!(kept.len(), 1, "the filter ran after the cap, not before it");
3725    }
3726
3727    /// **A walk that stopped early says so.** Landing exactly on the cap, or
3728    /// running out of page budget, returns the same short `Vec` as a small
3729    /// collection — and the caller cannot tell them apart afterwards. That
3730    /// silence is how the starvation this walk exists to prevent came back one
3731    /// order of magnitude further out: a quiet publication whose busy sibling
3732    /// fills every page returns nothing, forever, looking healthy.
3733    #[tokio::test]
3734    async fn a_walk_that_stops_early_reports_itself_incomplete() {
3735        let records: Vec<Value> = (0..2)
3736            .map(|i| serde_json::json!({"uri": format!("at://did:plc:x/c/{i}"), "value": {}}))
3737            .collect();
3738        // Every page is full AND advertises another — the shape that lands on
3739        // the cap with the collection still going.
3740        let body = serde_json::json!({ "records": records, "cursor": "next" }).to_string();
3741        let base = crate::net::tests::serve_body(body.into_bytes()).await;
3742        let port: u16 = base
3743            .trim_end_matches('/')
3744            .rsplit(':')
3745            .next()
3746            .unwrap()
3747            .parse()
3748            .unwrap();
3749        crate::net::test_host_override(
3750            "incomplete-pds.test",
3751            std::net::SocketAddr::from(([127, 0, 0, 1], port)),
3752        );
3753        let client = PdsClient::anonymous(
3754            ssrf_test_client(),
3755            format!("http://incomplete-pds.test:{port}"),
3756            "did:plc:x",
3757        );
3758
3759        let walk = client
3760            .list_recent_matching("c", 2, 100, |_| true)
3761            .await
3762            .expect("walk failed");
3763        assert_eq!(walk.records.len(), 2);
3764        assert!(
3765            !walk.complete,
3766            "a walk that filled its cap with pages still to come called itself complete"
3767        );
3768    }
3769
3770    /// **A repeated cursor is not the end of the collection (#203).** The
3771    /// stranger-repo walk keeps what it read rather than refusing, but a server
3772    /// that repeated its cursor on a non-empty page said it had more: the walk
3773    /// is partial, not complete.
3774    #[tokio::test]
3775    async fn a_walk_ended_by_a_repeated_cursor_reports_itself_incomplete() {
3776        let body = serde_json::json!({
3777            "records": [{"uri": "at://did:plc:x/c/1", "value": {}}],
3778            "cursor": "same-every-time",
3779        })
3780        .to_string();
3781        let (base, _) = host_for(vec![body.into_bytes()], "repeat-partial.test").await;
3782        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
3783        let walk = client
3784            .list_recent_matching("c", 100, 100, |_| true)
3785            .await
3786            .expect("walk failed");
3787        assert!(
3788            !walk.complete,
3789            "a walk the server cut off with a repeated cursor called itself complete"
3790        );
3791    }
3792
3793    /// The other side: a collection that runs out IS complete, so the caller
3794    /// does not warn about every ordinary small publication.
3795    #[tokio::test]
3796    async fn a_walk_that_exhausts_the_collection_reports_itself_complete() {
3797        let body = serde_json::json!({
3798            "records": [{"uri": "at://did:plc:x/c/1", "value": {}}]
3799        })
3800        .to_string();
3801        let base = crate::net::tests::serve_body(body.into_bytes()).await;
3802        let port: u16 = base
3803            .trim_end_matches('/')
3804            .rsplit(':')
3805            .next()
3806            .unwrap()
3807            .parse()
3808            .unwrap();
3809        crate::net::test_host_override(
3810            "complete-pds.test",
3811            std::net::SocketAddr::from(([127, 0, 0, 1], port)),
3812        );
3813        let client = PdsClient::anonymous(
3814            ssrf_test_client(),
3815            format!("http://complete-pds.test:{port}"),
3816            "did:plc:x",
3817        );
3818
3819        let walk = client
3820            .list_recent_matching("c", 100, 100, |_| true)
3821            .await
3822            .expect("walk failed");
3823        assert_eq!(walk.records.len(), 1);
3824        assert!(walk.complete, "an exhausted collection is a complete read");
3825    }
3826
3827    /// **`{}` is not a page of zero records.** The records-presence guard
3828    /// landed on the OAuth client first; a proxy answering
3829    /// `{"ok":true,"data":{}}` kept the same `sub_ref`-wipe open on the
3830    /// sidecar path, and `{}` from a stranger's PDS made an empty publication
3831    /// look healthy.
3832    #[test]
3833    fn a_body_without_a_records_field_is_not_an_empty_page() {
3834        let err =
3835            parse_list_records(br#"{}"#).expect_err("`{}` was read as a page of zero records");
3836        assert!(format!("{err:#}").contains("no records field"), "{err:#}");
3837        assert_eq!(
3838            crate::feed::publication_failure_kind(&err),
3839            crate::feed::FailureKind::Parse,
3840            "{err:#}"
3841        );
3842        let err = parse_list_records(br#"{"cursor":"c"}"#)
3843            .expect_err("a cursor-only body was read as a page");
3844        assert!(format!("{err:#}").contains("no records field"), "{err:#}");
3845        assert_eq!(
3846            crate::feed::publication_failure_kind(&err),
3847            crate::feed::FailureKind::Parse,
3848            "{err:#}"
3849        );
3850        // #227: an empty body is the same refusal, and files the same way.
3851        let err = parse_list_records(b"").expect_err("an empty body was read as a page");
3852        assert_eq!(
3853            crate::feed::publication_failure_kind(&err),
3854            crate::feed::FailureKind::Parse,
3855            "{err:#}"
3856        );
3857        let page = parse_list_records(br#"{"records":[]}"#).unwrap();
3858        assert!(page.records.is_empty());
3859    }
3860
3861    /// Serve one oversized-but-well-formed body and point `host` at it.
3862    async fn serve_oversized(host: &str, shape: &str) -> String {
3863        let filler = "x".repeat(crate::net::MAX_BODY_BYTES);
3864        let body = shape.replace("PAD", &filler);
3865        assert!(body.len() > crate::net::MAX_BODY_BYTES);
3866        let base = crate::net::tests::serve_body(body.into_bytes()).await;
3867        let port: u16 = base
3868            .trim_end_matches('/')
3869            .rsplit(':')
3870            .next()
3871            .unwrap()
3872            .parse()
3873            .unwrap();
3874        crate::net::test_host_override(host, std::net::SocketAddr::from(([127, 0, 0, 1], port)));
3875        format!("http://{host}:{port}")
3876    }
3877
3878    /// **The DID document is the most remote-controlled body of the lot.**
3879    ///
3880    /// For a `did:web:` the host comes straight out of the DID, so whoever
3881    /// supplies the DID chooses the server. The SSRF guard proves the address
3882    /// is public; it says nothing about the body being finite.
3883    #[tokio::test]
3884    async fn the_did_document_read_is_capped() {
3885        let base = serve_oversized(
3886            "did-doc-cap.test",
3887            r##"{"service":[{"id":"#atproto_pds","type":"AtprotoPersonalDataServer","serviceEndpoint":"https://pds.example"}],"pad":"PAD"}"##,
3888        )
3889        .await;
3890        let err = resolve_did_to_pds(
3891            &ssrf_test_client(),
3892            &base,
3893            "did:plc:ohutz6x5acjmpuulp3x7wxxc",
3894        )
3895        .await
3896        .expect_err("an oversized DID document was buffered whole");
3897        assert!(
3898            format!("{err:#}").contains("cap"),
3899            "failed for the wrong reason: {err:#}"
3900        );
3901    }
3902
3903    /// `resolver_base` is a user-influenced PDS host, as this function's own
3904    /// guard comment says.
3905    #[tokio::test]
3906    async fn the_resolve_handle_read_is_capped() {
3907        let base = serve_oversized(
3908            "resolve-handle-cap.test",
3909            r#"{"did":"did:plc:ohutz6x5acjmpuulp3x7wxxc","pad":"PAD"}"#,
3910        )
3911        .await;
3912        let err = resolve_handle(&ssrf_test_client(), &base, "alice.example.com")
3913            .await
3914            .expect_err("an oversized resolveHandle body was buffered whole");
3915        assert!(
3916            format!("{err:#}").contains("cap"),
3917            "failed for the wrong reason: {err:#}"
3918        );
3919    }
3920
3921    /// **The sidecar's body is capped like every other body we read.**
3922    ///
3923    /// `/internal/repo` proxies whatever the account's PDS returned, so its
3924    /// size is remote-controlled by a host the reader chose and we did not.
3925    /// Every other response in this codebase goes through
3926    /// [`crate::net::read_capped`]; this one buffered the whole thing with
3927    /// `resp.json()`, so the 8 MB ceiling that bounds the direct PDS client
3928    /// simply did not exist on the sidecar backend — which is the default.
3929    #[tokio::test]
3930    async fn the_sidecar_client_caps_the_body_it_will_buffer() {
3931        // Well-formed, and past the cap. The guard has to fire on size, not
3932        // on the shape being wrong.
3933        let filler = "x".repeat(crate::net::MAX_BODY_BYTES);
3934        let body = format!(r#"{{"ok":true,"data":{{"records":[],"pad":"{filler}"}}}}"#);
3935        assert!(body.len() > crate::net::MAX_BODY_BYTES);
3936        let base = crate::net::tests::serve_body(body.into_bytes()).await;
3937        let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
3938        let err = client
3939            .list_records(
3940                "did:plc:ewvi7nxzyoun6zhxrhs64oiz",
3941                "app.feather.subscription",
3942                None,
3943                None,
3944            )
3945            .await
3946            .expect_err("an oversized sidecar body was buffered whole");
3947        assert!(
3948            format!("{err:#}").contains("cap"),
3949            "failed for the wrong reason: {err:#}"
3950        );
3951    }
3952
3953    /// The sidecar path needs the records guard too, not only the envelope
3954    /// one: `{"ok":true,"data":{}}` is what a proxy makes of an empty or
3955    /// unexpected upstream body.
3956    #[tokio::test]
3957    async fn the_sidecar_client_refuses_a_data_object_without_records() {
3958        // `data: {}` reaches `page_from_body`; no `data` at all is refused
3959        // before it, by the sidecar client itself. Both are the same refusal.
3960        for body in [&br#"{"ok":true,"data":{}}"#[..], &br#"{"ok":true}"#[..]] {
3961            let base = crate::net::tests::serve_body(body.to_vec()).await;
3962            let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
3963            let err = client
3964                .list_records(
3965                    "did:plc:ewvi7nxzyoun6zhxrhs64oiz",
3966                    "app.feather.subscription",
3967                    None,
3968                    None,
3969                )
3970                .await
3971                .expect_err("a data object without records was read as an empty repo");
3972            assert!(format!("{err:#}").contains("no records field"), "{err:#}");
3973            assert_eq!(
3974                crate::feed::publication_failure_kind(&err),
3975                crate::feed::FailureKind::Parse,
3976                "{err:#}"
3977            );
3978        }
3979    }
3980
3981    /// An exactly-full final page dropped nothing, so it must not warn that it
3982    /// did: `>=` reported truncation whenever the last page landed flush.
3983    #[test]
3984    fn an_exactly_full_page_is_not_a_truncation() {
3985        let page = |n: usize| -> Vec<RecordEntry> {
3986            (0..n)
3987                .map(|i| RecordEntry {
3988                    uri: format!("at://did:plc:x/c/{i}"),
3989                    cid: None,
3990                    value: Value::Null,
3991                })
3992                .collect()
3993        };
3994        let mut out = Vec::new();
3995        assert!(
3996            !extend_truncating(&mut out, page(3), 3),
3997            "a page that exactly fills the cap dropped nothing"
3998        );
3999        assert_eq!(out.len(), 3);
4000        assert!(
4001            extend_truncating(&mut out, page(1), 3),
4002            "one more IS a drop"
4003        );
4004        assert_eq!(out.len(), 3);
4005    }
4006
4007    /// **The reading walk USES the truncating accumulator.** The helper being
4008    /// correct is not the point — the previous round's bug was a guard that
4009    /// existed and was not called. Driven through a real server: one page of
4010    /// five records under a cap of three.
4011    #[tokio::test]
4012    async fn the_reading_walk_returns_a_truncated_archive_rather_than_an_error() {
4013        let records: Vec<Value> = (0..5)
4014            .map(|i| serde_json::json!({"uri": format!("at://did:plc:x/c/{i}"), "value": {}}))
4015            .collect();
4016        let body = serde_json::json!({ "records": records }).to_string();
4017        let base = crate::net::tests::serve_body(body.into_bytes()).await;
4018        let port: u16 = base
4019            .trim_end_matches('/')
4020            .rsplit(':')
4021            .next()
4022            .unwrap()
4023            .parse()
4024            .unwrap();
4025        crate::net::test_host_override(
4026            "truncating-pds.test",
4027            std::net::SocketAddr::from(([127, 0, 0, 1], port)),
4028        );
4029        let client = PdsClient::anonymous(
4030            ssrf_test_client(),
4031            format!("http://truncating-pds.test:{port}"),
4032            "did:plc:x",
4033        );
4034
4035        let walk = client
4036            .list_recent_matching("site.standard.document", 3, 100, |_| true)
4037            .await
4038            .expect("a big archive must be readable, not an error");
4039        assert_eq!(
4040            walk.records.len(),
4041            3,
4042            "the walk did not truncate to its cap"
4043        );
4044        assert!(
4045            !walk.complete,
4046            "a truncated walk must not report completeness"
4047        );
4048
4049        // The fail-closed walk still refuses the same overshoot.
4050        let err = client
4051            .list_all_records("community.lexicon.rss.subscription")
4052            .await;
4053        assert!(
4054            err.is_ok() || format!("{:#}", err.unwrap_err()).contains("cap"),
4055            "the sub_ref walk must keep its refusal"
4056        );
4057    }
4058
4059    /// **A write is not "succeeded" because the status was 200.** The sidecar's
4060    /// `delete_record` and `apply_writes` discard the body entirely, so a
4061    /// `200 {"error": …}` reported success: the UI showed a reader
4062    /// unsubscribed while the record was still in their repo, and a whole
4063    /// batch of writes vanished silently.
4064    #[tokio::test]
4065    async fn the_sidecar_client_refuses_a_200_error_envelope_on_writes() {
4066        let base = crate::net::tests::serve_body(
4067            br#"{"ok":true,"data":{"error":"InvalidRequest","message":"nope"}}"#.to_vec(),
4068        )
4069        .await;
4070        let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
4071        let did = "did:plc:ewvi7nxzyoun6zhxrhs64oiz";
4072        let err = client
4073            .delete_subscription(did, "rk1")
4074            .await
4075            .expect_err("a failed delete was reported as success");
4076        assert!(format!("{err:#}").contains("InvalidRequest"), "{err:#}");
4077
4078        let err = client
4079            .apply_writes(
4080                did,
4081                &[WriteOp::Delete {
4082                    collection: lexicon::nsid::SUBSCRIPTION.to_string(),
4083                    rkey: "rk1".to_string(),
4084                }],
4085            )
4086            .await
4087            .expect_err("a failed batch was reported as success");
4088        assert!(format!("{err:#}").contains("InvalidRequest"), "{err:#}");
4089    }
4090
4091    /// The sidecar proxies the PDS's body, so the same 2xx envelope arrives
4092    /// through `RepoOk.data` — a defaulted `Value` that deserialised into an
4093    /// empty page just as happily. Driven through the real client.
4094    #[tokio::test]
4095    async fn the_sidecar_client_refuses_a_200_error_envelope() {
4096        let base = crate::net::tests::serve_body(
4097            br#"{"ok":true,"data":{"error":"InvalidRequest","message":"nope"}}"#.to_vec(),
4098        )
4099        .await;
4100        let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
4101        let err = client
4102            .list_records(
4103                "did:plc:ewvi7nxzyoun6zhxrhs64oiz",
4104                "app.feather.subscription",
4105                None,
4106                None,
4107            )
4108            .await
4109            .expect_err("an error envelope was read as an empty page");
4110        assert!(
4111            format!("{err:#}").contains("InvalidRequest"),
4112            "failed for the wrong reason: {err:#}"
4113        );
4114    }
4115
4116    /// **Every listRecords caller refuses a 2xx error envelope, not just the
4117    /// anonymous one.** `oauth::xrpc::Repo` reads `records` off the JSON with
4118    /// `unwrap_or(Array([]))` and the sidecar's `RepoOk.data` is a defaulted
4119    /// `Value`, so a PDS answering 200 with an envelope reached
4120    /// `resolve_subscriptions` as `Ok(empty)` — which is not the fail-closed
4121    /// branch, so `sync_sub_refs` DELETEd the DID's whole `sub_ref` projection:
4122    /// one bad response revokes a reader's access to every feed they have.
4123    #[test]
4124    fn an_error_envelope_is_refused_whatever_shape_it_arrives_in() {
4125        let envelope = serde_json::json!({"error": "InvalidRequest", "message": "bad cursor"});
4126        let err = reject_error_envelope(&envelope).expect_err("an envelope passed as data");
4127        assert!(format!("{err:#}").contains("InvalidRequest"), "{err:#}");
4128        assert_eq!(
4129            crate::feed::publication_failure_kind(&err),
4130            crate::feed::FailureKind::Status,
4131            "{err:#}"
4132        );
4133        // A real page, and an empty real page, are both data.
4134        reject_error_envelope(&serde_json::json!({"records": []})).expect("an empty page is data");
4135        reject_error_envelope(&serde_json::json!({"records": [], "cursor": "c"})).unwrap();
4136    }
4137
4138    /// **A 200 carrying an error envelope is not an empty page.** `records` is
4139    /// `#[serde(default)]`, so `{"error": "...", "message": "..."}` on a 200
4140    /// deserialised as zero records — and a walk over a stranger's documents
4141    /// then returned a healthy, empty feed instead of an error. Some PDS
4142    /// implementations do answer 200 for application-level failures.
4143    #[test]
4144    fn a_200_with_an_error_envelope_is_not_an_empty_page() {
4145        let err = parse_list_records(br#"{"error":"InvalidRequest","message":"bad cursor"}"#)
4146            .expect_err("an error envelope parsed as a page");
4147        assert!(format!("{err:#}").contains("InvalidRequest"), "{err:#}");
4148        // #227: an error said on a 200 is the PDS answering no, as a 400 is.
4149        assert_eq!(
4150            crate::feed::publication_failure_kind(&err),
4151            crate::feed::FailureKind::Status,
4152            "{err:#}"
4153        );
4154        let page = parse_list_records(br#"{"records":[]}"#).expect("an empty page is a page");
4155        assert!(page.records.is_empty() && page.cursor.is_none());
4156    }
4157
4158    /// **Both invariants, now read out of the bytes rather than out of a
4159    /// `Value`.** Parsing once is the point of the change; parsing once while
4160    /// quietly dropping a guard would be a much worse trade, and these are the
4161    /// shapes those guards exist for.
4162    #[test]
4163    fn parsing_a_page_from_bytes_keeps_both_invariants() {
4164        // Each shape names the reason it must fail for. Accepting either
4165        // message would let the envelope guard be deleted without a test
4166        // noticing, because an error envelope also has no `records` field — so
4167        // it keeps failing, for a reason that stops applying the day a PDS
4168        // returns an envelope alongside a records array.
4169        for (label, body, because) in [
4170            (
4171                "an error envelope on a 2xx",
4172                &br#"{"error":"InvalidRequest","message":"bad cursor"}"#[..],
4173                "error envelope",
4174            ),
4175            (
4176                "an envelope that also carries records",
4177                &br#"{"error":"InvalidRequest","records":[]}"#[..],
4178                "error envelope",
4179            ),
4180            (
4181                "a body with no records field",
4182                &br#"{"cursor":"c"}"#[..],
4183                "no records",
4184            ),
4185            ("a proxy's empty object", &br#"{}"#[..], "no records"),
4186            ("an empty body", &b""[..], "no records"),
4187        ] {
4188            let err = parse_list_records(body)
4189                .map(|p| panic!("{label} was read as a page of {} records", p.records.len()))
4190                .unwrap_err();
4191            let msg = format!("{err:#}");
4192            assert!(
4193                msg.contains(because),
4194                "{label} should have failed on {because:?}, got: {msg}"
4195            );
4196        }
4197        let page = parse_list_records(br#"{"records":[],"cursor":"c"}"#)
4198            .expect("a genuinely empty page is still a page");
4199        assert!(page.records.is_empty());
4200        assert_eq!(page.cursor.as_deref(), Some("c"));
4201    }
4202
4203    #[test]
4204    fn list_records_envelope_deserializes() {
4205        let resp: ListRecordsResponse =
4206            serde_json::from_value(subscription_list_json()).expect("envelope");
4207        assert_eq!(resp.records.len(), 2);
4208        assert_eq!(resp.cursor.as_deref(), Some("3ksub0002"));
4209        assert_eq!(resp.records[0].cid.as_deref(), Some("bafyreisubone"));
4210    }
4211
4212    #[test]
4213    fn record_entry_rkey_is_last_uri_segment() {
4214        let resp: ListRecordsResponse =
4215            serde_json::from_value(subscription_list_json()).expect("envelope");
4216        assert_eq!(resp.records[0].rkey(), Some("3ksub0001"));
4217        assert_eq!(resp.records[1].rkey(), Some("3ksub0002"));
4218    }
4219
4220    #[test]
4221    fn record_value_parses_into_lexicon_subscription() {
4222        let resp: ListRecordsResponse =
4223            serde_json::from_value(subscription_list_json()).expect("envelope");
4224
4225        let full: Subscription = resp.records[0].parse().expect("parse full sub");
4226        assert_eq!(full.r#type, lexicon::nsid::SUBSCRIPTION);
4227        assert_eq!(full.url, "https://example.com/feed.xml");
4228        assert_eq!(full.title.as_deref(), Some("Example Blog"));
4229        assert_eq!(full.site_url.as_deref(), Some("https://example.com/"));
4230        assert_eq!(full.fetch_hint, Some(lexicon::FetchHint::Hourly));
4231
4232        let minimal: Subscription = resp.records[1].parse().expect("parse minimal sub");
4233        assert_eq!(minimal.url, "https://blog.example.org/atom.xml");
4234        assert!(minimal.title.is_none());
4235    }
4236
4237    fn ssrf_test_client() -> Client {
4238        Client::builder()
4239            .user_agent(crate::USER_AGENT)
4240            .build()
4241            .unwrap()
4242    }
4243
4244    /// A hostile `did:web` whose host is the cloud-metadata address must be
4245    /// REFUSED before any request leaves the box — the DID-document fetch now
4246    /// routes through the SSRF guard (`guarded_get_no_privacy`), which rejects
4247    /// link-local / metadata targets.
4248    #[tokio::test]
4249    async fn resolve_did_web_blocks_metadata_host() {
4250        let client = ssrf_test_client();
4251        let err = resolve_did_to_pds(&client, "https://plc.directory", "did:web:169.254.169.254")
4252            .await
4253            .unwrap_err()
4254            .to_string();
4255        assert!(
4256            err.contains("forbidden") || err.contains("internal"),
4257            "expected an SSRF refusal, got: {err}"
4258        );
4259    }
4260
4261    /// A `did:web` pointing at loopback is likewise blocked (internal service
4262    /// reflection).
4263    #[tokio::test]
4264    async fn resolve_did_web_blocks_loopback_host() {
4265        let client = ssrf_test_client();
4266        let err = resolve_did_to_pds(&client, "https://plc.directory", "did:web:127.0.0.1")
4267            .await
4268            .unwrap_err()
4269            .to_string();
4270        assert!(
4271            err.contains("forbidden") || err.contains("internal"),
4272            "expected an SSRF refusal, got: {err}"
4273        );
4274    }
4275
4276    /// `resolve_handle` against a metadata/loopback resolver base is also guarded
4277    /// (the base can come from a prior hostile DID-doc resolution).
4278    #[tokio::test]
4279    async fn resolve_handle_blocks_metadata_resolver_base() {
4280        let client = ssrf_test_client();
4281        let err = resolve_handle(&client, "http://169.254.169.254", "alice.example.com")
4282            .await
4283            .unwrap_err()
4284            .to_string();
4285        assert!(
4286            err.contains("forbidden") || err.contains("internal"),
4287            "expected an SSRF refusal, got: {err}"
4288        );
4289    }
4290
4291    /// A resolved `serviceEndpoint` that targets an internal host is rejected at
4292    /// resolve time via [`crate::net::assert_public_target`], so it can never be
4293    /// handed to a raw XRPC client.
4294    #[tokio::test]
4295    async fn service_endpoint_internal_target_rejected() {
4296        assert!(crate::net::assert_public_target("http://169.254.169.254/")
4297            .await
4298            .is_err());
4299        assert!(crate::net::assert_public_target("http://127.0.0.1:3000/")
4300            .await
4301            .is_err());
4302        // A public endpoint literal passes.
4303        assert!(crate::net::assert_public_target("https://1.1.1.1/")
4304            .await
4305            .is_ok());
4306    }
4307
4308    /// A `PdsClient` pointed at an internal `pds_base`, as an attacker-controlled
4309    /// DID document could arrange between the `assert_public_target` at resolve
4310    /// time and the request.
4311    fn internal_target_client(pds_base: &str) -> PdsClient {
4312        PdsClient::new(
4313            ssrf_test_client(),
4314            pds_base,
4315            "did:plc:victim",
4316            Auth::Session(SessionAuth {
4317                did: "did:plc:victim".to_string(),
4318                handle: None,
4319                access_jwt: "session-bearer-must-not-leak".to_string(),
4320                refresh_jwt: None,
4321            }),
4322        )
4323    }
4324
4325    /// **Regression (v0.2.8):** every `com.atproto.repo.*` WRITE must go through
4326    /// the SSRF guard, not the shared client. Before the fix only `list_records`
4327    /// was guarded, so `createRecord` / `putRecord` / `deleteRecord` /
4328    /// `applyWrites` would happily deliver the session bearer to
4329    /// `169.254.169.254` or loopback on a rebound host.
4330    #[tokio::test]
4331    async fn every_repo_write_is_refused_against_an_internal_pds() {
4332        for base in [
4333            "http://169.254.169.254",
4334            "http://127.0.0.1:9",
4335            "http://[::1]",
4336        ] {
4337            let client = internal_target_client(base);
4338            let sub = Subscription::new("https://example.com/feed.xml", "2026-08-13T00:00:00Z");
4339
4340            let mut errors = vec![
4341                client
4342                    .create_record(lexicon::nsid::SUBSCRIPTION, &sub)
4343                    .await
4344                    .unwrap_err()
4345                    .to_string(),
4346                client
4347                    .put_record(lexicon::nsid::SUBSCRIPTION, "rkey", &sub, None)
4348                    .await
4349                    .unwrap_err()
4350                    .to_string(),
4351                client
4352                    .delete_record(lexicon::nsid::SUBSCRIPTION, "rkey")
4353                    .await
4354                    .unwrap_err()
4355                    .to_string(),
4356            ];
4357            errors.push(
4358                client
4359                    .apply_writes(&[WriteOp::Delete {
4360                        collection: lexicon::nsid::SUBSCRIPTION.to_string(),
4361                        rkey: "rkey".to_string(),
4362                    }])
4363                    .await
4364                    .unwrap_err()
4365                    .to_string(),
4366            );
4367
4368            for err in errors {
4369                assert!(
4370                    err.contains("forbidden") || err.contains("internal"),
4371                    "{base}: expected an SSRF refusal, got: {err}"
4372                );
4373            }
4374        }
4375    }
4376
4377    /// **Regression (v0.2.8):** the app password travels in the request BODY,
4378    /// where reqwest's cross-origin header sanitisation cannot protect it — so
4379    /// `createSession` is guarded too, and a rebound/internal `pds_base` never
4380    /// receives it.
4381    #[tokio::test]
4382    async fn app_password_login_is_refused_against_an_internal_pds() {
4383        let client = ssrf_test_client();
4384        for base in ["http://169.254.169.254", "http://127.0.0.1:9"] {
4385            let err = login_with_app_password(&client, base, "alice.example.com", "hunter2-app-pw")
4386                .await
4387                .unwrap_err()
4388                .to_string();
4389            assert!(
4390                err.contains("forbidden") || err.contains("internal"),
4391                "{base}: expected an SSRF refusal, got: {err}"
4392            );
4393        }
4394    }
4395
4396    /// An anonymous client is read-only: the write paths fail closed on
4397    /// [`Auth::bearer`] before any socket work, so `Auth::Anonymous` can never
4398    /// become a credential-less write primitive against a stranger's PDS.
4399    #[tokio::test]
4400    async fn anonymous_client_cannot_write() {
4401        let client = PdsClient::anonymous(
4402            ssrf_test_client(),
4403            "https://pds.example.com",
4404            "did:plc:stranger",
4405        );
4406        let err = client
4407            .delete_record(lexicon::nsid::SUBSCRIPTION, "rkey")
4408            .await
4409            .unwrap_err()
4410            .to_string();
4411        assert!(
4412            err.contains("no credentials") || err.contains("anonymous") || err.contains("bearer"),
4413            "expected a fail-closed auth error, got: {err}"
4414        );
4415    }
4416
4417    #[test]
4418    fn write_result_deserializes() {
4419        let wr: WriteResult = serde_json::from_value(json!({
4420            "uri": "at://did:plc:abc123/community.lexicon.rss.subscription/3ksubnew",
4421            "cid": "bafyreinew"
4422        }))
4423        .expect("write result");
4424        assert!(wr.uri.ends_with("3ksubnew"));
4425        assert_eq!(wr.cid.as_deref(), Some("bafyreinew"));
4426    }
4427
4428    #[test]
4429    fn did_document_finds_pds_endpoint() {
4430        let doc: DidDocument = serde_json::from_value(json!({
4431            "id": "did:plc:abc123",
4432            "service": [
4433                {
4434                    "id": "#atproto_pds",
4435                    "type": "AtprotoPersonalDataServer",
4436                    "serviceEndpoint": "https://pds.example.com/"
4437                }
4438            ]
4439        }))
4440        .expect("did doc");
4441        assert_eq!(
4442            doc.pds_endpoint().as_deref(),
4443            Some("https://pds.example.com")
4444        );
4445    }
4446
4447    #[test]
4448    fn did_document_without_pds_yields_none() {
4449        let doc: DidDocument = serde_json::from_value(json!({
4450            "id": "did:plc:abc123",
4451            "service": []
4452        }))
4453        .expect("did doc");
4454        assert!(doc.pds_endpoint().is_none());
4455    }
4456
4457    #[test]
4458    fn session_auth_deserializes_create_session_shape() {
4459        let session: SessionAuth = serde_json::from_value(json!({
4460            "did": "did:plc:abc123",
4461            "handle": "alice.example.com",
4462            "accessJwt": "eyJh...access",
4463            "refreshJwt": "eyJh...refresh"
4464        }))
4465        .expect("session");
4466        assert_eq!(session.did, "did:plc:abc123");
4467        assert_eq!(session.handle.as_deref(), Some("alice.example.com"));
4468        let auth = Auth::Session(session);
4469        assert_eq!(auth.bearer().expect("bearer"), "eyJh...access");
4470    }
4471
4472    #[test]
4473    fn oauth_variant_carries_no_direct_bearer() {
4474        let auth = Auth::Oauth(OauthPlaceholder::default());
4475        assert!(
4476            auth.bearer().is_err(),
4477            "Auth::Oauth carries no direct bearer — the sidecar owns the OAuth path"
4478        );
4479    }
4480
4481    #[test]
4482    fn anonymous_variant_carries_no_bearer() {
4483        let err = Auth::Anonymous.bearer().unwrap_err().to_string();
4484        assert!(
4485            err.contains("anonymous"),
4486            "the anonymous refusal must name itself, got: {err}"
4487        );
4488    }
4489
4490    #[test]
4491    fn anonymous_client_targets_the_requested_repo() {
4492        let client = PdsClient::anonymous(
4493            ssrf_test_client(),
4494            "https://pds.example.com/",
4495            "did:plc:abc123",
4496        );
4497        // The trailing slash is trimmed so `xrpc_url` joins cleanly.
4498        assert_eq!(client.pds_base(), "https://pds.example.com");
4499        assert_eq!(client.did(), "did:plc:abc123");
4500        // …and it holds no credential.
4501        assert!(client.auth.bearer().is_err());
4502    }
4503
4504    /// The regression test for the defect this milestone fixes: `list_records`
4505    /// used to send on the shared client, bypassing the SSRF guard entirely. It
4506    /// now routes through `net::guarded_get_no_privacy`, so an internal
4507    /// `pds_base` is refused before a packet leaves the box. Hermetic — the hosts
4508    /// are IP literals, rejected without any DNS lookup or connect.
4509    #[tokio::test]
4510    async fn list_records_blocks_internal_pds_base() {
4511        for base in ["http://169.254.169.254", "http://127.0.0.1:1"] {
4512            let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
4513            let err = client
4514                .list_records(lexicon::nsid::SUBSCRIPTION, Some(1), None)
4515                .await
4516                .unwrap_err()
4517                .to_string();
4518            assert!(
4519                err.contains("forbidden") || err.contains("internal"),
4520                "expected an SSRF refusal for {base}, got: {err}"
4521            );
4522        }
4523    }
4524
4525    /// The guard is not anonymous-only: an *authenticated* client reading a
4526    /// hostile PDS base is blocked identically. (That path was only ever safe by
4527    /// accident of usage.)
4528    #[tokio::test]
4529    async fn list_records_guard_applies_to_authed_clients_too() {
4530        let auth = Auth::Session(SessionAuth {
4531            did: "did:plc:x".to_string(),
4532            handle: None,
4533            access_jwt: "x".to_string(),
4534            refresh_jwt: None,
4535        });
4536        let client = PdsClient::new(
4537            ssrf_test_client(),
4538            "http://169.254.169.254",
4539            "did:plc:x",
4540            auth,
4541        );
4542        let err = client
4543            .list_records(lexicon::nsid::SUBSCRIPTION, Some(1), None)
4544            .await
4545            .unwrap_err()
4546            .to_string();
4547        assert!(
4548            err.contains("forbidden") || err.contains("internal"),
4549            "expected an SSRF refusal, got: {err}"
4550        );
4551    }
4552
4553    #[test]
4554    fn apply_writes_ops_render_tagged_union() {
4555        let create = WriteOp::Create {
4556            collection: lexicon::nsid::SUBSCRIPTION.to_string(),
4557            rkey: None,
4558            value: json!({"url": "https://example.com/feed.xml"}),
4559        };
4560        let update = WriteOp::Update {
4561            collection: lexicon::nsid::READ_STATE.to_string(),
4562            rkey: "feedhash01".to_string(),
4563            value: json!({"feedUrl": "https://example.com/feed.xml"}),
4564        };
4565        let delete = WriteOp::Delete {
4566            collection: lexicon::nsid::SAVED.to_string(),
4567            rkey: "3ksaved01".to_string(),
4568        };
4569
4570        assert_eq!(
4571            create.to_json()["$type"],
4572            json!("com.atproto.repo.applyWrites#create")
4573        );
4574        // A create with no explicit rkey omits the field (server assigns a tid).
4575        assert!(create.to_json().get("rkey").is_none());
4576
4577        assert_eq!(
4578            update.to_json()["$type"],
4579            json!("com.atproto.repo.applyWrites#update")
4580        );
4581        assert_eq!(update.to_json()["rkey"], json!("feedhash01"));
4582
4583        assert_eq!(
4584            delete.to_json()["$type"],
4585            json!("com.atproto.repo.applyWrites#delete")
4586        );
4587        assert_eq!(delete.to_json()["rkey"], json!("3ksaved01"));
4588    }
4589
4590    #[test]
4591    fn read_state_flush_creates_first_then_updates() {
4592        // A cursor whose PDS record does NOT yet exist (pds_created = false) must
4593        // become a CREATE op at its stable rkey — NOT a bare update, which would
4594        // error on the missing record and (applyWrites being atomic per-repo) drop
4595        // the whole batch on a feed's first flush.
4596        let fresh = (
4597            "rs-fresh".to_string(),
4598            ReadState::new("https://a.example/feed.xml", None, "2026-07-12T00:00:00Z"),
4599            false,
4600        );
4601        // An already-created cursor updates in place.
4602        let existing = (
4603            "rs-existing".to_string(),
4604            ReadState::new(
4605                "https://b.example/feed.xml",
4606                Some("2026-07-11T00:00:00Z".to_string()),
4607                "2026-07-12T00:00:00Z",
4608            ),
4609            true,
4610        );
4611
4612        let ops = read_state_write_ops(&[fresh, existing]).expect("build ops");
4613        assert_eq!(ops.len(), 2);
4614
4615        // First op: a create carrying the stable rkey (put/create, not update).
4616        let create = ops[0].to_json();
4617        assert_eq!(
4618            create["$type"],
4619            json!("com.atproto.repo.applyWrites#create"),
4620            "first flush of a new feed must CREATE its readState record"
4621        );
4622        assert_eq!(create["rkey"], json!("rs-fresh"));
4623        // The created record omits readThrough (F1): backlog not implicitly read.
4624        assert!(create["value"].get("readThrough").is_none());
4625
4626        // Second op: an update for the already-created record.
4627        let update = ops[1].to_json();
4628        assert_eq!(
4629            update["$type"],
4630            json!("com.atproto.repo.applyWrites#update")
4631        );
4632        assert_eq!(update["rkey"], json!("rs-existing"));
4633
4634        // Both ride the SAME batch — batching is preserved.
4635        assert_eq!(ops.len(), 2);
4636    }
4637
4638    #[test]
4639    fn urlencode_escapes_did_colons_and_keeps_unreserved() {
4640        assert_eq!(urlencode("did:plc:abc123"), "did%3Aplc%3Aabc123");
4641        assert_eq!(
4642            urlencode("community.lexicon.rss.subscription"),
4643            "community.lexicon.rss.subscription"
4644        );
4645        assert_eq!(urlencode("a b&c"), "a%20b%26c");
4646    }
4647
4648    #[test]
4649    fn xrpc_record_not_found_is_detected() {
4650        let err = AtProtoError::Xrpc {
4651            status: StatusCode::BAD_REQUEST,
4652            error: "RecordNotFound".to_string(),
4653            message: Some("Could not locate record".to_string()),
4654        };
4655        assert!(err.is_record_not_found());
4656    }
4657
4658    // -- reader-facing CRUD: rkey extraction --------------------------------
4659
4660    #[test]
4661    fn write_result_extracts_rkey_from_uri() {
4662        let wr: WriteResult = serde_json::from_value(json!({
4663            "uri": "at://did:plc:abc123/community.lexicon.rss.subscription/3ksubnew",
4664            "cid": "bafyreinew"
4665        }))
4666        .expect("write result");
4667        assert_eq!(wr.rkey(), Some("3ksubnew"));
4668        assert_eq!(wr.into_rkey(), "3ksubnew");
4669    }
4670
4671    // -- reader-facing CRUD: deterministic sort orders ----------------------
4672    //
4673    // The `list_*_sorted` wrappers only add an ordering on top of the network
4674    // `list_*` read, so we exercise the *comparator* here on representative
4675    // data (parsed from a listRecords-shaped envelope) with no network.
4676
4677    // -- reader-facing CRUD: bulk applyWrites shape (OPML import) ------------
4678
4679    /// **Bulk subscribe, through the real client, asserted on the bytes it
4680    /// sent.** The test this replaces built the `WriteOp::Create` ops itself
4681    /// ("mirror what `add_subscriptions_bulk` builds") and asserted on its own
4682    /// construction; the function was never called, and writing every feed
4683    /// into the wrong collection with server-assigned rkeys left the suite
4684    /// green. Three atproto sort tests that re-implemented the comparator
4685    /// inline are deleted alongside — `lexicon::sort_tests` fails their
4686    /// mutation, and they added nothing but a misleading name.
4687    #[tokio::test]
4688    async fn bulk_subscribe_writes_client_assigned_ordered_rkeys_to_the_right_collection() {
4689        let (base, log) =
4690            crate::net::tests::serve_json_capturing(br#"{"ok":true,"data":{}}"#.to_vec()).await;
4691        let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
4692        let subs: Vec<crate::vetted::VettedSubscription> = (0..3)
4693            .map(|i| {
4694                crate::vetted::VettedSubscription::new(&lexicon::Subscription::new(
4695                    format!("https://f{i}.example/feed.xml"),
4696                    "2026-07-12T00:00:00.000Z",
4697                ))
4698            })
4699            .collect();
4700
4701        let rkeys = client
4702            .add_subscriptions_bulk("did:plc:ewvi7nxzyoun6zhxrhs64oiz", &subs)
4703            .await
4704            .expect("bulk write failed");
4705
4706        let sent = log.lock().unwrap().clone();
4707        assert_eq!(
4708            sent.len(),
4709            1,
4710            "expected one applyWrites request, got {sent:?}"
4711        );
4712        let body: Value = serde_json::from_str(sent[0].split("\r\n\r\n").nth(1).unwrap())
4713            .expect("request body is JSON");
4714        let writes = body["writes"].as_array().expect("writes array");
4715        assert_eq!(writes.len(), 3);
4716        for (i, w) in writes.iter().enumerate() {
4717            assert_eq!(
4718                w["collection"],
4719                lexicon::nsid::SUBSCRIPTION,
4720                "write {i} went to the wrong collection"
4721            );
4722            assert_eq!(
4723                w["rkey"].as_str(),
4724                Some(rkeys[i].as_str()),
4725                "write {i} does not carry the rkey the client returned"
4726            );
4727        }
4728        let mut sorted = rkeys.clone();
4729        sorted.sort();
4730        assert_eq!(rkeys, sorted, "client-assigned rkeys must ascend");
4731        assert_eq!(
4732            rkeys.iter().collect::<std::collections::HashSet<_>>().len(),
4733            3,
4734            "rkeys must be distinct"
4735        );
4736    }
4737
4738    /// **The walk refuses a repeated cursor (#203).** A PDS that echoes the
4739    /// same cursor forever would otherwise be walked for 200 pages, so the
4740    /// walk must not follow it — but it must not return what it has either.
4741    /// This walk feeds `replace_sub_refs`, and the two pages it read are a
4742    /// list the server said was unfinished: returned as `Ok`, every
4743    /// subscription past them is deleted from the reader's projection.
4744    #[tokio::test]
4745    async fn list_all_records_refuses_a_repeated_cursor() {
4746        let body = serde_json::json!({
4747            "records": [{"uri": "at://did:plc:x/c/1", "value": {}}],
4748            "cursor": "same-every-time"
4749        })
4750        .to_string();
4751        let base = crate::net::tests::serve_body(body.into_bytes()).await;
4752        let port: u16 = base
4753            .trim_end_matches('/')
4754            .rsplit(':')
4755            .next()
4756            .unwrap()
4757            .parse()
4758            .unwrap();
4759        crate::net::test_host_override(
4760            "repeated-cursor.test",
4761            std::net::SocketAddr::from(([127, 0, 0, 1], port)),
4762        );
4763        let client = PdsClient::anonymous(
4764            ssrf_test_client(),
4765            format!("http://repeated-cursor.test:{port}"),
4766            "did:plc:x",
4767        );
4768        // Page 1: cursor None → "same". Page 2: "same" again, with a record →
4769        // refused. Two requests, not two hundred, and no short list.
4770        let err = client
4771            .list_all_records("c")
4772            .await
4773            .expect_err("a repeated cursor ended the walk with a short list");
4774        assert!(
4775            format!("{err:#}").contains("repeated cursor"),
4776            "refused for the wrong reason: {err:#}"
4777        );
4778    }
4779
4780    /// Twin of `list_all_records_refuses_a_repeated_cursor` for the sidecar
4781    /// walk, the default backend.
4782    #[tokio::test]
4783    async fn the_sidecar_walk_refuses_a_repeated_cursor() {
4784        let body = serde_json::json!({
4785            "ok": true,
4786            "data": {
4787                "records": [{ "uri": "at://did:plc:x/c/3labONE", "value": {} }],
4788                "cursor": "same-every-time",
4789            }
4790        })
4791        .to_string()
4792        .into_bytes();
4793        let base = crate::net::tests::serve_body(body).await;
4794        let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
4795        let err = client
4796            .list_all_records("did:plc:ewvi7nxzyoun6zhxrhs64oiz", "c")
4797            .await
4798            .expect_err("a repeated cursor ended the walk with a short list");
4799        assert!(
4800            format!("{err:#}").contains("repeated cursor"),
4801            "refused for the wrong reason: {err:#}"
4802        );
4803    }
4804
4805    /// **A cursor on an EMPTY page still ends the walk normally.** This
4806    /// project's own PDS returns a cursor alongside its last page, so the
4807    /// repeated-cursor refusal must not reach a page with no records.
4808    #[tokio::test]
4809    async fn the_sidecar_walk_ends_on_an_empty_page_that_carries_a_cursor() {
4810        let bodies = vec![
4811            serde_json::json!({ "ok": true, "data": {
4812                "records": [{ "uri": "at://did:plc:x/c/3labONE", "value": {} }],
4813                "cursor": "p1",
4814            }})
4815            .to_string()
4816            .into_bytes(),
4817            serde_json::json!({ "ok": true, "data": { "records": [], "cursor": "p1" }})
4818                .to_string()
4819                .into_bytes(),
4820        ];
4821        let base = crate::net::tests::serve_bodies_in_sequence(bodies).await;
4822        let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
4823        let records = client
4824            .list_all_records("did:plc:ewvi7nxzyoun6zhxrhs64oiz", "c")
4825            .await
4826            .expect("an empty page with a cursor is how a real PDS ends a list");
4827        assert_eq!(records.len(), 1);
4828    }
4829
4830    // -- bounding the parse before it allocates -----------------------------
4831
4832    /// **Strings cannot invent structure.** The subtle half of the bound: an
4833    /// article containing a million commas is one node, and counting naively
4834    /// would refuse it.
4835    #[test]
4836    fn structure_inside_a_string_is_not_structure() {
4837        let prose = format!(
4838            r#"{{"records":[{{"uri":"at://d/c/r","value":{{"t":"{}"}}}}]}}"#,
4839            "a,b,[c],{d}:e,".repeat(50_000)
4840        );
4841        let bound = count_structural_chars(prose.as_bytes());
4842        assert!(
4843            bound < 100,
4844            "a page of prose full of punctuation was counted as {bound} nodes"
4845        );
4846        assert!(
4847            parse_list_records(prose.as_bytes()).is_ok(),
4848            "a legitimate page of prose was refused"
4849        );
4850    }
4851
4852    #[test]
4853    fn a_node_explosion_is_refused_before_it_is_parsed() {
4854        // ~8 MB of the cheapest node there is, which is the measured attack.
4855        let mut body = String::from(r#"{"records":[{"uri":"at://d/c/r","value":["#);
4856        for _ in 0..1_200_000 {
4857            body.push_str("{},");
4858        }
4859        body.push_str(r#"{}]}]}"#);
4860        assert!(
4861            body.len() > 3_000_000,
4862            "the probe body is {} bytes",
4863            body.len()
4864        );
4865
4866        let bound = count_structural_chars(body.as_bytes());
4867        assert!(
4868            bound > MAX_LIST_STRUCTURAL_CHARS,
4869            "the attack shape was counted as only {bound} nodes"
4870        );
4871        let err = parse_list_records(body.as_bytes())
4872            .expect_err("a node explosion was parsed rather than refused");
4873        assert!(
4874            format!("{err:#}").contains("structural characters"),
4875            "failed for the wrong reason: {err:#}"
4876        );
4877    }
4878
4879    /// **The densest page the lexicons permit must fit, with room.**
4880    ///
4881    /// This is the floor under [`MAX_LIST_STRUCTURAL_CHARS`], and it is the reason the cap
4882    /// is 640 000 rather than the ~150 000 that would otherwise hold the memory
4883    /// claim comfortably. A `readState` record carries up to
4884    /// [`crate::lexicon::ReadState::MAX_IDS`] read ids, and a page carries 100 of
4885    /// them — far denser in nodes than a page of articles, which is mostly
4886    /// prose. Tighten the cap below this and a reader with a lot of history
4887    /// stops being able to sync at all.
4888    #[test]
4889    fn a_full_read_state_page_fits_under_the_cap() {
4890        let ids: Vec<String> = (0..crate::lexicon::ReadState::MAX_IDS)
4891            .map(|i| format!("https://example.com/blog/post-{i}"))
4892            .collect();
4893        let records: Vec<serde_json::Value> = (0..100)
4894            .map(|i| {
4895                serde_json::json!({
4896                    "uri": format!("at://did:plc:ohutz6x5acjmpuulp3x7wxxc/community.lexicon.rss.readState/3lab{i}"),
4897                    "cid": "bafyreiabc123def456ghi789jkl012mno345pqr678stu901",
4898                    "value": {
4899                        "$type": "community.lexicon.rss.readState",
4900                        "feedUrl": "https://example.com/feed.xml",
4901                        "readThrough": "2026-07-11T09:30:00Z",
4902                        // **BOTH arrays, because the lexicon permits both.**
4903                        // Filling only `readIds` counted 203 503 nodes, so a cap
4904                        // as low as 300 000 passed every test in the suite while
4905                        // refusing the very page this test exists to protect.
4906                        "readIds": ids,
4907                        "unreadIds": ids,
4908                    }
4909                })
4910            })
4911            .collect();
4912        let body = serde_json::json!({ "records": records }).to_string();
4913        let bound = count_structural_chars(body.as_bytes());
4914        assert!(
4915            bound < MAX_LIST_STRUCTURAL_CHARS,
4916            "the densest legitimate page counts {bound} nodes against a cap of {MAX_LIST_STRUCTURAL_CHARS}"
4917        );
4918        // And it is dense enough to be the floor the cap was chosen for: a page
4919        // counting only a fifth of the cap would pass the assertion above while
4920        // leaving the cap free to drop far below real traffic.
4921        assert!(
4922            bound > MAX_LIST_STRUCTURAL_CHARS / 2,
4923            "this page counts only {bound} nodes, so it is no longer the floor \
4924             `MAX_LIST_STRUCTURAL_CHARS` was measured against and a much tighter cap would \
4925             pass it"
4926        );
4927        assert!(
4928            parse_list_records(body.as_bytes()).is_ok(),
4929            "a full read-state page was refused"
4930        );
4931    }
4932
4933    /// The write path takes the same guard as the listing path.
4934    #[tokio::test]
4935    async fn the_write_path_refuses_a_node_explosion() {
4936        let mut data = String::from(r#"{"ok":true,"data":{"records":["#);
4937        for _ in 0..700_000 {
4938            data.push_str("{},");
4939        }
4940        data.push_str(r#"{}]}}"#);
4941        let base = crate::net::tests::serve_body(data.into_bytes()).await;
4942        let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
4943        let err = client
4944            .delete_record("did:plc:ewvi7nxzyoun6zhxrhs64oiz", "c", "r")
4945            .await
4946            .expect_err("a node explosion reached the parser on the write path");
4947        assert!(
4948            format!("{err:#}").contains("structural characters"),
4949            "failed for the wrong reason: {err:#}"
4950        );
4951    }
4952
4953    #[test]
4954    fn an_ordinary_page_is_nowhere_near_the_structure_bound() {
4955        // **A page, not a record.** This served `paged_bodies(1, 17_000, false)` —
4956        // ONE page holding ONE record — which counts about eleven characters, so
4957        // the old `< 1_000` assertion held by three orders of magnitude and would
4958        // have passed with the cap at 1 000. It also made the doc comment's "a
4959        // page of 100 documents measures 40 000" untested, and that figure was
4960        // wrong by 10x.
4961        let records: Vec<serde_json::Value> = (0..100)
4962            .map(|i| {
4963                serde_json::json!({
4964                    "uri": format!("at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.document/3lab{i}"),
4965                    "cid": "bafyreiabc123def456ghi789jkl012mno345pqr678stu901",
4966                    "value": {
4967                        "$type": "site.standard.document",
4968                        "title": "A reasonably typical post title",
4969                        "path": format!("/posts/{i}"),
4970                        "site": "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
4971                        "publishedAt": "2026-07-11T09:30:00Z",
4972                        "description": "x".repeat(120),
4973                        "textContent": "y".repeat(15_000),
4974                    }
4975                })
4976            })
4977            .collect();
4978        let body = serde_json::json!({ "records": records }).to_string();
4979        // A real page of documents is mostly prose: 1.5 MB on the wire for 4 003
4980        // counted characters.
4981        assert!(
4982            body.len() > 1_000_000,
4983            "the probe page is only {} bytes, so it is not a full page",
4984            body.len(),
4985        );
4986        let counted = count_structural_chars(body.as_bytes());
4987        assert!(
4988            (3_500..4_500).contains(&counted),
4989            "a page of 100 documents counted {counted}, not the ~4 003 the cap's \
4990             doc comment claims — the ordinary-traffic end of the bracket moved",
4991        );
4992        assert!(
4993            counted * 100 < MAX_LIST_STRUCTURAL_CHARS,
4994            "ordinary traffic is within 100x of the cap ({counted} against \
4995             {MAX_LIST_STRUCTURAL_CHARS}), which is not the headroom the cap claims",
4996        );
4997    }
4998
4999    /// **An escaped quote does not end the string**, asserted without a magic
5000    /// number: the same document with the escape replaced by a plain letter has
5001    /// the same structure, so it must count the same. Get the escape wrong and the
5002    /// scanner leaves the string early and counts the rest as structure.
5003    #[test]
5004    fn an_escaped_quote_does_not_end_the_string() {
5005        let escaped = br#"{"records":[{"uri":"a\"b","value":{}}],"cursor":"x"}"#;
5006        let plain = br#"{"records":[{"uri":"axb","value":{}}],"cursor":"x"}"#;
5007        assert_eq!(
5008            count_structural_chars(escaped),
5009            count_structural_chars(plain),
5010            "an escaped quote changed the structure count"
5011        );
5012        let backslash = br#"{"records":[],"cursor":"x\\"}"#;
5013        let letter = br#"{"records":[],"cursor":"xy"}"#;
5014        assert_eq!(
5015            count_structural_chars(backslash),
5016            count_structural_chars(letter),
5017            "an escaped backslash changed the structure count"
5018        );
5019    }
5020
5021    /// A string is itself a node, so a page of strings costs more than a page of
5022    /// numbers. Without that, an array of a million short strings reads as cheap.
5023    #[test]
5024    fn a_string_counts_as_a_node() {
5025        assert!(
5026            count_structural_chars(br#"["a","b","c"]"#) > count_structural_chars(br#"[1,1,1]"#),
5027            "strings were not counted, so an array of them looks free"
5028        );
5029    }
5030
5031    #[tokio::test]
5032    async fn the_sidecar_refuses_a_node_explosion_too() {
5033        let mut data =
5034            String::from(r#"{"ok":true,"data":{"records":[{"uri":"at://d/c/r","value":["#);
5035        for _ in 0..1_200_000 {
5036            data.push_str("{},");
5037        }
5038        data.push_str(r#"{}]}]}}"#);
5039        let base = crate::net::tests::serve_body(data.into_bytes()).await;
5040        let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
5041        let err = client
5042            .list_records("did:plc:ewvi7nxzyoun6zhxrhs64oiz", "c", None, None)
5043            .await
5044            .expect_err("a node explosion reached the parser");
5045        assert!(
5046            format!("{err:#}").contains("structural characters"),
5047            "failed for the wrong reason: {err:#}"
5048        );
5049    }
5050
5051    // -- walk byte budget ---------------------------------------------------
5052
5053    /// What a parsed value really costs, counted independently of the code
5054    /// under test: every node occupies a `Value`, wherever it sits.
5055    fn node_count(v: &serde_json::Value) -> usize {
5056        1 + match v {
5057            serde_json::Value::Array(a) => a.iter().map(node_count).sum::<usize>(),
5058            serde_json::Value::Object(o) => o.values().map(node_count).sum::<usize>(),
5059            _ => 0,
5060        }
5061    }
5062
5063    fn record_of(value: serde_json::Value) -> RecordEntry {
5064        RecordEntry {
5065            uri: "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/c/3lab".to_string(),
5066            cid: Some("bafyreiabc123def456ghi789jkl012mno345pqr678stu901".to_string()),
5067            value,
5068        }
5069    }
5070
5071    /// **The estimate must never under-report, on any shape.**
5072    ///
5073    /// The version this replaces charged serialized length, which is accurate
5074    /// on prose-shaped records and 42x optimistic on the shapes an attacker
5075    /// picks. A bound that is only correct on benign input is not a bound.
5076    #[test]
5077    fn the_estimate_charges_every_node_at_least_what_a_parsed_value_costs() {
5078        let deep: serde_json::Value =
5079            serde_json::from_str(&format!("{}{}", "[".repeat(100), "]".repeat(100))).unwrap();
5080        let shapes: Vec<(&str, serde_json::Value)> = vec![
5081            ("100 nested empty arrays", deep),
5082            (
5083                "4096 empty arrays",
5084                serde_json::json!(vec![serde_json::json!([]); 4096]),
5085            ),
5086            ("4096 empty strings", serde_json::json!(vec![""; 4096])),
5087            (
5088                "4096 nulls",
5089                serde_json::json!(vec![serde_json::Value::Null; 4096]),
5090            ),
5091            ("4096 bools", serde_json::json!(vec![true; 4096])),
5092            ("4096 small numbers", serde_json::json!(vec![0; 4096])),
5093            (
5094                "object with short keys",
5095                serde_json::Value::Object(
5096                    (0..4096)
5097                        .map(|i| (format!("k{i}"), serde_json::json!([])))
5098                        .collect(),
5099                ),
5100            ),
5101            (
5102                "a realistic document",
5103                serde_json::json!({
5104                    "$type": "site.standard.document",
5105                    "title": "A post with a reasonably typical title",
5106                    "path": "/posts/one",
5107                    "publishedAt": "2026-07-11T09:30:00Z",
5108                    "textContent": "x".repeat(17_000),
5109                }),
5110            ),
5111        ];
5112        for (label, value) in shapes {
5113            let entry = record_of(value);
5114            let charged = approx_bytes(&entry);
5115            let floor = node_count(&entry.value) * std::mem::size_of::<serde_json::Value>();
5116            assert!(
5117                charged >= floor,
5118                "{label}: charged {charged} for {} nodes, which cannot cost less than {floor}",
5119                node_count(&entry.value)
5120            );
5121            let wire = serde_json::to_vec(&entry.value).unwrap().len();
5122            assert!(
5123                charged >= wire,
5124                "{label}: charged {charged}, under the {wire} bytes it takes on the wire alone"
5125            );
5126        }
5127    }
5128
5129    /// **Known answers, taken from a real allocator elsewhere.**
5130    ///
5131    /// The property above models `Value` nodes and nothing else, which is how an
5132    /// object-shaped under-charge of about half slipped past it: a
5133    /// `serde_json::Map` is a `BTreeMap` whose leaf is allocated whole, so the
5134    /// entries' own nodes are not the cost.
5135    ///
5136    /// **This test does not measure anything.** The two figures were obtained
5137    /// with a counting global allocator against the `serde_json` in this
5138    /// lockfile and are hardcoded here, because a global allocator is not
5139    /// something to install in the suite for one assertion. That makes this a
5140    /// tripwire for the *estimate* changing, not for the *real cost* changing: a
5141    /// dependency or toolchain bump that grows a map's true footprint leaves this
5142    /// green and the estimate quietly short again. Re-taking these numbers is the
5143    /// price of trusting them.
5144    #[test]
5145    fn the_estimate_covers_shapes_measured_against_a_real_allocator() {
5146        let many_small = serde_json::json!(vec![serde_json::json!({"a": 0}); 5000]);
5147        let mut deep = serde_json::json!({"a": 0});
5148        for _ in 0..99 {
5149            deep = serde_json::json!({ "a": deep });
5150        }
5151        for (label, value, measured) in [
5152            ("5000 one-key objects", many_small, 3_430_000usize),
5153            ("a 100-deep chain of one-key objects", deep, 63_350),
5154        ] {
5155            let charged = approx_bytes(&record_of(value));
5156            assert!(
5157                charged >= measured,
5158                "{label}: charged {charged} against {measured} bytes actually held"
5159            );
5160        }
5161    }
5162
5163    #[test]
5164    fn the_estimate_counts_the_uri_and_cid_too() {
5165        let bare = RecordEntry {
5166            uri: String::new(),
5167            cid: None,
5168            value: serde_json::json!(null),
5169        };
5170        let addressed = record_of(serde_json::json!(null));
5171        assert!(
5172            approx_bytes(&addressed) > approx_bytes(&bare),
5173            "a record's own identifiers are retained alongside its value"
5174        );
5175    }
5176
5177    #[test]
5178    fn the_budget_admits_a_page_that_exactly_fills_it() {
5179        let page = vec![record_of(serde_json::json!({"t": "x".repeat(1000)}))];
5180        let exact: usize = page.iter().map(approx_bytes).sum();
5181        assert!(
5182            ByteBudget::new(exact).admit(&page),
5183            "a page that exactly fits was refused; the fence-post is one byte out"
5184        );
5185        assert!(
5186            !ByteBudget::new(exact - 1).admit(&page),
5187            "a page one byte over the budget was admitted"
5188        );
5189    }
5190
5191    #[test]
5192    fn a_refused_page_leaves_the_running_total_alone() {
5193        let small = vec![record_of(serde_json::json!({"t": "x".repeat(100)}))];
5194        let huge = vec![record_of(serde_json::json!({"t": "x".repeat(100_000)}))];
5195        let cost: usize = small.iter().map(approx_bytes).sum();
5196        let mut budget = ByteBudget::new(cost * 3);
5197
5198        assert!(budget.admit(&small), "the first page fits");
5199        let after_one = budget.used();
5200        assert!(after_one > 0, "an admitted page must be charged");
5201
5202        assert!(!budget.admit(&huge), "the oversized page must be refused");
5203        assert_eq!(
5204            budget.used(),
5205            after_one,
5206            "a refused page moved the total — either charged, or reset"
5207        );
5208        assert!(
5209            budget.admit(&small),
5210            "the walk could not continue against the total it had before the refusal"
5211        );
5212    }
5213
5214    /// Build `pages` responses, each holding one record of about `bytes`, each
5215    /// pointing at the next. Returns the base URL and what one page costs.
5216    ///
5217    /// **Pages that differ is the whole point.** A walk served the same body
5218    /// twice is refused by its repeated-cursor guard, so every test built on the
5219    /// fixed-body server refuses on page one and never exercises accumulation
5220    /// at all — which is how a per-page budget once passed a whole suite.
5221    pub(crate) fn paged_bodies(
5222        pages: usize,
5223        bytes: usize,
5224        envelope: bool,
5225    ) -> (Vec<Vec<u8>>, usize) {
5226        let record = |i: usize| {
5227            serde_json::json!({
5228                "uri": format!("at://did:plc:ohutz6x5acjmpuulp3x7wxxc/c/3lab{i}"),
5229                "cid": "bafyreiabc123def456ghi789jkl012mno345pqr678stu901",
5230                "value": { "t": "x".repeat(bytes) }
5231            })
5232        };
5233        let bodies = (0..pages)
5234            .map(|i| {
5235                let mut page = serde_json::json!({ "records": [record(i)] });
5236                if i + 1 < pages {
5237                    page["cursor"] = serde_json::json!(format!("p{}", i + 1));
5238                }
5239                if envelope {
5240                    page = serde_json::json!({ "ok": true, "data": page });
5241                }
5242                page.to_string().into_bytes()
5243            })
5244            .collect();
5245        let entry: RecordEntry = serde_json::from_value(record(0)).unwrap();
5246        (bodies, approx_bytes(&entry))
5247    }
5248
5249    async fn host_for(bodies: Vec<Vec<u8>>, host: &str) -> (String, u16) {
5250        let base = crate::net::tests::serve_bodies_in_sequence(bodies).await;
5251        let port: u16 = base
5252            .trim_end_matches('/')
5253            .rsplit(':')
5254            .next()
5255            .unwrap()
5256            .parse()
5257            .unwrap();
5258        crate::net::test_host_override(host, std::net::SocketAddr::from(([127, 0, 0, 1], port)));
5259        (format!("http://{host}:{port}"), port)
5260    }
5261
5262    // ---- #177: one malformed envelope ---------------------------------------
5263
5264    /// A record with no `uri`, which is what #177 probed on `main`.
5265    fn malformed_page(cursor: Option<&str>) -> Vec<u8> {
5266        let mut page = serde_json::json!({
5267            "records": [
5268                { "uri": "at://did:plc:x/c/3labGOOD", "value": {} },
5269                { "cid": "bafy", "value": {} },
5270            ]
5271        });
5272        if let Some(c) = cursor {
5273            page["cursor"] = serde_json::json!(c);
5274        }
5275        page.to_string().into_bytes()
5276    }
5277
5278    #[test]
5279    fn one_malformed_envelope_is_counted_not_fatal_to_the_page() {
5280        let page = parse_list_records(&malformed_page(None))
5281            .expect("one malformed envelope failed the whole page");
5282        assert_eq!(page.records.len(), 1, "the good record was not kept");
5283        assert_eq!(page.records[0].uri, "at://did:plc:x/c/3labGOOD");
5284        assert_eq!(page.malformed, 1, "the malformed record was not counted");
5285    }
5286
5287    /// The publications listing in `standard_site::fetch` reads a stranger's
5288    /// repo through this walk: it skips, counts, and keeps paging.
5289    #[tokio::test]
5290    async fn the_skipping_walk_skips_malformed_records_and_keeps_paging() {
5291        let only_bad = serde_json::json!({
5292            "records": [{ "cid": "bafy", "value": {} }], "cursor": "p1"
5293        })
5294        .to_string()
5295        .into_bytes();
5296        let bodies = vec![
5297            only_bad,
5298            malformed_page(Some("p2")),
5299            serde_json::json!({ "records": [] })
5300                .to_string()
5301                .into_bytes(),
5302        ];
5303        let (base, _) = host_for(bodies, "skipping-walk-malformed.test").await;
5304        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5305        let (records, skipped) = client
5306            .list_all_records_skipping_within("c", &mut ByteBudget::new(MAX_LIST_BYTES))
5307            .await
5308            .expect("a malformed record failed a stranger's walk");
5309        assert_eq!(
5310            records.len(),
5311            1,
5312            "the good record behind the bad page was lost"
5313        );
5314        assert_eq!(skipped, 2, "skipped records were not counted");
5315    }
5316
5317    /// Pages of nothing but junk records, each ~`junk` bytes, with fresh
5318    /// cursors, so only the budget can stop the walk.
5319    fn junk_pages(n: usize, junk: usize) -> Vec<Vec<u8>> {
5320        (0..n)
5321            .map(|i| {
5322                serde_json::json!({
5323                    "records": [{ "cid": "bafy", "value": "x".repeat(junk) }],
5324                    "cursor": format!("p{}", i + 1),
5325                })
5326                .to_string()
5327                .into_bytes()
5328            })
5329            .collect()
5330    }
5331
5332    /// Review of #224: skipped records were never charged, so a stranger's
5333    /// repo serving junk pages walked all MAX_LIST_PAGES of them — gigabytes
5334    /// per poll — under a budget meant to stop at 128 MiB.
5335    #[tokio::test]
5336    async fn skipped_records_are_charged_against_the_budget() {
5337        let (base, _) = host_for(junk_pages(40, 256 * 1024), "junk-recent.test").await;
5338        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5339        let mut budget = ByteBudget::new(1024 * 1024);
5340        let walk = client
5341            .list_recent_matching_within("c", 100, &mut budget, 25, |_| true)
5342            .await
5343            .unwrap();
5344        assert!(
5345            !walk.complete,
5346            "40 pages of junk were walked to the end under a 1 MiB budget"
5347        );
5348
5349        let (base, _) = host_for(junk_pages(40, 256 * 1024), "junk-skipping.test").await;
5350        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5351        client
5352            .list_all_records_skipping_within("c", &mut ByteBudget::new(1024 * 1024))
5353            .await
5354            .expect_err("40 pages of junk were walked to the end under a 1 MiB budget");
5355    }
5356
5357    /// Review of #224: a page that skipped a record was charged its whole
5358    /// wire size AND its good records again, so a large publication near the
5359    /// budget failed only because one tiny malformed record sat beside it.
5360    #[tokio::test]
5361    async fn a_skipped_record_does_not_double_charge_its_page() {
5362        let page = |i: usize, with_bad: bool| {
5363            let mut records = vec![serde_json::json!({
5364                "uri": format!("at://did:plc:x/c/3lab{i}"), "value": "v".repeat(100_000)
5365            })];
5366            if with_bad {
5367                records.push(serde_json::json!({ "cid": "b", "value": {} }));
5368            }
5369            let mut body = serde_json::json!({ "records": records });
5370            if i < 4 {
5371                body["cursor"] = serde_json::json!(format!("p{}", i + 1));
5372            }
5373            body.to_string().into_bytes()
5374        };
5375        // A budget that holds the five clean pages with room to spare, and
5376        // less than twice that.
5377        let clean: Vec<_> = (0..5).map(|i| page(i, false)).collect();
5378        let (base, _) = host_for(clean, "double-charge-clean.test").await;
5379        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5380        let mut budget = ByteBudget::new(MAX_LIST_BYTES);
5381        let (clean_records, _) = client
5382            .list_all_records_skipping_within("c", &mut budget)
5383            .await
5384            .unwrap();
5385        let fits = budget.used() + budget.used() / 2;
5386
5387        let mixed: Vec<_> = (0..5).map(|i| page(i, true)).collect();
5388        let (base, _) = host_for(mixed, "double-charge-mixed.test").await;
5389        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5390        let (records, skipped) = client
5391            .list_all_records_skipping_within("c", &mut ByteBudget::new(fits))
5392            .await
5393            .expect("one tiny malformed record per page failed a walk that fits");
5394        assert_eq!(records.len(), clean_records.len());
5395        assert_eq!(skipped, 5);
5396
5397        // The documents walk, the same way: a page that skipped a record pays
5398        // its wire size once, not that and its kept records again.
5399        let mixed: Vec<_> = (0..5).map(|i| page(i, true)).collect();
5400        let (base, _) = host_for(mixed, "double-charge-recent.test").await;
5401        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5402        let walk = client
5403            .list_recent_matching_within("c", 100, &mut ByteBudget::new(fits), 25, |_| true)
5404            .await
5405            .unwrap();
5406        assert!(
5407            walk.complete,
5408            "one tiny malformed record per page cut short a walk that fits"
5409        );
5410        assert_eq!(walk.records.len(), clean_records.len());
5411    }
5412
5413    /// Third review of #224: charging a page that skipped a record only its
5414    /// wire size let a walk retain what it never paid for — a parsed record
5415    /// can hold up to 42x its wire size. Pages of dense `[[],[],…]` values, each
5416    /// with one malformed record beside them, retained 18x the budget.
5417    #[tokio::test]
5418    async fn a_page_that_skipped_a_record_still_pays_for_what_it_keeps() {
5419        let dense = format!("[{}]", vec!["[]"; 2000].join(","));
5420        let pages = |with_bad: bool| -> Vec<Vec<u8>> {
5421            (0..3)
5422                .map(|i| {
5423                    let mut records: Vec<String> = (0..99)
5424                        .map(|r| {
5425                            format!(r#"{{"uri":"at://did:plc:x/c/3l{i}x{r}","value":{dense}}}"#)
5426                        })
5427                        .collect();
5428                    if with_bad {
5429                        records.push(r#"{"cid":"b","value":{}}"#.to_string());
5430                    }
5431                    let cursor = if i < 2 {
5432                        format!(r#","cursor":"p{}""#, i + 1)
5433                    } else {
5434                        String::new()
5435                    };
5436                    format!(r#"{{"records":[{}]{cursor}}}"#, records.join(",")).into_bytes()
5437                })
5438                .collect()
5439        };
5440        const BUDGET: usize = 4 * 1024 * 1024;
5441        // Control: without the malformed records the walk is refused.
5442        let (base, _) = host_for(pages(false), "retained-control.test").await;
5443        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5444        let err = client
5445            .list_all_records_skipping_within("c", &mut ByteBudget::new(BUDGET))
5446            .await
5447            .expect_err("control: the clean pages fit a budget they exceed");
5448        assert_eq!(
5449            crate::feed::publication_failure_kind(&err),
5450            crate::feed::FailureKind::Body,
5451            "{err:#}"
5452        );
5453
5454        let (base, _) = host_for(pages(true), "retained-skipping.test").await;
5455        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5456        let err = client
5457            .list_all_records_skipping_within("c", &mut ByteBudget::new(BUDGET))
5458            .await
5459            .expect_err("one malformed record per page bought pages the budget refuses");
5460        assert!(format!("{err:#}").contains("malformed"), "{err:#}");
5461        assert_eq!(
5462            crate::feed::publication_failure_kind(&err),
5463            crate::feed::FailureKind::Body,
5464            "{err:#}"
5465        );
5466
5467        // The documents walk, with pages that each FIT the remaining budget
5468        // but together exceed it: only charging what the kept records retain
5469        // can stop it. (Pages each bigger than the budget are stopped by the
5470        // transient check alone and prove nothing about the charge.)
5471        let small_dense = format!("[{}]", vec!["[]"; 120].join(","));
5472        let fitting_pages: Vec<Vec<u8>> = (0..6)
5473            .map(|i| {
5474                let mut records: Vec<String> = (0..99)
5475                    .map(|r| {
5476                        format!(r#"{{"uri":"at://did:plc:x/c/3m{i}x{r}","value":{small_dense}}}"#)
5477                    })
5478                    .collect();
5479                records.push(r#"{"cid":"b","value":{}}"#.to_string());
5480                let cursor = if i < 5 {
5481                    format!(r#","cursor":"q{}""#, i + 1)
5482                } else {
5483                    String::new()
5484                };
5485                format!(r#"{{"records":[{}]{cursor}}}"#, records.join(",")).into_bytes()
5486            })
5487            .collect();
5488        let (base, _) = host_for(fitting_pages, "retained-recent.test").await;
5489        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5490        let walk = client
5491            .list_recent_matching_within("c", 10_000, &mut ByteBudget::new(BUDGET), 100, |_| true)
5492            .await
5493            .unwrap();
5494        let retained: usize = walk.records.iter().map(approx_bytes).sum();
5495        assert!(
5496            retained <= BUDGET,
5497            "retained {retained} under a {BUDGET}-byte budget"
5498        );
5499        assert!(
5500            !walk.complete,
5501            "pages totalling more than the budget were all kept"
5502        );
5503    }
5504
5505    /// The reader's own repo: this walk feeds `replace_sub_refs`, so skipping
5506    /// would drop a subscription silently. It refuses, by type.
5507    #[tokio::test]
5508    async fn the_own_repo_walk_refuses_a_page_with_a_malformed_record() {
5509        let (base, _) = host_for(vec![malformed_page(None)], "own-repo-malformed.test").await;
5510        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5511        let err = client
5512            .list_all_records("c")
5513            .await
5514            .expect_err("a page with a malformed record was accepted");
5515        let refused = err
5516            .downcast_ref::<MalformedRecords>()
5517            .unwrap_or_else(|| panic!("refused for the wrong reason: {err:#}"));
5518        assert_eq!(refused.count, 1);
5519        assert_eq!(refused.collection, "c");
5520    }
5521
5522    #[tokio::test]
5523    async fn the_sidecar_walk_refuses_a_page_with_a_malformed_record() {
5524        let body = serde_json::json!({
5525            "ok": true,
5526            "data": { "records": [
5527                { "uri": "at://did:plc:x/c/3labGOOD", "value": {} },
5528                { "cid": "bafy", "value": {} },
5529            ]}
5530        })
5531        .to_string()
5532        .into_bytes();
5533        let base = crate::net::tests::serve_bodies_in_sequence(vec![body]).await;
5534        let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
5535        let err = client
5536            .list_all_records("did:plc:x", "c")
5537            .await
5538            .expect_err("a page with a malformed record was accepted");
5539        assert!(
5540            err.downcast_ref::<MalformedRecords>().is_some(),
5541            "refused for the wrong reason: {err:#}"
5542        );
5543    }
5544
5545    /// A stranger's publication: skipping is right here, and the walk must keep
5546    /// paging past a page whose ONLY records were malformed — that page is not
5547    /// the end of the collection.
5548    #[tokio::test]
5549    async fn a_publication_walk_skips_malformed_records_and_keeps_paging() {
5550        let only_bad = serde_json::json!({
5551            "records": [{ "cid": "bafy", "value": {} }], "cursor": "p1"
5552        })
5553        .to_string()
5554        .into_bytes();
5555        let bodies = vec![
5556            only_bad,
5557            malformed_page(Some("p2")),
5558            serde_json::json!({ "records": [] })
5559                .to_string()
5560                .into_bytes(),
5561        ];
5562        let (base, _) = host_for(bodies, "publication-malformed.test").await;
5563        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5564        let mut budget = ByteBudget::new(MAX_LIST_BYTES);
5565        let walk = client
5566            .list_recent_matching_within("c", 100, &mut budget, 25, |_| true)
5567            .await
5568            .expect("a malformed record failed a stranger's publication walk");
5569        assert_eq!(
5570            walk.records.len(),
5571            1,
5572            "the good record behind the bad page was lost"
5573        );
5574        assert_eq!(walk.malformed, 2, "skipped records were not counted");
5575        assert!(
5576            walk.complete,
5577            "the walk stopped at a page of only malformed records"
5578        );
5579    }
5580
5581    /// **The budget is spent across pages, not reset by each one.**
5582    ///
5583    /// The single test this project most needed and did not have. Without it,
5584    /// moving the budget's construction inside the page loop — making the cap
5585    /// 200x weaker and effectively inert — passed every test in the suite.
5586    #[tokio::test]
5587    async fn a_refusing_walk_spends_its_budget_across_pages() {
5588        let (bodies, per_page) = paged_bodies(3, 4096, false);
5589        let (base, _) = host_for(bodies, "budget-accumulate.test").await;
5590        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5591
5592        let err = client
5593            .list_all_records_within("c", &mut ByteBudget::new(per_page * 2))
5594            .await
5595            .expect_err("three pages cannot fit in a two-page budget");
5596        let msg = format!("{err:#}");
5597        assert!(msg.contains("byte cap"), "wrong bound reported: {msg}");
5598        assert!(
5599            msg.contains("2 held"),
5600            "the walk did not keep exactly the two pages that fit: {msg}"
5601        );
5602        assert_eq!(
5603            crate::feed::publication_failure_kind(&err),
5604            crate::feed::FailureKind::Body,
5605            "{msg}"
5606        );
5607    }
5608
5609    #[tokio::test]
5610    async fn a_truncating_walk_keeps_the_pages_that_fit() {
5611        let (bodies, per_page) = paged_bodies(3, 4096, false);
5612        let (base, _) = host_for(bodies, "budget-accumulate-trunc.test").await;
5613        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5614
5615        let walk = client
5616            .list_recent_matching_within("c", 100, &mut ByteBudget::new(per_page * 2), 100, |_| {
5617                true
5618            })
5619            .await
5620            .expect("an additive walk truncates rather than failing");
5621        assert_eq!(
5622            walk.records.len(),
5623            2,
5624            "the pages that fit were not kept, or the refused one was"
5625        );
5626        assert!(
5627            !walk.complete,
5628            "a walk stopped by the budget called itself complete"
5629        );
5630    }
5631
5632    /// **The budget must not bind before the record cap does, with room spare.**
5633    ///
5634    /// The walks that carry `MAX_LIST_RECORDS` REFUSE when a bound is hit, and a
5635    /// refusal drops the reader into `resolve_subscriptions`' fail-closed branch
5636    /// — so an account near the record cap would serve a stale projection on
5637    /// every poll, forever. The figure quoted in `MAX_LIST_BYTES`'s own comment
5638    /// is this calculation, and a review caught that figure being wrong by a
5639    /// factor of two because nothing computed it. This does.
5640    ///
5641    /// Double, not merely under: the margin is what stops a slightly longer
5642    /// title or one more optional field from turning a working account into a
5643    /// permanently failing one.
5644    #[test]
5645    fn a_full_subscription_repo_fits_the_budget_twice_over() {
5646        let record = record_of(serde_json::json!({
5647            "$type": "community.lexicon.rss.subscription",
5648            "url": "https://example.com/blog/feed.xml",
5649            "title": "Some Blog With A Longish Name",
5650            "siteUrl": "https://example.com/blog",
5651            "createdAt": "2026-07-11T09:30:00Z",
5652            "folder": "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/community.lexicon.rss.folder/3lab999",
5653            "fetchHint": "hourly",
5654        }));
5655        let per_record = approx_bytes(&record);
5656        let full_repo = per_record * MAX_LIST_RECORDS;
5657        assert!(
5658            full_repo * 2 <= MAX_LIST_BYTES,
5659            "a full repo charges {per_record} B x {MAX_LIST_RECORDS} = {} MB against a {} MB \
5660             budget — too close for a walk whose verdict is a refusal",
5661            full_repo / (1024 * 1024),
5662            MAX_LIST_BYTES / (1024 * 1024)
5663        );
5664    }
5665
5666    /// **Running out of pages is a refusal, not a short answer.**
5667    ///
5668    /// The three refusing walks fell out of `for _ in 0..MAX_LIST_PAGES` into a
5669    /// bare `Ok(out)`, so a repo bigger than the page budget returned a truncated
5670    /// list that looks exactly like a complete one. `resolve_subscriptions` needs
5671    /// an `Err` to take its fail-closed branch; given `Ok` it hands the short list
5672    /// to `replace_sub_refs`, which DELETEs the reader's whole `sub_ref`
5673    /// projection and reinserts only what it was given. Everything past the cap
5674    /// is gone from their account, on an ordinary poll, with no attacker.
5675    ///
5676    /// `extend_bounded`'s refusal cannot catch this: `MAX_LIST_PAGES` x the 100
5677    /// records we ask for is exactly `MAX_LIST_RECORDS`, so against any server
5678    /// that honours `limit` the page budget runs out first, every time.
5679    #[tokio::test]
5680    async fn a_walk_that_runs_out_of_pages_refuses_rather_than_truncating() {
5681        // One more page than the budget, every page still offering a cursor.
5682        let bodies: Vec<Vec<u8>> = (0..MAX_LIST_PAGES + 1)
5683            .map(|i| {
5684                serde_json::json!({
5685                    "records": [{ "uri": format!("at://did:plc:x/c/3lab{i}"), "value": {} }],
5686                    "cursor": format!("p{}", i + 1),
5687                })
5688                .to_string()
5689                .into_bytes()
5690            })
5691            .collect();
5692        let (base, _) = host_for(bodies, "pages-exhausted.test").await;
5693        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5694
5695        let err = client
5696            .list_all_records("c")
5697            .await
5698            .expect_err("a truncated list was returned as a complete one");
5699        let msg = format!("{err:#}");
5700        assert!(
5701            msg.contains("did not finish"),
5702            "failed for the wrong reason: {msg}"
5703        );
5704        assert_eq!(
5705            crate::feed::publication_failure_kind(&err),
5706            crate::feed::FailureKind::Body,
5707            "{msg}"
5708        );
5709    }
5710
5711    /// **A walk that finishes cleanly across several pages still returns `Ok`.**
5712    ///
5713    /// The refusal's dangerous direction. Removing the flag's reset makes *every*
5714    /// multi-page walk refuse, which puts a reader with more than one page of
5715    /// records permanently into the fail-closed branch — and a review found that
5716    /// mutation surviving on the sidecar walk, which is the default backend,
5717    /// because nothing walked it to a clean finish and asserted success.
5718    #[tokio::test]
5719    async fn the_sidecar_walk_that_finishes_cleanly_returns_the_records() {
5720        let mut bodies: Vec<Vec<u8>> = (0..3)
5721            .map(|i| {
5722                serde_json::json!({
5723                    "ok": true,
5724                    "data": {
5725                        "records": [{ "uri": format!("at://did:plc:x/c/3lab{i}"), "value": {} }],
5726                        "cursor": format!("p{}", i + 1),
5727                    }
5728                })
5729                .to_string()
5730                .into_bytes()
5731            })
5732            .collect();
5733        // **The terminator CARRIES a record.** Ending on an empty page left a
5734        // second mutation alive: drop the last page's records and the assertion
5735        // below still counts three, because the last page had none to drop. A
5736        // real PDS ends on a partial page, and that page's records are the ones
5737        // an off-by-one loses.
5738        bodies.push(
5739            serde_json::json!({
5740                "ok": true,
5741                "data": { "records": [{ "uri": "at://did:plc:x/c/3labLAST", "value": {} }] }
5742            })
5743            .to_string()
5744            .into_bytes(),
5745        );
5746        let base = crate::net::tests::serve_bodies_in_sequence(bodies).await;
5747        let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
5748        let records = client
5749            .list_all_records("did:plc:ewvi7nxzyoun6zhxrhs64oiz", "c")
5750            .await
5751            .expect("a walk that ran out of records is not a short list");
5752        assert_eq!(
5753            records.len(),
5754            4,
5755            "the pages that were served were not all kept"
5756        );
5757        assert!(
5758            records.iter().any(|r| r.uri.ends_with("3labLAST")),
5759            "the LAST page's records were dropped — the walk kept the right \
5760             count only because every page held one: {:?}",
5761            records.iter().map(|r| r.uri.as_str()).collect::<Vec<_>>(),
5762        );
5763    }
5764
5765    /// **The page cap is pinned exactly, not to within one.**
5766    ///
5767    /// `the_sidecar_walk_that_runs_out_of_pages_refuses` serves
5768    /// `MAX_LIST_PAGES + 1` pages, so a budget one page SHORT refuses too and
5769    /// that mutation survives it. A walk whose last allowed request is the
5770    /// terminating one must come back `Ok` — which fails the moment the loop
5771    /// allows one page fewer, and is the direction that costs a reader their
5772    /// subscriptions.
5773    #[tokio::test]
5774    async fn a_sidecar_walk_that_terminates_on_its_last_allowed_page_succeeds() {
5775        let mut bodies: Vec<Vec<u8>> = (0..MAX_LIST_PAGES - 1)
5776            .map(|i| {
5777                serde_json::json!({
5778                    "ok": true,
5779                    "data": {
5780                        "records": [{ "uri": format!("at://did:plc:x/c/3lab{i}"), "value": {} }],
5781                        "cursor": format!("p{}", i + 1),
5782                    }
5783                })
5784                .to_string()
5785                .into_bytes()
5786            })
5787            .collect();
5788        // Request number `MAX_LIST_PAGES` — the last the loop allows — is the one
5789        // that terminates, and it carries a record of its own.
5790        bodies.push(
5791            serde_json::json!({
5792                "ok": true,
5793                "data": { "records": [{ "uri": "at://did:plc:x/c/3labLAST", "value": {} }] }
5794            })
5795            .to_string()
5796            .into_bytes(),
5797        );
5798        assert_eq!(bodies.len(), MAX_LIST_PAGES);
5799        let base = crate::net::tests::serve_bodies_in_sequence(bodies).await;
5800        let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
5801
5802        let records = client
5803            .list_all_records("did:plc:ewvi7nxzyoun6zhxrhs64oiz", "c")
5804            .await
5805            .expect("a walk that terminated inside its budget is not a short list");
5806        assert_eq!(
5807            records.len(),
5808            MAX_LIST_PAGES,
5809            "a walk that used its whole page budget and finished lost records",
5810        );
5811    }
5812
5813    /// **The direct walk's page cap, pinned exactly.**
5814    ///
5815    /// Twin of `a_sidecar_walk_that_terminates_on_its_last_allowed_page_succeeds`
5816    /// for the anonymous client. Verified needed: with only the `+ 1` refusal test
5817    /// above, `for _ in 0..MAX_LIST_PAGES - 1` left all 914 tests passing.
5818    #[tokio::test]
5819    async fn a_direct_walk_that_terminates_on_its_last_allowed_page_succeeds() {
5820        let mut bodies: Vec<Vec<u8>> = (0..MAX_LIST_PAGES - 1)
5821            .map(|i| {
5822                serde_json::json!({
5823                    "records": [{ "uri": format!("at://did:plc:x/c/3lab{i}"), "value": {} }],
5824                    "cursor": format!("p{}", i + 1),
5825                })
5826                .to_string()
5827                .into_bytes()
5828            })
5829            .collect();
5830        bodies.push(
5831            serde_json::json!({
5832                "records": [{ "uri": "at://did:plc:x/c/3labLAST", "value": {} }]
5833            })
5834            .to_string()
5835            .into_bytes(),
5836        );
5837        assert_eq!(bodies.len(), MAX_LIST_PAGES);
5838        let (base, _) = host_for(bodies, "last-allowed-page.test").await;
5839        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5840
5841        let records = client
5842            .list_all_records("c")
5843            .await
5844            .expect("a walk that terminated inside its budget is not a short list");
5845        assert_eq!(
5846            records.len(),
5847            MAX_LIST_PAGES,
5848            "a walk that used its whole page budget and finished lost records",
5849        );
5850        assert!(
5851            records.iter().any(|r| r.uri.ends_with("3labLAST")),
5852            "the LAST page's records were dropped",
5853        );
5854    }
5855
5856    /// **The TRUNCATING walk's page cap, pinned exactly — it reports completeness
5857    /// rather than refusing, so an off-by-one here is a silent short read.**
5858    ///
5859    /// A publication whose archive needs exactly the page budget to exhaust is
5860    /// `complete`; one page fewer makes it `complete = false`, which
5861    /// `store_publication` treats as a partial read. Verified needed:
5862    /// `for _ in 0..MAX_LIST_PAGES - 1` on this walk left all 914 tests passing.
5863    #[tokio::test]
5864    async fn a_truncating_walk_that_exhausts_on_its_last_allowed_page_is_complete() {
5865        let mut bodies: Vec<Vec<u8>> = (0..MAX_LIST_PAGES - 1)
5866            .map(|i| {
5867                serde_json::json!({
5868                    "records": [{ "uri": format!("at://did:plc:x/c/3lab{i}"), "value": {} }],
5869                    "cursor": format!("p{}", i + 1),
5870                })
5871                .to_string()
5872                .into_bytes()
5873            })
5874            .collect();
5875        bodies.push(
5876            serde_json::json!({
5877                "records": [{ "uri": "at://did:plc:x/c/3labLAST", "value": {} }]
5878            })
5879            .to_string()
5880            .into_bytes(),
5881        );
5882        assert_eq!(bodies.len(), MAX_LIST_PAGES);
5883        let (base, _) = host_for(bodies, "last-allowed-page-truncating.test").await;
5884        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5885
5886        // `max_records` well above what is served, so the cap under test is the
5887        // PAGE budget and not the record one.
5888        let walk = client
5889            .list_recent_matching("c", MAX_LIST_PAGES * 10, 1, |_| true)
5890            .await
5891            .expect("walk failed");
5892        assert_eq!(
5893            walk.records.len(),
5894            MAX_LIST_PAGES,
5895            "a walk that used its whole page budget and exhausted the collection \
5896             lost records",
5897        );
5898        assert!(
5899            walk.complete,
5900            "a collection that ran out on the last allowed page was reported as a \
5901             partial read, which is a starvation warning for a complete archive",
5902        );
5903    }
5904
5905    /// The sidecar walk refuses a short list too — and it is the default backend.
5906    #[tokio::test]
5907    async fn the_sidecar_walk_that_runs_out_of_pages_refuses() {
5908        let bodies: Vec<Vec<u8>> = (0..MAX_LIST_PAGES + 1)
5909            .map(|i| {
5910                serde_json::json!({
5911                    "ok": true,
5912                    "data": {
5913                        "records": [{ "uri": format!("at://did:plc:x/c/3lab{i}"), "value": {} }],
5914                        "cursor": format!("p{}", i + 1),
5915                    }
5916                })
5917                .to_string()
5918                .into_bytes()
5919            })
5920            .collect();
5921        let base = crate::net::tests::serve_bodies_in_sequence(bodies).await;
5922        let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
5923        let err = client
5924            .list_all_records("did:plc:ewvi7nxzyoun6zhxrhs64oiz", "c")
5925            .await
5926            .expect_err("a truncated list was returned as a complete one");
5927        assert!(
5928            format!("{err:#}").contains("did not finish"),
5929            "failed for the wrong reason: {err:#}"
5930        );
5931        assert_eq!(
5932            crate::feed::publication_failure_kind(&err),
5933            crate::feed::FailureKind::Body,
5934            "{err:#}"
5935        );
5936    }
5937
5938    /// **A budget passed to two walks is spent by both of them.**
5939    ///
5940    /// The reason it is passed rather than constructed: a publication read runs
5941    /// a second walk while still holding the first's records, so two independent
5942    /// ceilings let one read hold twice the bound. Here the first walk spends
5943    /// the budget and the second finds it spent.
5944    #[tokio::test]
5945    async fn two_walks_sharing_a_budget_do_not_each_get_the_whole_of_it() {
5946        let (bodies, per_page) = paged_bodies(4, 4096, false);
5947        let (base, _) = host_for(bodies, "budget-shared.test").await;
5948        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5949        let mut budget = ByteBudget::new(per_page * 3);
5950
5951        let err = client
5952            .list_all_records_within("c", &mut budget)
5953            .await
5954            .expect_err("four pages cannot fit a three-page budget");
5955        assert!(format!("{err:#}").contains("3 held"), "{err:#}");
5956
5957        // Same budget, nothing left in it.
5958        let err = client
5959            .list_all_records_within("c", &mut budget)
5960            .await
5961            .expect_err("the second walk was handed a fresh ceiling");
5962        assert!(
5963            format!("{err:#}").contains("0 held"),
5964            "the second walk kept something out of an exhausted budget: {err:#}"
5965        );
5966    }
5967
5968    /// **A transient page has to fit what is LEFT of the budget.**
5969    ///
5970    /// Measuring it against the ceiling lets a walk that has already retained
5971    /// most of its budget hold a further ceiling's worth of page on top. The
5972    /// filter keeps nothing here, so the running total cannot stop the walk and
5973    /// only the remaining-budget comparison can.
5974    #[tokio::test]
5975    async fn a_transient_page_must_fit_what_is_left_not_the_ceiling() {
5976        let (bodies, per_page) = paged_bodies(3, 4096, false);
5977        let (base, _) = host_for(bodies, "budget-remaining.test").await;
5978        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5979
5980        // Ceiling of two and a half pages, two of them already spent.
5981        let mut budget = ByteBudget::new(per_page * 5 / 2);
5982        let spent = vec![
5983            record_of(serde_json::json!({ "t": "x".repeat(4096) })),
5984            record_of(serde_json::json!({ "t": "x".repeat(4096) })),
5985        ];
5986        assert!(budget.admit(&spent), "the pre-spend has to fit");
5987        assert!(
5988            budget.remaining() < per_page,
5989            "and has to leave less than a page"
5990        );
5991
5992        let walk = client
5993            .list_recent_matching_within("c", 100, &mut budget, 100, |_| false)
5994            .await
5995            .expect("an additive walk truncates rather than failing");
5996        assert!(
5997            !walk.complete,
5998            "a page larger than the remaining budget was walked past"
5999        );
6000    }
6001
6002    /// **A filter that keeps nothing must not let the walk run unbounded.**
6003    ///
6004    /// The running total charges what is kept, so a filter matching nothing
6005    /// charges zero and the total can never stop the walk. What it holds is
6006    /// another matter: each page is fully parsed before the filter sees it, and
6007    /// `read_capped`'s 8 MB bounds the wire, not the tree. Only the per-page
6008    /// bound stands between that and the box.
6009    #[tokio::test]
6010    async fn a_filter_that_keeps_nothing_still_cannot_outrun_the_budget() {
6011        let (bodies, per_page) = paged_bodies(3, 4096, false);
6012        let (base, _) = host_for(bodies, "budget-filtered.test").await;
6013        let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
6014
6015        let walk = client
6016            .list_recent_matching_within("c", 100, &mut ByteBudget::new(per_page / 2), 100, |_| {
6017                false
6018            })
6019            .await
6020            .expect("an additive walk truncates rather than failing");
6021        assert!(
6022            !walk.complete,
6023            "a page too large to hold was walked past because the filter dropped it"
6024        );
6025        assert!(walk.records.is_empty(), "the filter kept nothing");
6026    }
6027
6028    #[tokio::test]
6029    async fn the_sidecar_walk_spends_its_budget_across_pages() {
6030        let (bodies, per_page) = paged_bodies(3, 4096, true);
6031        let base = crate::net::tests::serve_bodies_in_sequence(bodies).await;
6032        let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
6033
6034        let err = client
6035            .list_all_records_within(
6036                "did:plc:ewvi7nxzyoun6zhxrhs64oiz",
6037                "app.feather.subscription",
6038                &mut ByteBudget::new(per_page * 2),
6039            )
6040            .await
6041            .expect_err("the sidecar walk was the one with no budget at all");
6042        let msg = format!("{err:#}");
6043        assert!(msg.contains("byte cap"), "wrong bound reported: {msg}");
6044        assert!(
6045            msg.contains("2 held"),
6046            "did not accumulate across pages: {msg}"
6047        );
6048        assert_eq!(
6049            crate::feed::publication_failure_kind(&err),
6050            crate::feed::FailureKind::Body,
6051            "{msg}"
6052        );
6053    }
6054
6055    /// **Shapes that used to read as a healthy empty page.**
6056    ///
6057    /// Each of these was accepted by the `Value` route as `records: []`, and an
6058    /// empty page is not inert: `resolve_subscriptions` passes it to
6059    /// `replace_sub_refs`, which DELETEs the reader's projection and rewrites
6060    /// what it was handed. A page that is wrong in this direction costs them
6061    /// every feed.
6062    #[test]
6063    fn a_page_that_is_not_a_listing_is_never_read_as_an_empty_one() {
6064        for (label, body) in [
6065            (
6066                "a non-string error alongside records",
6067                &br#"{"error":404,"records":[]}"#[..],
6068            ),
6069            (
6070                "an object error alongside records",
6071                &br#"{"error":{"code":"x"},"records":[]}"#[..],
6072            ),
6073            (
6074                "a duplicated records key, the second one empty",
6075                &br#"{"records":[{"uri":"at://d/c/r","value":{}}],"records":[]}"#[..],
6076            ),
6077            ("an explicit null records", &br#"{"records":null}"#[..]),
6078        ] {
6079            assert!(
6080                parse_list_records(body).is_err(),
6081                "{label} was read as a page"
6082            );
6083        }
6084    }
6085
6086    /// **A non-string `error` is an envelope, and is reported as one.**
6087    ///
6088    /// Typing the field as a `String` made these fail as "invalid type" — the
6089    /// wrong reason for the exact shape the guard exists for, which is the same
6090    /// looseness that once let the guard be deleted unnoticed. So the reason is
6091    /// asserted, not just the refusal.
6092    #[test]
6093    fn a_non_string_error_is_reported_as_an_envelope() {
6094        for body in [
6095            &br#"{"error":404,"records":[]}"#[..],
6096            &br#"{"error":{"code":"x"},"records":[]}"#[..],
6097            &br#"{"error":[],"records":[]}"#[..],
6098            &br#"{"error":true,"records":[]}"#[..],
6099        ] {
6100            let err = parse_list_records(body)
6101                .expect_err("a non-string error envelope was read as an empty page");
6102            assert!(
6103                format!("{err:#}").contains("error envelope"),
6104                "{} failed for the wrong reason: {err:#}",
6105                String::from_utf8_lossy(body)
6106            );
6107        }
6108        // An empty name IS an envelope, as it was before this work: the route
6109        // this replaced keyed on `as_str`, so `Some("")` bailed. Exempting it
6110        // was a loosening made on speculation about proxy conventions, and a
6111        // loosening in this direction is a page accepted that used to be
6112        // refused.
6113        for body in [
6114            &br#"{"error":"","records":[]}"#[..],
6115            // Not zero on the wire, but zero once read: an exemption keyed on
6116            // `as_f64` swallowed anything that underflows.
6117            &br#"{"error":1e-400,"records":[]}"#[..],
6118        ] {
6119            let err = parse_list_records(body).expect_err("this is an envelope");
6120            assert!(
6121                format!("{err:#}").contains("error envelope"),
6122                "{} failed for the wrong reason: {err:#}",
6123                String::from_utf8_lossy(body)
6124            );
6125        }
6126        // The four spellings of "no error". `null` is what an ordinary listing
6127        // carries; `false` and integer `0` are a proxy convention, and refusing
6128        // those would fail a good page outright.
6129        for body in [
6130            &br#"{"error":null,"records":[]}"#[..],
6131            &br#"{"error":false,"records":[]}"#[..],
6132            &br#"{"error":0,"records":[]}"#[..],
6133            &br#"{"records":[]}"#[..],
6134        ] {
6135            assert!(
6136                parse_list_records(body).is_ok(),
6137                "{} is not an error envelope",
6138                String::from_utf8_lossy(body)
6139            );
6140        }
6141    }
6142
6143    /// **A non-string `error` is named by its type, never by its contents.**
6144    ///
6145    /// Rendering the value would serialise the whole attacker-chosen subtree
6146    /// before truncating it, allocating a full extra copy of up to the body cap
6147    /// — in a change whose purpose is cutting peak allocation. The earlier
6148    /// version of this did exactly that and the comment claimed otherwise.
6149    #[test]
6150    fn a_structured_error_is_named_by_its_type_not_serialised() {
6151        let payload = "s".repeat(20_000);
6152        let body = format!(r#"{{"error":{{"deep":"{payload}"}},"records":[]}}"#);
6153        let err = parse_list_records(body.as_bytes()).expect_err("an envelope is a refusal");
6154        let msg = format!("{err:#}");
6155        assert!(
6156            !msg.contains("ssss"),
6157            "the error's contents reached the message: {} chars",
6158            msg.len()
6159        );
6160        assert!(
6161            msg.contains("non-string error: object"),
6162            "it should name the shape instead: {msg}"
6163        );
6164    }
6165
6166    /// `data` absent is not `data` empty, on the sidecar envelope too.
6167    ///
6168    /// `{"ok":true}` is what a proxy makes of an unexpected upstream body, and
6169    /// reading it as a page of zero records is the wipe this whole family of
6170    /// guards exists to prevent.
6171    #[tokio::test]
6172    async fn the_sidecar_refuses_an_envelope_with_no_data() {
6173        for body in [&br#"{"ok":true}"#[..], &br#"{"ok":true,"data":{}}"#[..]] {
6174            let base = crate::net::tests::serve_body(body.to_vec()).await;
6175            let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
6176            let err = client
6177                .list_records(
6178                    "did:plc:ewvi7nxzyoun6zhxrhs64oiz",
6179                    "app.feather.subscription",
6180                    None,
6181                    None,
6182                )
6183                .await
6184                .expect_err("an envelope without a listing was read as an empty page");
6185            assert!(
6186                format!("{err:#}").contains("no records"),
6187                "{} failed for the wrong reason: {err:#}",
6188                String::from_utf8_lossy(body)
6189            );
6190        }
6191    }
6192
6193    /// **The sidecar gets the duplicated-key refusal too.**
6194    ///
6195    /// It was the one client still reading a listing through a `Value`, where a
6196    /// repeated key resolves last-wins — so a body carrying a second, empty
6197    /// `records` array read as a successful empty page, and an empty page on this
6198    /// path is `replace_sub_refs` deleting every `sub_ref` the reader has. It is
6199    /// also the default backend, so it was the one that mattered most.
6200    #[tokio::test]
6201    async fn the_sidecar_refuses_a_duplicated_records_key() {
6202        let base = crate::net::tests::serve_body(
6203            br#"{"ok":true,"data":{"records":[{"uri":"at://d/c/r","value":{}}],"records":[]}}"#
6204                .to_vec(),
6205        )
6206        .await;
6207        let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
6208        let err = client
6209            .list_records(
6210                "did:plc:ewvi7nxzyoun6zhxrhs64oiz",
6211                "app.feather.subscription",
6212                None,
6213                None,
6214            )
6215            .await
6216            .expect_err("a duplicated records key was read as an empty page");
6217        assert!(
6218            format!("{err:#}").contains("duplicate"),
6219            "failed for the wrong reason: {err:#}"
6220        );
6221    }
6222
6223    /// A non-string `message` must not fail an otherwise good page.
6224    /// **A listing has to be an object.**
6225    ///
6226    /// serde's derived `Deserialize` takes a struct positionally too, so with
6227    /// every field defaulted `[null,null,[]]` bound `records` to an empty vector
6228    /// and read as a healthy page — and a body with no keys defeats the envelope
6229    /// guard and the duplicated-key refusal at the same time, because neither has
6230    /// anything to look at. Fourteen bytes, and `replace_sub_refs` deletes every
6231    /// feed the reader has.
6232    #[test]
6233    fn a_listing_that_is_not_an_object_is_not_a_page() {
6234        for body in [
6235            &b"[null,null,[]]"[..],
6236            &b"[null,null,[],null]"[..],
6237            &br#"[null,null,[{"uri":"at://d/c/r","value":{}}],"c"]"#[..],
6238            &b"[]"[..],
6239            &br#""a string""#[..],
6240            &b"0"[..],
6241            &b"true"[..],
6242        ] {
6243            assert!(
6244                parse_list_records(body).is_err(),
6245                "{} was read as a page",
6246                String::from_utf8_lossy(body)
6247            );
6248        }
6249    }
6250
6251    /// **The rendering is bounded in bytes, whatever the input is made of.**
6252    ///
6253    /// Counting characters bounds nothing a log cares about: 120 astral-plane
6254    /// code points are 480 bytes. The invariant is on the output's byte length.
6255    #[test]
6256    fn a_truncated_message_is_bounded_in_bytes() {
6257        for (label, input) in [
6258            ("ascii", "e".repeat(50_000)),
6259            ("astral", "\u{1f600}".repeat(20_000)),
6260            (
6261                "mixed",
6262                format!("{}{}", "e".repeat(200), "\u{1f600}".repeat(200)),
6263            ),
6264            (
6265                "just over in bytes, just under in chars",
6266                "\u{1f600}".repeat(40),
6267            ),
6268        ] {
6269            let out = truncate_for_message(&input);
6270            assert!(
6271                out.len() <= 200,
6272                "{label}: rendered {} bytes from {} bytes of input",
6273                out.len(),
6274                input.len()
6275            );
6276        }
6277        // Short inputs pass through untouched.
6278        assert_eq!(truncate_for_message("Boom"), "Boom");
6279    }
6280
6281    /// **The sidecar has two envelope layers, and both are guards.**
6282    ///
6283    /// `page_from_body` covers the PDS's, which arrives inside `data`. The
6284    /// sidecar's own can say `ok:false` or carry its own `error` on a 200 while
6285    /// `data` still holds something that reads as a perfectly good empty page —
6286    /// and an empty page here is `replace_sub_refs` deleting every feed.
6287    #[tokio::test]
6288    async fn the_sidecar_refuses_its_own_error_envelope_on_a_2xx() {
6289        for body in [
6290            &br#"{"ok":false,"error":"ExpiredToken","data":{"records":[]}}"#[..],
6291            &br#"{"ok":true,"error":"ExpiredToken","data":{"records":[]}}"#[..],
6292            &br#"{"ok":false,"data":{"records":[]}}"#[..],
6293        ] {
6294            let base = crate::net::tests::serve_body(body.to_vec()).await;
6295            let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
6296            let err = client
6297                .list_records(
6298                    "did:plc:ewvi7nxzyoun6zhxrhs64oiz",
6299                    "app.feather.subscription",
6300                    None,
6301                    None,
6302                )
6303                .await
6304                .expect_err("the sidecar's own envelope was read as a page");
6305            let msg = format!("{err:#}");
6306            assert!(
6307                msg.contains("sidecar answered 2xx"),
6308                "{} failed for the wrong reason: {msg}",
6309                String::from_utf8_lossy(body)
6310            );
6311        }
6312    }
6313
6314    /// **An unknown field's contents are still validated.**
6315    ///
6316    /// `IgnoredAny` skips without validating, so a body that is not valid JSON at
6317    /// all read as a healthy empty page where the route this replaced refused it.
6318    #[test]
6319    fn an_unknown_field_holding_invalid_json_is_not_a_page() {
6320        for body in [
6321            &b"{\"records\":[],\"x\":\"\xff\xfe\"}"[..],
6322            &br#"{"records":[],"x":"\ud800"}"#[..],
6323        ] {
6324            assert!(
6325                parse_list_records(body).is_err(),
6326                "{} was read as a page",
6327                String::from_utf8_lossy(body)
6328            );
6329        }
6330    }
6331
6332    #[test]
6333    fn a_non_string_message_does_not_cost_the_page() {
6334        let page = parse_list_records(br#"{"records":[],"message":5,"cursor":"c"}"#)
6335            .expect("message carries no guard; typing it strictly failed whole listings");
6336        assert_eq!(page.cursor.as_deref(), Some("c"));
6337    }
6338
6339    /// The name that reaches the log is bounded, because the PDS chooses it.
6340    #[test]
6341    fn an_enormous_error_name_is_truncated_before_it_reaches_a_log() {
6342        let huge = "e".repeat(50_000);
6343        let body = format!(r#"{{"error":"{huge}","records":[]}}"#);
6344        let err = parse_list_records(body.as_bytes()).expect_err("an envelope is a refusal");
6345        let msg = format!("{err:#}");
6346        assert!(
6347            msg.len() < 400,
6348            "the error message carried {} bytes of attacker-chosen text",
6349            msg.len()
6350        );
6351        assert!(
6352            msg.contains("50000 bytes"),
6353            "it should say what it dropped: {msg}"
6354        );
6355
6356        // Astral-plane code points: the bound must hold in BYTES, because a log
6357        // line is bytes. Counting characters made this four times the stated cap.
6358        let wide = "\u{1f600}".repeat(20_000);
6359        let body = format!(r#"{{"error":"{wide}","message":"{wide}","records":[]}}"#);
6360        let err = parse_list_records(body.as_bytes()).expect_err("an envelope is a refusal");
6361        let msg = format!("{err:#}");
6362        assert!(
6363            msg.len() < 400,
6364            "a wide-character error rendered {} bytes",
6365            msg.len()
6366        );
6367    }
6368
6369    // -- TID rkeys ----------------------------------------------------------
6370
6371    #[test]
6372    fn tid_rkeys_are_13_char_s32_and_monotonic() {
6373        let mut gen = TidGenerator::new();
6374        let mut prev: Option<String> = None;
6375        for _ in 0..1000 {
6376            let tid = gen.next();
6377            assert_eq!(tid.len(), 13, "a TID is 13 s32 chars");
6378            assert!(
6379                tid.bytes().all(|b| S32_ALPHABET.contains(&b)),
6380                "TID {tid} uses only the s32 alphabet"
6381            );
6382            if let Some(p) = &prev {
6383                assert!(*p < tid, "TIDs must be strictly increasing ({p} < {tid})");
6384            }
6385            prev = Some(tid);
6386        }
6387    }
6388
6389    #[test]
6390    fn tid_rkeys_are_valid_atproto_record_keys() {
6391        // atproto rkey charset: [A-Za-z0-9._~:-], length 1..=512, not "."/"..".
6392        let mut gen = TidGenerator::new();
6393        let tid = gen.next();
6394        assert!(is_valid_rkey(&tid), "{tid:?}");
6395        assert!(tid
6396            .bytes()
6397            .all(|b| b.is_ascii_alphanumeric() || matches!(b, b'.' | b'_' | b'~' | b':' | b'-')));
6398    }
6399
6400    #[test]
6401    fn tid_values_round_trip_through_the_decoder() {
6402        // The decoder is the inverse of the encoder across the whole range a
6403        // TID can hold, boundaries included.
6404        let max_tid = (0x001f_ffff_ffff_ffffu64 << 10) | 0x3ff;
6405        for v in [0u64, 1, 31, 32, 1023, 1024, 1_000_000, max_tid] {
6406            let encoded = encode_s32_tid(v);
6407            assert_eq!(
6408                decode_s32_tid(&encoded),
6409                Some(v),
6410                "{v} encoded to {encoded}, which did not decode back"
6411            );
6412        }
6413
6414        // **A round trip alone proves too little.** Encoder and decoder share
6415        // the alphabet, so swapping two of its symbols round-trips perfectly
6416        // and still reads every real record key wrong. These two are the
6417        // known answer: a record key from a real atproto repo, and the value
6418        // it holds, computed independently of this code.
6419        assert_eq!(
6420            decode_s32_tid("3jzfcijpj2z2a"),
6421            Some(1_728_652_679_052_295_174)
6422        );
6423        assert_eq!(encode_s32_tid(1_728_652_679_052_295_174), "3jzfcijpj2z2a");
6424        assert_eq!(
6425            decode_s32_tid("3jzfcijpj2z2a").map(|raw| raw >> 10),
6426            Some(1_688_137_381_887_007),
6427            "that key was written at 2023-06-30T15:03:01.887007Z"
6428        );
6429    }
6430
6431    #[test]
6432    fn the_first_tid_of_a_generator_decodes_to_the_microsecond_it_was_minted() {
6433        let micros = || {
6434            std::time::SystemTime::now()
6435                .duration_since(std::time::UNIX_EPOCH)
6436                .map(|d| d.as_micros() as u64)
6437                .unwrap_or(0)
6438        };
6439        // The FIRST `next()` only. `TidGenerator` bumps a TID to `last + 1`
6440        // to stay strictly increasing, and on a generator whose clock id is
6441        // already at its maximum that carry lands in the timestamp bits — so a
6442        // later TID can decode a microsecond or two past when it was really
6443        // minted. A fresh generator has `last: 0`, where the bump cannot fire.
6444        let before = micros();
6445        let tid = TidGenerator::new().next();
6446        let after = micros();
6447        let raw = decode_s32_tid(&tid).expect("a generated TID must decode");
6448        let minted = raw >> 10;
6449        assert!(
6450            (before..=after).contains(&minted),
6451            "TID {tid} decoded to {minted}, outside the {before}..={after} window it was minted in"
6452        );
6453    }
6454
6455    #[test]
6456    fn the_decoder_rejects_strings_that_are_not_13_char_s32_values() {
6457        for rkey in [
6458            "",               // empty
6459            "self",           // the common non-TID rkey
6460            "3jzfcijpj2z2",   // 12 chars: one short
6461            "3jzfcijpj2z2aa", // 14 chars: one long
6462            "3jzfcijpj2z2A",  // uppercase is outside the s32 alphabet
6463            "3jzfcijpj2z-a",  // a legal rkey character, but not an s32 one
6464            "3jzfcijpj2z2!",  // not a legal rkey character at all
6465            "c222222222222",  // decodes with bit 63 set: the reserved top bit
6466            "k222222222222",  // decodes past 64 bits entirely
6467            "zzzzzzzzzzzzz",  // the largest 13-char s32 string
6468        ] {
6469            assert_eq!(
6470                decode_s32_tid(rkey),
6471                None,
6472                "{rkey:?} is not a 13-character s32 value"
6473            );
6474        }
6475    }
6476
6477    /// **The window is not a slug detector, and this is what that costs.**
6478    ///
6479    /// A 13-character slug beginning `3` decodes into the last few years just
6480    /// as a record key does, and nothing in the string tells them apart. These
6481    /// are read as dates, and pinning that here is the honest alternative to a
6482    /// doc comment claiming otherwise. The damage is bounded: a wrong date is
6483    /// an ordinary past instant that ages, sweeps and is outranked normally.
6484    #[test]
6485    fn a_slug_that_decodes_inside_the_window_is_read_as_a_date() {
6486        for (slug, reads_as) in [
6487            ("3hoursinparis", "2020-11-24T08:17:26Z"),
6488            ("3ideasforjune", "2021-08-12T00:19:38Z"),
6489            ("3jokesaweekly", "2023-02-12T15:50:26Z"),
6490        ] {
6491            assert_eq!(
6492                tid_timestamp(slug).map(crate::feed::fmt_time),
6493                Some(reads_as.to_string()),
6494                "{slug} is indistinguishable from a record key written then"
6495            );
6496        }
6497    }
6498
6499    #[test]
6500    fn a_tid_minted_slightly_ahead_of_our_clock_is_still_believed() {
6501        let now = chrono::Utc::now();
6502        let of = |at: chrono::DateTime<chrono::Utc>| {
6503            encode_s32_tid((at.timestamp_micros() as u64) << 10)
6504        };
6505        assert!(
6506            tid_timestamp(&of(now + chrono::Duration::seconds(2))).is_some(),
6507            "a PDS two seconds fast must not leave a fresh document undated"
6508        );
6509        assert_eq!(
6510            tid_timestamp(&of(now + chrono::Duration::hours(1))),
6511            None,
6512            "an hour ahead is a broken clock or a slug, not skew"
6513        );
6514    }
6515
6516    #[test]
6517    fn a_tid_timestamp_is_bounded_at_both_ends() {
6518        let now = chrono::Utc::now();
6519        let of = |micros: i64| encode_s32_tid((micros as u64) << 10);
6520
6521        // A TID minted now dates to now.
6522        let fresh = TidGenerator::new().next();
6523        let dated = tid_timestamp(&fresh).expect("a freshly minted TID has a timestamp");
6524        assert!(
6525            (now - chrono::Duration::minutes(1)..=now + chrono::Duration::minutes(1))
6526                .contains(&dated),
6527            "{fresh} dated to {dated}, not to now ({now})"
6528        );
6529
6530        // Before atproto existed: not a date.
6531        assert_eq!(
6532            tid_timestamp(&of(TID_FLOOR_MICROS - 1)),
6533            None,
6534            "a TID predating atproto must not date an entry"
6535        );
6536        assert!(
6537            tid_timestamp(&of(TID_FLOOR_MICROS)).is_some(),
6538            "the floor itself is a real instant"
6539        );
6540
6541        // In the future: not a date. A slug of 13 s32 characters lands here,
6542        // which is the case this bound exists for.
6543        let far_future = (now + chrono::Duration::days(365)).timestamp_micros();
6544        assert_eq!(
6545            tid_timestamp(&of(far_future)),
6546            None,
6547            "a TID from the future must not date an entry"
6548        );
6549        assert_eq!(
6550            tid_timestamp("abcdefghijklm"),
6551            None,
6552            "a 13-character slug decodes to the year 2192; it is not a date"
6553        );
6554    }
6555
6556    #[test]
6557    fn s32_encoding_is_ascending_for_ascending_values() {
6558        // The whole point of s32: numeric order == lexicographic string order.
6559        assert!(encode_s32_tid(1) < encode_s32_tid(2));
6560        assert!(encode_s32_tid(31) < encode_s32_tid(32));
6561        assert!(encode_s32_tid(1_000_000) < encode_s32_tid(1_000_001));
6562        // Ordering holds all the way to the largest real TID value (a 53-bit
6563        // microsecond timestamp shifted into bits 63..10, plus the clock id).
6564        let max_tid = (0x001f_ffff_ffff_ffffu64 << 10) | 0x3ff;
6565        assert!(encode_s32_tid(max_tid - 1) < encode_s32_tid(max_tid));
6566    }
6567    /// **Exceeding the record cap is an ERROR, not a silent truncation.**
6568    ///
6569    /// The page cap bounds how many requests a walk makes; it bounds the
6570    /// accumulated memory only if the server honours `limit=100`, and a host we
6571    /// did not choose has no obligation to. A review measured an 8 MB page
6572    /// holding ~95 000 minimal records and retaining 23 MB as
6573    /// `Vec&lt;RecordEntry&gt;` — 200 such pages is gigabytes on a 512 MB box.
6574    ///
6575    /// Truncating instead would be worse than the OOM it prevents. The caller
6576    /// of the live walk is `resolve_subscriptions`, whose result feeds
6577    /// `replace_sub_refs` — a `DELETE` plus reinsert of exactly what it was
6578    /// handed. A short list there is not a short list, it is **revoked access**
6579    /// to the feeds that fell off the end. That is the failure PR #167 was
6580    /// closed for reintroducing, so this returns `Err` and lets the existing
6581    /// fail-closed branch serve the last-known projection.
6582    #[test]
6583    fn exceeding_the_record_cap_is_an_error_not_a_truncation() {
6584        let page = |n: usize| -> Vec<RecordEntry> {
6585            (0..n)
6586                .map(|i| RecordEntry {
6587                    uri: format!("at://did:plc:x/c/{i}"),
6588                    cid: None,
6589                    value: serde_json::Value::Null,
6590                })
6591                .collect()
6592        };
6593
6594        let mut out = page(90);
6595        let err = extend_bounded(&mut out, page(20), 100, "c")
6596            .expect_err("a page past the cap was accepted");
6597        let msg = format!("{err:#}");
6598        assert!(msg.contains("100"), "the cap is not named: {msg}");
6599        assert_eq!(
6600            crate::feed::publication_failure_kind(&err),
6601            crate::feed::FailureKind::Body,
6602            "{msg}"
6603        );
6604        assert_eq!(
6605            out.len(),
6606            90,
6607            "the partial page was kept — a truncated list must not survive the error"
6608        );
6609    }
6610
6611    #[test]
6612    fn accumulating_within_the_cap_succeeds() {
6613        let page = |n: usize| -> Vec<RecordEntry> {
6614            (0..n)
6615                .map(|i| RecordEntry {
6616                    uri: format!("at://did:plc:x/c/{i}"),
6617                    cid: None,
6618                    value: serde_json::Value::Null,
6619                })
6620                .collect()
6621        };
6622        let mut out = Vec::new();
6623        extend_bounded(&mut out, page(60), 100, "c").unwrap();
6624        extend_bounded(&mut out, page(40), 100, "c").unwrap();
6625        assert_eq!(out.len(), 100, "exactly the cap must be allowed");
6626    }
6627
6628    // -- applyWrites chunking (#240) ----------------------------------------
6629
6630    /// Every `applyWrites` call a fake saw, as the `writes` array it carried.
6631    pub(crate) type ApplyWritesLog = Arc<std::sync::Mutex<Vec<Vec<Value>>>>;
6632
6633    /// A fake that answers `applyWrites` the way a strict PDS does, on BOTH
6634    /// shapes this crate sends it in: the PDS's own
6635    /// `/xrpc/com.atproto.repo.applyWrites` (the direct and OAuth clients) and
6636    /// the sidecar's `/internal/repo` (`action: "applyWrites"`).
6637    ///
6638    /// It refuses what the reference PDS refuses — more than 200 writes
6639    /// (`InvalidRequest: Too many writes. Max: 200`, from
6640    /// `packages/pds/src/api/com/atproto/repo/applyWrites.ts`) and a body over
6641    /// the 150 KiB `jsonLimit` every reference PDS before atproto#4989 applied
6642    /// to it — so a test that sends an unchunked batch fails the way production
6643    /// would, rather than passing against a fake that accepts anything.
6644    ///
6645    /// Every call is logged, refused or not, so a test can assert that a later
6646    /// chunk was never SENT. `fail_call` (1-based) answers that call with a 500.
6647    pub(crate) async fn serve_apply_writes(fail_call: Option<usize>) -> (String, ApplyWritesLog) {
6648        use axum::body::Bytes;
6649        use axum::http::{StatusCode as Status, Uri};
6650        use axum::response::IntoResponse;
6651
6652        const PRE_4989_JSON_LIMIT: usize = 150 * 1024;
6653        let log: ApplyWritesLog = Arc::default();
6654        let sink = Arc::clone(&log);
6655        let app = axum::Router::new()
6656            .fallback(move |uri: Uri, body: Bytes| {
6657                let sink = Arc::clone(&sink);
6658                async move {
6659                    let sidecar = uri.path() == "/internal/repo";
6660                    let reply = |status: Status, error: &str, message: &str| {
6661                        let body = if sidecar {
6662                            json!({ "ok": false, "error": error, "message": message, "status": status.as_u16() })
6663                        } else {
6664                            json!({ "error": error, "message": message })
6665                        };
6666                        (status, axum::Json(body)).into_response()
6667                    };
6668                    let parsed: Value = serde_json::from_slice(&body).unwrap_or(Value::Null);
6669                    // Anything else a handler sends on the way (a folder
6670                    // listing, say) is answered empty and not logged, so the
6671                    // log and `fail_call` count `applyWrites` calls only.
6672                    let Some(writes) = parsed["writes"].as_array().cloned() else {
6673                        let empty = json!({ "records": [] });
6674                        return if sidecar {
6675                            axum::Json(json!({ "ok": true, "data": empty })).into_response()
6676                        } else {
6677                            axum::Json(empty).into_response()
6678                        };
6679                    };
6680                    let call = {
6681                        let mut calls = sink.lock().unwrap();
6682                        calls.push(writes.clone());
6683                        calls.len()
6684                    };
6685                    if body.len() > PRE_4989_JSON_LIMIT {
6686                        return reply(
6687                            Status::PAYLOAD_TOO_LARGE,
6688                            "PayloadTooLarge",
6689                            "request entity too large",
6690                        );
6691                    }
6692                    if writes.len() > 200 {
6693                        return reply(
6694                            Status::BAD_REQUEST,
6695                            "InvalidRequest",
6696                            "Too many writes. Max: 200",
6697                        );
6698                    }
6699                    if fail_call == Some(call) {
6700                        return reply(
6701                            Status::INTERNAL_SERVER_ERROR,
6702                            "InternalServerError",
6703                            "boom",
6704                        );
6705                    }
6706                    let data =
6707                        json!({ "commit": { "cid": "bafycommit", "rev": "3l" }, "results": [] });
6708                    if sidecar {
6709                        axum::Json(json!({ "ok": true, "data": data })).into_response()
6710                    } else {
6711                        axum::Json(data).into_response()
6712                    }
6713                }
6714            })
6715            // The fake must see an oversized body to refuse it, not have axum
6716            // refuse it first at its own 2 MB default.
6717            .layer(axum::extract::DefaultBodyLimit::disable());
6718        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
6719        let addr = listener.local_addr().unwrap();
6720        tokio::spawn(async move { axum::serve(listener, app).await.unwrap() });
6721        (format!("http://{addr}"), log)
6722    }
6723
6724    /// `n` distinct vetted subscriptions, in a known order.
6725    fn vetted_subs(n: usize) -> Vec<crate::vetted::VettedSubscription> {
6726        (0..n)
6727            .map(|i| {
6728                crate::vetted::VettedSubscription::new(&lexicon::Subscription::new(
6729                    format!("https://f{i}.example/feed.xml"),
6730                    "2026-07-12T00:00:00.000Z",
6731                ))
6732            })
6733            .collect()
6734    }
6735
6736    /// The `rkey` of every write a run of calls carried, flattened in send order.
6737    pub(crate) fn sent_rkeys(log: &ApplyWritesLog) -> Vec<String> {
6738        log.lock()
6739            .unwrap()
6740            .iter()
6741            .flatten()
6742            .map(|w| w["rkey"].as_str().unwrap_or_default().to_string())
6743            .collect()
6744    }
6745
6746    /// How many writes each call carried, in send order.
6747    pub(crate) fn call_sizes(log: &ApplyWritesLog) -> Vec<usize> {
6748        log.lock().unwrap().iter().map(Vec::len).collect()
6749    }
6750
6751    const CHUNK_DID: &str = "did:plc:ewvi7nxzyoun6zhxrhs64oiz";
6752
6753    fn chunk_sidecar(base: &str) -> SidecarClient {
6754        SidecarClient::new(Client::new(), base, base, "secret")
6755    }
6756
6757    /// **201 writes are two calls, 200 then 1, in order.** The OPML import used
6758    /// to send every feed in one `applyWrites`, which the reference PDS refuses
6759    /// past 200 — so any import over 200 feeds failed outright.
6760    #[tokio::test]
6761    async fn sidecar_bulk_add_of_201_is_two_calls_in_order() {
6762        let (base, log) = serve_apply_writes(None).await;
6763        let rkeys = chunk_sidecar(&base)
6764            .add_subscriptions_bulk(CHUNK_DID, &vetted_subs(201))
6765            .await
6766            .expect("a 201-feed import must succeed against a PDS that caps at 200");
6767
6768        assert_eq!(call_sizes(&log), vec![200, 1]);
6769        let urls: Vec<String> = log
6770            .lock()
6771            .unwrap()
6772            .iter()
6773            .flatten()
6774            .map(|w| w["value"]["url"].as_str().unwrap().to_string())
6775            .collect();
6776        let expected: Vec<String> = (0..201)
6777            .map(|i| format!("https://f{i}.example/feed.xml"))
6778            .collect();
6779        assert_eq!(urls, expected, "ops must keep input order across chunks");
6780        assert_eq!(
6781            sent_rkeys(&log),
6782            rkeys,
6783            "the returned rkeys are the ones written, in order"
6784        );
6785    }
6786
6787    /// 500 — the default per-DID cap, so the largest import a stock instance
6788    /// sends — is three calls.
6789    #[tokio::test]
6790    async fn sidecar_bulk_add_of_500_is_three_calls() {
6791        let (base, log) = serve_apply_writes(None).await;
6792        chunk_sidecar(&base)
6793            .add_subscriptions_bulk(CHUNK_DID, &vetted_subs(500))
6794            .await
6795            .expect("bulk write");
6796        assert_eq!(call_sizes(&log), vec![200, 200, 100]);
6797    }
6798
6799    /// Exactly the limit is ONE call: chunking must not split a batch that fits.
6800    #[tokio::test]
6801    async fn sidecar_bulk_add_of_exactly_200_is_one_call() {
6802        let (base, log) = serve_apply_writes(None).await;
6803        chunk_sidecar(&base)
6804            .add_subscriptions_bulk(CHUNK_DID, &vetted_subs(200))
6805            .await
6806            .expect("bulk write");
6807        assert_eq!(call_sizes(&log), vec![200]);
6808    }
6809
6810    /// Nothing to write is no call at all — the behaviour both live clients
6811    /// already had (the sidecar refuses an empty `writes[]` with a 400).
6812    #[tokio::test]
6813    async fn sidecar_bulk_add_of_nothing_sends_nothing() {
6814        let (base, log) = serve_apply_writes(None).await;
6815        let rkeys = chunk_sidecar(&base)
6816            .add_subscriptions_bulk(CHUNK_DID, &[])
6817            .await
6818            .expect("an empty import is not an error");
6819        assert!(rkeys.is_empty());
6820        assert!(
6821            call_sizes(&log).is_empty(),
6822            "an empty batch must not be sent"
6823        );
6824    }
6825
6826    /// **A failed chunk stops the run.** Chunk 2 of 3 fails: the call returns
6827    /// an error, and chunk 3 is never sent — sending it would commit writes
6828    /// after a gap, which no caller could describe as a prefix.
6829    #[tokio::test]
6830    async fn sidecar_bulk_add_stops_at_the_first_failed_chunk() {
6831        let (base, log) = serve_apply_writes(Some(2)).await;
6832        let err = chunk_sidecar(&base)
6833            .add_subscriptions_bulk(CHUNK_DID, &vetted_subs(500))
6834            .await
6835            .expect_err("a failed chunk must fail the call");
6836        assert_eq!(call_sizes(&log), vec![200, 200], "chunk 3 must NOT be sent");
6837        assert!(
6838            format!("{err:#}").contains("boom"),
6839            "the PDS's reason was lost: {err:#}"
6840        );
6841        let progress = ApplyWritesIncomplete::of(&err).expect("the error says how far it got");
6842        assert_eq!(
6843            (progress.landed, progress.in_doubt, progress.total),
6844            (200, 200, 500),
6845            "chunk 1 landed, chunk 2 is in doubt, chunk 3 was never sent"
6846        );
6847        // A plain `%err` log line still names the PDS's reason.
6848        assert!(err.to_string().contains("boom"), "{err}");
6849    }
6850
6851    /// A batch that fit in one call fails exactly as it did before chunking:
6852    /// the same message, and nothing landed.
6853    #[tokio::test]
6854    async fn a_single_chunk_failure_reads_as_it_always_did() {
6855        let (base, _log) = serve_apply_writes(Some(1)).await;
6856        let err = chunk_sidecar(&base)
6857            .add_subscriptions_bulk(CHUNK_DID, &vetted_subs(3))
6858            .await
6859            .expect_err("refused");
6860        let progress = ApplyWritesIncomplete::of(&err).expect("progress");
6861        assert_eq!((progress.landed, progress.in_doubt), (0, 3));
6862        assert!(
6863            !err.to_string().contains("applyWrites call"),
6864            "a one-call batch has no progress to report: {err}"
6865        );
6866        assert_eq!(
6867            format!("{err:#}").matches("boom").count(),
6868            1,
6869            "the cause must not print twice in the chain: {err:#}"
6870        );
6871    }
6872
6873    fn create_ops(n: usize, value_bytes: usize) -> Vec<WriteOp> {
6874        (0..n)
6875            .map(|i| WriteOp::Create {
6876                collection: lexicon::nsid::SUBSCRIPTION.to_string(),
6877                rkey: Some(format!("rk{i:05}")),
6878                value: json!({ "pad": "x".repeat(value_bytes) }),
6879            })
6880            .collect()
6881    }
6882
6883    /// The op-count boundary, on small ops the byte bound never touches.
6884    #[test]
6885    // A one-range Vec IS the expected value here: one chunk spanning the batch.
6886    #[allow(clippy::single_range_in_vec_init)]
6887    fn chunks_split_at_200_ops_and_not_before() {
6888        for (n, want) in [
6889            (0, vec![]),
6890            (1, vec![0..1]),
6891            (200, vec![0..200]),
6892            (201, vec![0..200, 200..201]),
6893            (500, vec![0..200, 200..400, 400..500]),
6894        ] {
6895            assert_eq!(chunk_writes(&create_ops(n, 8)), want, "{n} ops");
6896        }
6897    }
6898
6899    /// **The byte bound splits under 200 ops, and every chunk fits it.**
6900    #[test]
6901    fn chunks_split_on_bytes_and_each_fits() {
6902        let ops = create_ops(40, 10_000);
6903        let chunks = chunk_writes(&ops);
6904        assert!(chunks.len() > 1, "400 KB went out as {chunks:?}");
6905        let mut next = 0;
6906        for range in &chunks {
6907            assert_eq!(range.start, next, "chunks must be consecutive: {chunks:?}");
6908            next = range.end;
6909            let body = json!({
6910                "repo": CHUNK_DID,
6911                "writes": ops[range.clone()].iter().map(WriteOp::to_json).collect::<Vec<_>>(),
6912            })
6913            .to_string();
6914            assert!(
6915                body.len() <= APPLY_WRITES_MAX_BYTES + 200,
6916                "a {}-byte body for {range:?}",
6917                body.len()
6918            );
6919        }
6920        assert_eq!(next, ops.len(), "every op, once");
6921    }
6922
6923    /// **The byte bound does not split an ordinary import.** 200 subscriptions
6924    /// with a title and a site URL each fit one call, so the common OPML
6925    /// import pays one round trip per 200 feeds and no more — the figure the
6926    /// bound's doc comment rests on.
6927    #[test]
6928    fn a_realistic_200_feed_import_is_one_call() {
6929        let ops: Vec<WriteOp> = (0..200)
6930            .map(|i| {
6931                let mut sub = lexicon::Subscription::new(
6932                    format!("https://www.example-blog-{i:03}.com/feeds/posts/default.xml"),
6933                    "2026-07-12T00:00:00.000Z",
6934                );
6935                sub.title = Some(format!("An Example Blog With A Fairly Long Title {i}"));
6936                sub.site_url = Some(format!("https://www.example-blog-{i:03}.com/"));
6937                WriteOp::Create {
6938                    collection: lexicon::nsid::SUBSCRIPTION.to_string(),
6939                    rkey: Some(format!("3lab2c4d5e{i:03}")),
6940                    value: serde_json::to_value(&sub).unwrap(),
6941                }
6942            })
6943            .collect();
6944        // ~76 KB measured.
6945        let bytes: usize = ops.iter().map(|op| op.to_json().to_string().len()).sum();
6946        assert_eq!(chunk_writes(&ops), vec![0..200], "{bytes} bytes");
6947    }
6948
6949    /// An op bigger than the bound cannot be split: it goes alone, and the
6950    /// ops around it are not dragged into its call.
6951    #[test]
6952    fn an_oversized_op_goes_alone() {
6953        let mut ops = create_ops(3, 8);
6954        ops.insert(1, create_ops(1, APPLY_WRITES_MAX_BYTES + 1).remove(0));
6955        assert_eq!(chunk_writes(&ops), vec![0..1, 1..2, 2..4]);
6956    }
6957
6958    /// Read-state cursors as large as the lexicon allows: 1,000 ids each.
6959    fn big_cursors(n: usize) -> Vec<(String, ReadState, bool)> {
6960        (0..n)
6961            .map(|i| {
6962                let mut state = ReadState::new(
6963                    format!("https://f{i}.example/feed.xml"),
6964                    None,
6965                    "2026-07-12T00:00:00.000Z",
6966                );
6967                state.read_ids = (0..ReadState::MAX_IDS)
6968                    .map(|j| format!("https://f{i}.example/posts/{j:04}/an-entry-permalink"))
6969                    .collect();
6970                (format!("rk{i:04}"), state, i % 2 == 0)
6971            })
6972            .collect()
6973    }
6974
6975    /// **The byte bound splits a batch well under 200 ops.** Ten full cursors
6976    /// are ~500 KB: under the op cap, over every older reference PDS's 150 KiB
6977    /// body limit. Each call must fit, and together they must carry every
6978    /// cursor, once, in order.
6979    #[tokio::test]
6980    async fn sidecar_read_state_flush_splits_on_bytes_under_200_ops() {
6981        let (base, log) = serve_apply_writes(None).await;
6982        let cursors = big_cursors(10);
6983        chunk_sidecar(&base)
6984            .flush_read_states(CHUNK_DID, &cursors)
6985            .await
6986            .expect("a byte-heavy flush must succeed in chunks");
6987
6988        let sizes = call_sizes(&log);
6989        assert!(
6990            sizes.len() > 1,
6991            "a ~500 KB flush went out as one call: {sizes:?}"
6992        );
6993        let want: Vec<String> = cursors.iter().map(|(rkey, _, _)| rkey.clone()).collect();
6994        assert_eq!(sent_rkeys(&log), want, "every cursor, once, in order");
6995    }
6996
6997    /// The direct (app-password) client gets the same chunking: its
6998    /// `flush_read_states` is the same op list on a different wire.
6999    #[tokio::test]
7000    async fn direct_client_read_state_flush_of_201_is_two_calls() {
7001        let (base, log) = serve_apply_writes(None).await;
7002        let port: u16 = base.rsplit(':').next().unwrap().parse().unwrap();
7003        let host = format!("chunk-direct-{port}.test");
7004        crate::net::test_host_override(&host, std::net::SocketAddr::from(([127, 0, 0, 1], port)));
7005        let client = PdsClient::new(
7006            ssrf_test_client(),
7007            format!("http://{host}:{port}"),
7008            CHUNK_DID,
7009            Auth::Session(SessionAuth {
7010                did: CHUNK_DID.to_string(),
7011                handle: None,
7012                access_jwt: "jwt".to_string(),
7013                refresh_jwt: None,
7014            }),
7015        );
7016        let cursors: Vec<(String, ReadState, bool)> = (0..201)
7017            .map(|i| {
7018                let feed = format!("https://f{i}.example/feed.xml");
7019                let state = ReadState::new(feed, None, "2026-07-12T00:00:00.000Z");
7020                (format!("rk{i:04}"), state, true)
7021            })
7022            .collect();
7023        client.flush_read_states(&cursors).await.expect("flush");
7024        assert_eq!(call_sizes(&log), vec![200, 1]);
7025        let want: Vec<String> = cursors.iter().map(|(rkey, _, _)| rkey.clone()).collect();
7026        assert_eq!(sent_rkeys(&log), want);
7027    }
7028
7029    // -- #149: compare-and-swap putRecord ------------------------------------
7030
7031    pub(crate) const SWAP_DID: &str = "did:plc:ewvi7nxzyoun6zhxrhs64oiz";
7032    pub(crate) const OLD_CID: &str = "bafyreigh2akiscaildcqabsyg3dfr6chu3fgpregiymsck7e7aqa4s52zy";
7033
7034    /// A server answering every request with `status` and `body`, logging each
7035    /// request's JSON body (`Null` for a GET). Returns its loopback base URL,
7036    /// a hostname routed to it (the guarded clients refuse loopback), and the
7037    /// log.
7038    pub(crate) async fn serve_status_json(
7039        status: u16,
7040        body: Value,
7041    ) -> (String, String, Arc<std::sync::Mutex<Vec<Value>>>) {
7042        use axum::response::IntoResponse as _;
7043        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
7044        let addr = listener.local_addr().unwrap();
7045        let host = format!("swap-{}.atproto.test", addr.port());
7046        crate::net::test_host_override(&host, addr);
7047        let log: Arc<std::sync::Mutex<Vec<Value>>> = Arc::default();
7048        let sink = Arc::clone(&log);
7049        let app = axum::Router::new().fallback(move |raw: axum::body::Bytes| {
7050            let sink = Arc::clone(&sink);
7051            let body = body.clone();
7052            async move {
7053                sink.lock()
7054                    .unwrap()
7055                    .push(serde_json::from_slice(&raw).unwrap_or(Value::Null));
7056                (
7057                    axum::http::StatusCode::from_u16(status).unwrap(),
7058                    axum::Json(body),
7059                )
7060                    .into_response()
7061            }
7062        });
7063        tokio::spawn(async move { axum::serve(listener, app).await.unwrap() });
7064        (
7065            format!("http://{addr}"),
7066            format!("http://{host}:{}", addr.port()),
7067            log,
7068        )
7069    }
7070
7071    fn swap_direct_client(pds: &str) -> PdsClient {
7072        PdsClient::new(
7073            ssrf_test_client(),
7074            pds,
7075            SWAP_DID,
7076            Auth::Session(SessionAuth {
7077                did: SWAP_DID.to_string(),
7078                handle: None,
7079                access_jwt: "jwt".to_string(),
7080                refresh_jwt: None,
7081            }),
7082        )
7083    }
7084
7085    pub(crate) fn swap_sub() -> crate::vetted::VettedSubscription {
7086        crate::vetted::VettedSubscription::new(&Subscription::new(
7087            "https://example.com/feed.xml",
7088            "2024-03-01T00:00:00.000Z",
7089        ))
7090    }
7091
7092    pub(crate) fn write_ok() -> Value {
7093        json!({
7094            "uri": format!("at://{SWAP_DID}/{}/rk", lexicon::nsid::SUBSCRIPTION),
7095            "cid": "bafyreiafter",
7096        })
7097    }
7098
7099    /// **The direct client puts `swapRecord` on the wire when given one, and
7100    /// leaves the key out entirely when not.** Asserted on the request the PDS
7101    /// received: a value accepted by the method and dropped on the way out is
7102    /// the failure mode, and only the bytes can show it.
7103    #[tokio::test]
7104    async fn direct_put_record_sends_swap_record_only_when_given() {
7105        let (_, pds, log) = serve_status_json(200, write_ok()).await;
7106        let client = swap_direct_client(&pds);
7107
7108        client
7109            .put_record(
7110                lexicon::nsid::SUBSCRIPTION,
7111                "rk",
7112                &swap_sub(),
7113                Some(OLD_CID),
7114            )
7115            .await
7116            .expect("put with a swap");
7117        client
7118            .put_record(lexicon::nsid::SUBSCRIPTION, "rk", &swap_sub(), None)
7119            .await
7120            .expect("put without a swap");
7121
7122        let sent = log.lock().unwrap().clone();
7123        assert_eq!(sent.len(), 2, "{sent:?}");
7124        assert_eq!(sent[0]["rkey"], "rk", "captured no usable body: {sent:?}");
7125        assert_eq!(
7126            sent[0]["swapRecord"], OLD_CID,
7127            "the CID the caller read never reached the PDS: {}",
7128            sent[0]
7129        );
7130        assert_eq!(sent[1]["rkey"], "rk");
7131        assert!(
7132            sent[1].get("swapRecord").is_none(),
7133            "no swap was asked for, so none may be sent: {}",
7134            sent[1]
7135        );
7136    }
7137
7138    /// The same, for the sidecar client — whose body is the sidecar's own
7139    /// `/internal/repo` shape, not XRPC's.
7140    #[tokio::test]
7141    async fn sidecar_put_sends_swap_record_only_when_given() {
7142        let (base, _, log) =
7143            serve_status_json(200, json!({ "ok": true, "data": write_ok() })).await;
7144        let client = SidecarClient::new(Client::new(), &base, &base, "secret");
7145
7146        client
7147            .update_subscription(SWAP_DID, "rk", &swap_sub(), Some(OLD_CID))
7148            .await
7149            .expect("put with a swap");
7150        client
7151            .update_subscription(SWAP_DID, "rk", &swap_sub(), None)
7152            .await
7153            .expect("put without a swap");
7154
7155        let sent = log.lock().unwrap().clone();
7156        assert_eq!(sent.len(), 2, "{sent:?}");
7157        assert_eq!(
7158            sent[0]["action"], "put",
7159            "captured no usable body: {sent:?}"
7160        );
7161        assert_eq!(sent[0]["swapRecord"], OLD_CID, "{}", sent[0]);
7162        assert_eq!(sent[1]["action"], "put");
7163        assert!(sent[1].get("swapRecord").is_none(), "{}", sent[1]);
7164    }
7165
7166    /// What the reference PDS answers a stale `swapRecord` with.
7167    pub(crate) fn invalid_swap_xrpc() -> Value {
7168        json!({ "error": "InvalidSwap", "message": format!("Record was at {OLD_CID}") })
7169    }
7170
7171    /// The same refusal, after the sidecar has wrapped it.
7172    pub(crate) fn invalid_swap_sidecar() -> Value {
7173        json!({
7174            "ok": false,
7175            "error": "InvalidSwap",
7176            "message": format!("Record was at {OLD_CID}"),
7177            "status": 400,
7178        })
7179    }
7180
7181    /// **A refused swap is recognised from each client's real error.** Driven
7182    /// through the clients against a server answering what the PDS answers,
7183    /// not built by hand, so a client that changes how it wraps a rejection
7184    /// breaks this rather than the rename that depends on it.
7185    #[tokio::test]
7186    async fn an_invalid_swap_is_recognised_from_both_clients_errors() {
7187        let (_, pds, _) = serve_status_json(400, invalid_swap_xrpc()).await;
7188        let err = swap_direct_client(&pds)
7189            .put_record(
7190                lexicon::nsid::SUBSCRIPTION,
7191                "rk",
7192                &swap_sub(),
7193                Some(OLD_CID),
7194            )
7195            .await
7196            .expect_err("the PDS refused the swap");
7197        assert!(is_invalid_swap(&err), "direct client: {err:#}");
7198
7199        let (base, _, _) = serve_status_json(400, invalid_swap_sidecar()).await;
7200        let err = SidecarClient::new(Client::new(), &base, &base, "secret")
7201            .update_subscription(SWAP_DID, "rk", &swap_sub(), Some(OLD_CID))
7202            .await
7203            .expect_err("the PDS refused the swap");
7204        assert!(is_invalid_swap(&err), "sidecar client: {err:#}");
7205    }
7206
7207    /// **Every other failure is NOT a lost race.** Reading one of these as
7208    /// `InvalidSwap` would re-read and retry a write the PDS refused for a
7209    /// reason a retry cannot fix — or tell the reader someone else edited a
7210    /// record nobody touched.
7211    #[tokio::test]
7212    async fn other_failures_are_not_an_invalid_swap() {
7213        for (status, body) in [
7214            (
7215                400,
7216                json!({ "error": "InvalidRequest", "message": "bad record" }),
7217            ),
7218            (400, json!({ "error": "RecordNotFound" })),
7219            (500, json!({ "error": "InternalServerError" })),
7220            (401, json!({ "error": "AuthRequired" })),
7221            // The name in the MESSAGE, not the error field, is not the signal.
7222            (
7223                400,
7224                json!({ "error": "InvalidRequest", "message": "InvalidSwap" }),
7225            ),
7226        ] {
7227            let (_, pds, _) = serve_status_json(status, body.clone()).await;
7228            let err = swap_direct_client(&pds)
7229                .put_record(
7230                    lexicon::nsid::SUBSCRIPTION,
7231                    "rk",
7232                    &swap_sub(),
7233                    Some(OLD_CID),
7234                )
7235                .await
7236                .expect_err("refused");
7237            assert!(!is_invalid_swap(&err), "{status} {body}: {err:#}");
7238        }
7239
7240        // A transport failure: nothing listening.
7241        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
7242        let dead = format!("http://{}", listener.local_addr().unwrap());
7243        drop(listener);
7244        let err = SidecarClient::new(Client::new(), &dead, &dead, "secret")
7245            .update_subscription(SWAP_DID, "rk", &swap_sub(), Some(OLD_CID))
7246            .await
7247            .expect_err("nothing is listening");
7248        assert!(!is_invalid_swap(&err), "transport: {err:#}");
7249
7250        // A string that merely SAYS it is not a typed refusal.
7251        assert!(!is_invalid_swap(&anyhow::anyhow!("InvalidSwap")));
7252    }
7253
7254    /// The typed refusal is found wherever it sits in the chain: under a
7255    /// context, and under [`ApplyWritesIncomplete`], whose `source()` skips its
7256    /// cause's top error — the trap `readstate::may_be_existence_mismatch` fell
7257    /// into on the merge with #240.
7258    #[tokio::test]
7259    async fn an_invalid_swap_is_found_under_context_and_a_split_batch() {
7260        let refusal = || AtProtoError::Xrpc {
7261            status: StatusCode::BAD_REQUEST,
7262            error: "InvalidSwap".to_string(),
7263            message: None,
7264        };
7265        assert!(is_invalid_swap(&anyhow::Error::new(refusal())));
7266        assert!(is_invalid_swap(
7267            &anyhow::Error::new(refusal()).context("com.atproto.repo.putRecord failed")
7268        ));
7269
7270        let writes: Vec<WriteOp> = (0..3)
7271            .map(|i| WriteOp::Delete {
7272                collection: lexicon::nsid::SUBSCRIPTION.to_string(),
7273                rkey: format!("rk{i}"),
7274            })
7275            .collect();
7276        let err = apply_writes_chunked(&writes, |_| async { Err(refusal().into()) })
7277            .await
7278            .expect_err("the only call failed");
7279        assert!(ApplyWritesIncomplete::of(&err).is_some(), "{err:#}");
7280        assert!(is_invalid_swap(&err), "{err:#}");
7281    }
7282
7283    /// Two subscription records with distinct CIDs, as `listRecords` returns.
7284    pub(crate) fn two_subs_page() -> Value {
7285        let rec = |rkey: &str, cid: &str, url: &str| {
7286            json!({
7287                "uri": format!("at://{SWAP_DID}/{}/{rkey}", lexicon::nsid::SUBSCRIPTION),
7288                "cid": cid,
7289                "value": {
7290                    "$type": lexicon::nsid::SUBSCRIPTION,
7291                    "url": url,
7292                    "createdAt": "2024-03-01T00:00:00.000Z",
7293                },
7294            })
7295        };
7296        json!({ "records": [
7297            rec("rk-a", "bafyreiaaaaaaaaaa", "https://a.example/feed.xml"),
7298            rec("rk-b", "bafyreibbbbbbbbbb", "https://b.example/feed.xml"),
7299        ] })
7300    }
7301
7302    /// Asserts a CID listing paired each record with ITS CID.
7303    pub(crate) fn assert_listed_with_cids(listed: &[(String, Option<String>, Subscription)]) {
7304        let got: Vec<(&str, Option<&str>, &str)> = listed
7305            .iter()
7306            .map(|(rkey, cid, sub)| (rkey.as_str(), cid.as_deref(), sub.url.as_str()))
7307            .collect();
7308        assert_eq!(
7309            got,
7310            vec![
7311                (
7312                    "rk-a",
7313                    Some("bafyreiaaaaaaaaaa"),
7314                    "https://a.example/feed.xml"
7315                ),
7316                (
7317                    "rk-b",
7318                    Some("bafyreibbbbbbbbbb"),
7319                    "https://b.example/feed.xml"
7320                ),
7321            ],
7322            "each record must come back with the CID it was listed at"
7323        );
7324    }
7325
7326    /// Two folder records at distinct CIDs, each carrying a field this build
7327    /// does not know (#268).
7328    pub(crate) fn two_folders_page() -> Value {
7329        let rec = |rkey: &str, cid: &str, name: &str| {
7330            json!({
7331                "uri": format!("at://{SWAP_DID}/{}/{rkey}", lexicon::nsid::FOLDER),
7332                "cid": cid,
7333                "value": {
7334                    "$type": lexicon::nsid::FOLDER,
7335                    "name": name,
7336                    "position": 3,
7337                    "createdAt": "2024-01-01T00:00:00.000Z",
7338                    "color": "#abc",
7339                },
7340            })
7341        };
7342        json!({ "records": [
7343            rec("fk-a", "bafyreifolderaaaa", "Tech"),
7344            rec("fk-b", "bafyreifolderbbbb", "News"),
7345        ] })
7346    }
7347
7348    /// Asserts a folder CID listing paired each record with ITS CID, and kept
7349    /// the record whole.
7350    pub(crate) fn assert_folders_listed_with_cids(listed: &[(String, Option<String>, Folder)]) {
7351        let got: Vec<(&str, Option<&str>, &str)> = listed
7352            .iter()
7353            .map(|(rkey, cid, f)| (rkey.as_str(), cid.as_deref(), f.name.as_str()))
7354            .collect();
7355        assert_eq!(
7356            got,
7357            vec![
7358                ("fk-a", Some("bafyreifolderaaaa"), "Tech"),
7359                ("fk-b", Some("bafyreifolderbbbb"), "News"),
7360            ],
7361            "each folder must come back with the CID it was listed at"
7362        );
7363        for (_, _, folder) in listed {
7364            assert_eq!(folder.position, Some(3));
7365            assert_eq!(folder.created_at, "2024-01-01T00:00:00.000Z");
7366            assert_eq!(folder.extra.get("color"), Some(&json!("#abc")));
7367        }
7368    }
7369
7370    #[tokio::test]
7371    async fn both_clients_list_folders_with_the_cid_each_was_read_at() {
7372        let (_, pds, _) = serve_status_json(200, two_folders_page()).await;
7373        let listed = swap_direct_client(&pds)
7374            .list_folders_with_cids()
7375            .await
7376            .expect("direct listing");
7377        assert_folders_listed_with_cids(&listed);
7378
7379        let (base, _, _) =
7380            serve_status_json(200, json!({ "ok": true, "data": two_folders_page() })).await;
7381        let listed = SidecarClient::new(Client::new(), &base, &base, "secret")
7382            .list_folders_with_cids(SWAP_DID)
7383            .await
7384            .expect("sidecar listing");
7385        assert_folders_listed_with_cids(&listed);
7386    }
7387
7388    #[tokio::test]
7389    async fn both_clients_list_subscriptions_with_the_cid_each_was_read_at() {
7390        let (_, pds, _) = serve_status_json(200, two_subs_page()).await;
7391        let listed = swap_direct_client(&pds)
7392            .list_subscriptions_with_cids()
7393            .await
7394            .expect("direct listing");
7395        assert_listed_with_cids(&listed);
7396
7397        let (base, _, _) =
7398            serve_status_json(200, json!({ "ok": true, "data": two_subs_page() })).await;
7399        let listed = SidecarClient::new(Client::new(), &base, &base, "secret")
7400            .list_subscriptions_with_cids(SWAP_DID)
7401            .await
7402            .expect("sidecar listing");
7403        assert_listed_with_cids(&listed);
7404    }
7405}