feather_reader/atproto.rs
1//! The atproto identity + PDS record layer.
2//!
3//! FeatherReader's defining bet is that a user's feed
4//! subscriptions, folders, saved items, and batched read-state live as records
5//! in the user's **own** atproto PDS under the open `community.lexicon.rss.*`
6//! community lexicon — not in the app's database. This module is the client that
7//! reads and writes those records.
8//!
9//! It has three layers:
10//!
11//! 1. **Identity resolution** ([`resolve_handle`], [`resolve_did_to_pds`]) —
12//! turn a handle (`alice.example.com`) into a DID (`did:plc:…`), then resolve
13//! the DID document to the PDS service endpoint. Handles resolve via the
14//! account's PDS `com.atproto.identity.resolveHandle` (or the well-known
15//! `/.well-known/atproto-did`); DIDs resolve via the PLC directory
16//! (`did:plc:*`) or the `did:web` well-known document.
17//! 2. **A lightweight [`PdsClient`]** — holds the resolved DID, the PDS base URL,
18//! and an [`Auth`] token, and exposes typed calls over `com.atproto.repo.*`:
19//! [`list_records`](PdsClient::list_records),
20//! `create_record`,
21//! `put_record`,
22//! [`delete_record`](PdsClient::delete_record), and
23//! `apply_writes` (the **batch** call the
24//! read-state flusher uses to coalesce many per-feed cursor writes into one
25//! round-trip).
26//! 3. **Typed convenience wrappers** wired to the [`crate::lexicon`] record
27//! types (list/create [`Subscription`]/[`Folder`]/[`Saved`], put
28//! [`ReadState`], batch-flush many `ReadState` cursors).
29//!
30//! ## Auth — the OAuth sidecar is the live path
31//!
32//! Auth is a **trait/enum boundary** so the mechanism can vary without touching
33//! call sites. There are three paths:
34//!
35//! * **The live path — the atproto OAuth confidential client, via [`SidecarClient`].**
36//! atproto OAuth (DPoP, PAR, token refresh) is fiddly and is **not** hand-rolled
37//! in Rust: it runs in a small, supported `@atproto/oauth-client-node` sidecar.
38//! The Rust server never holds
39//! PDS tokens — it POSTs every `com.atproto.repo.*` op to the sidecar's
40//! `/internal/repo` endpoint (gated by a shared `X-Internal-Secret`), and the
41//! sidecar restores the DID's OAuth session (transparent DPoP + token refresh)
42//! and runs the matching XRPC call. [`SidecarClient`] is that client; the typed
43//! convenience wrappers (list/create/put/delete subscriptions, batch-flush
44//! read-state) live on it and map 1:1 to the old [`PdsClient`] surface.
45//! * **The interim path — [`Auth::Session`] (app password).** A session obtained
46//! from `com.atproto.server.createSession`. Kept behind the [`Auth`] seam, but
47//! it is **no longer the live path**: [`PdsClient`] and
48//! [`login_with_app_password`] remain for tests, while [`SidecarClient`] is
49//! what the web layer routes through.
50//!
51//! ⚠️ **This is no longer a working "local runs without the sidecar" fallback,
52//! and the docs used to claim otherwise.** Since v0.2.8 every [`PdsClient`]
53//! request goes through the SSRF guard, which refuses loopback, RFC1918, ULA
54//! and `100.64/10` (Tailscale). So pointing this at `http://localhost:2583`
55//! or a tailnet PDS now fails with *"refusing to fetch forbidden (internal)
56//! address"* rather than returning a session. That is the guard behaving
57//! correctly — the target host is attacker-influenced in the cases that
58//! matter, and a dev-only escape hatch is exactly the kind of flag that ends
59//! up set in production — but it does mean a local-PDS workflow needs the
60//! PDS reachable on a public address, or a deliberate change here.
61//! * **The public-read path — [`Auth::Anonymous`], via [`PdsClient::anonymous`].**
62//! `com.atproto.repo.listRecords` is public on a standard PDS, so a stranger's
63//! `community.lexicon.rss.*` records can be read with no credentials at all.
64//! An anonymous client sends no `Authorization` header and is **read-only** —
65//! every write fails closed on [`Auth::bearer`]. Because the target host is
66//! then chosen by a stranger, the read is routed through
67//! [`crate::net::guarded_get_no_privacy`] (per-hop SSRF re-validation +
68//! connect-pinning) and capped by [`crate::net::read_capped`].
69//!
70//! ## Every PDS request goes through the SSRF guard
71//!
72//! A PDS host is *never* a host FeatherReader chose: it comes out of a DID
73//! document, which is attacker-controllable. So identity resolution, the record
74//! **reads**, and the record **writes** all route through [`crate::net`] —
75//! [`crate::net::guarded_get_no_privacy`] and
76//! [`crate::net::guarded_post_json`] — rather than the shared
77//! `reqwest::Client`. [`resolve_did_to_pds`] runs
78//! [`crate::net::assert_public_target`] on the `serviceEndpoint` it returns, but
79//! that check is a *separate DNS resolution* from the later request; only
80//! re-vetting and connect-pinning at request time closes the rebinding window.
81//! The writes matter most: they carry the session bearer, and
82//! [`login_with_app_password`] carries the app password in the request **body**,
83//! where reqwest's cross-origin header sanitisation offers no protection at all
84//! — which is why the guarded POST refuses redirects outright.
85//!
86//! All network I/O is `reqwest` (rustls, no OpenSSL); every fallible path returns
87//! [`anyhow::Result`] or the typed [`AtProtoError`] — nothing panics.
88
89use std::sync::Arc;
90
91use anyhow::{Context, Result};
92use reqwest::header::{HeaderName, HeaderValue, AUTHORIZATION};
93use reqwest::{Client, StatusCode};
94use serde::de::DeserializeOwned;
95use serde::{Deserialize, Serialize};
96use serde_json::{json, Value};
97
98use crate::lexicon::{self, Folder, ReadState, Saved, Subscription};
99
100/// The public PLC directory, used to resolve `did:plc:*` DIDs to their DID
101/// document (and thus their PDS service endpoint).
102pub const DEFAULT_PLC_DIRECTORY: &str = "https://plc.directory";
103
104/// The default appview/entryway used only as a bootstrap host for handle
105/// resolution when the caller has no PDS hint yet. Handle resolution ultimately
106/// works against any atproto host that implements
107/// `com.atproto.identity.resolveHandle`; `bsky.social` is a reliable default.
108pub const DEFAULT_RESOLVER_HOST: &str = "https://bsky.social";
109
110/// Hard cap on cursor pages any `list_all_records` walk will follow.
111///
112/// [`crate::net::read_capped`] bounds each individual response, but nothing
113/// bounded the *accumulation* across pages: a repo host that returns a full page
114/// and a fresh cursor forever walks memory until the (512 MB) box dies. At 100
115/// records per page this admits 20 000 records — far past any real
116/// `community.lexicon.rss.*` collection — while making the loop finite against a
117/// host we do not control. Mirrors [`crate::network::MAX_PAGES`], which bounds
118/// the relay walk for the same reason.
119const MAX_LIST_PAGES: usize = 200;
120
121/// Hard cap on the records a single `list_all_records` walk will accumulate.
122///
123/// [`MAX_LIST_PAGES`] bounds how many REQUESTS a walk makes. It bounds the
124/// accumulated memory only if the server honours `limit=100` — and a repo host
125/// we did not choose has no obligation to. Measured: an 8 MB page (the
126/// [`crate::net::read_capped`] ceiling) holds ~95 000 minimal records and
127/// retains ~23 MB as `Vec<RecordEntry>`, so the page cap alone admits gigabytes
128/// on a 512 MB box.
129///
130/// 20 000 is the number [`MAX_LIST_PAGES`]'s own comment already claimed — this
131/// makes the claim true rather than conditional on the server's cooperation.
132const MAX_LIST_RECORDS: usize = 20_000;
133
134/// The same cap for a collection whose records are **large**.
135///
136/// [`MAX_LIST_RECORDS`]'s figure was measured against *minimal* records
137/// (~1 KB). A `site.standard.document` carries the whole article — ~17 KB
138/// measured across 449 real ones — so 20 000 of them is ~340 MB retained on a
139/// 512 MB box. Sized to the record, not to the protocol.
140pub(crate) const MAX_LARGE_RECORDS: usize = 2_000;
141
142/// The at-URI scheme prefix, **the one Rust spelling**. Every Rust guard that
143/// asks "is this an at-URI" strips or compares this.
144///
145/// **SQL no longer holds a second opinion.** There used to be a matching string
146/// predicate in `store`, and this comment claimed a test pinned the two in
147/// agreement. Both are gone: the predicate was deleted when `feeds.kind` became
148/// a cache of [`crate::feed::FeedKind::of`], re-derived from the URL rather than
149/// re-described in SQL, and no such test survived it. Nothing outside Rust
150/// decides what an at-URI is, so there is nothing left to keep in step.
151pub(crate) const AT_URI_PREFIX: &str = "at://";
152
153/// Strip the at-URI scheme **case-insensitively**, returning the body.
154///
155/// Schemes are case-insensitive per RFC 3986 and `Url::parse` folds them, so
156/// `At://` names the same thing as `at://`. Recognition has to match that, or a
157/// mixed-case row is an at-URI to the fetcher (which refuses it) and an
158/// ordinary URL to every guard — polled forever, failing forever. Whether such
159/// a spelling may be STORED is a separate question, answered no.
160pub(crate) fn strip_at_prefix(url: &str) -> Option<&str> {
161 url.get(..AT_URI_PREFIX.len())
162 .filter(|p| p.eq_ignore_ascii_case(AT_URI_PREFIX))
163 .map(|p| &url[p.len()..])
164}
165
166/// atproto's record-key rules, all of them: charset `[A-Za-z0-9._:~-]`, length
167/// 1..=512, and not `.` or `..`. The repo's TID tests state the same rule; this
168/// is the one place it is enforced on a key that arrives from outside.
169pub(crate) fn is_valid_rkey(rkey: &str) -> bool {
170 !rkey.is_empty()
171 && rkey.len() <= 512
172 && rkey != "."
173 && rkey != ".."
174 && rkey
175 .chars()
176 .all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '_' | ':' | '~' | '-'))
177}
178
179/// Accumulate a page for a **reading** walk, keeping what fits and reporting
180/// whether anything was dropped.
181///
182/// **A truncation, never an error** — the opposite of [`extend_bounded`], and
183/// deliberately so. That function's refusal exists because its caller feeds
184/// `replace_sub_refs`, where a short list is revoked access. A walk that only
185/// ADDS entries has no such hazard, and refusing there is strictly worse: a
186/// publication with more documents than the cap would fail on every poll, so
187/// an ordinary long-running blog becomes permanently unreadable instead of
188/// partially read. The records kept are the ones the PDS returned first.
189pub(crate) fn extend_truncating(
190 out: &mut Vec<RecordEntry>,
191 page: Vec<RecordEntry>,
192 max: usize,
193) -> bool {
194 let room = max.saturating_sub(out.len());
195 // **Strictly greater.** `>=` called an exactly-full final page a
196 // truncation, so a collection holding exactly `max` records warned that it
197 // had dropped something on every poll.
198 let dropped = page.len() > room;
199 out.extend(page.into_iter().take(room));
200 dropped
201}
202
203/// The result of a bounded walk: what was read, and whether that is all of it.
204///
205/// **`complete` is a fact the caller cannot recover afterwards.** A short list
206/// from a truncating walk looks exactly like a short collection, and the
207/// difference is the one that matters: "this publication has nine articles" and
208/// "this reader gave up after nine" are the same `Vec` and very different
209/// answers.
210#[derive(Debug)]
211pub struct RecordWalk {
212 /// The records kept, in the order the PDS returned them.
213 pub records: Vec<RecordEntry>,
214 /// True when the collection ran out before any bound did.
215 pub complete: bool,
216 /// Records skipped because their envelope was malformed (#177). Only a
217 /// walk that SKIPS can report this; the walks that feed `replace_sub_refs`
218 /// refuse instead, with [`MalformedRecords`].
219 pub malformed: usize,
220}
221
222impl RecordWalk {
223 fn complete(records: Vec<RecordEntry>) -> Self {
224 Self {
225 records,
226 complete: true,
227 malformed: 0,
228 }
229 }
230 fn partial(records: Vec<RecordEntry>) -> Self {
231 Self {
232 records,
233 complete: false,
234 malformed: 0,
235 }
236 }
237}
238
239/// The memory one walk may retain.
240///
241/// **A record cap bounds memory only if you know what a record costs.** The
242/// caps above are counts, chosen against a measured ~17 KB document, and
243/// `MAX_LIST_PAGES` bounds requests rather than bytes. A PDS whose records are
244/// not that shape satisfies every count and still exhausts the box.
245///
246/// **128 MiB of ACCUMULATION per read — which is not the same as 128 MiB of
247/// memory, and an earlier version of this comment said it was.**
248///
249/// Every charge here is taken after `serde_json` has already built the page, so
250/// the true peak is this ceiling plus one page's tree, and a page's tree is not
251/// small: measured, an 8 MiB response of `{"":0}` objects retains 824 MB, a
252/// wire-to-heap amplification of 98x. A bound consulted after the allocation
253/// cannot prevent that allocation. What it does prevent is the accumulation
254/// across pages and across the walks of one read, which is the part that scales
255/// with how long a walk runs rather than with one response.
256///
257/// Closing the single-page case needs a smaller wire cap for `listRecords` or a
258/// parser that counts as it goes. Neither belongs to this bound; both are filed
259/// as #197 rather than implied here.
260///
261/// A caller passes one [`ByteBudget`] into every walk it makes, so a publication
262/// read — which runs a second walk while still holding the first's records — is
263/// bounded once rather than twice. Two independent ceilings put roughly 384 MB of
264/// accumulation in flight: 128 for the publications, 128 for the documents, and a
265/// further 128 of transient page because the per-page check compared against the
266/// ceiling instead of what was left.
267///
268/// **Only one caller threads it today**: the publication reader, polled by the
269/// scheduler's publication loop since 0.4.0. Every other live read
270/// builds its own ceiling per walk, so the per-request total is still a multiple
271/// of this number — two walks on an OPML export, four on a login — and nothing
272/// bounds concurrent requests at all.
273/// An earlier version of this constant was also 128 MiB while
274/// [`approx_bytes`] charged serialized length — 42x optimistic on hostile
275/// shapes, so the bound was nearly fiction. It was then cut to 64 MiB to
276/// compensate. Charging nodes removed the reason for the cut: the charge is now
277/// at or above what the page really retains, so 128 MiB of budget is at most
278/// 128 MiB of memory, which a 512 MB box carries.
279///
280/// The cut had a cost, measured rather than assumed. Per walk:
281///
282/// - The subscription walks cap at 20 000 records (5 000 on the live one). A
283/// real five-field subscription charges **2 188 bytes** here, and one carrying
284/// a folder and a fetch hint charges **2 764** — so a full repo is 42 to 53 MB,
285/// which was 65 to 82 % of a 64 MiB budget. Their verdict is a hard refusal
286/// that drops the reader into the fail-closed branch, so an account near the
287/// record cap with slightly longer titles would have served a stale projection
288/// on every poll, permanently. At 128 MiB that is 41 % and the count still
289/// binds first. An earlier version of this comment claimed 1.5 KB and 30 MB;
290/// that is a three-field record, not a real one.
291/// - The publication walk caps at [`MAX_LARGE_RECORDS`] (2 000). At the measured
292/// ~17 KB document that is about 37 MB either way. Above roughly 66 KB per
293/// article the budget binds first and the walk truncates early, reporting
294/// `complete: false` as it already does for the record cap.
295///
296/// So the counts bind first on everything measured, and a publication of
297/// extremely long articles truncates sooner than the count would. It remains
298/// untrue that nothing truncates that did not truncate before.
299pub(crate) const MAX_LIST_BYTES: usize = 128 * 1024 * 1024;
300
301/// What one record retains once parsed.
302///
303/// **Nodes, not serialized text.** An earlier version of this charged the
304/// length of the JSON, which is the wrong quantity by up to 42x: a parsed value
305/// is a tree of 32-byte nodes held in vectors that over-allocate, so `[[],[]…]`
306/// costs three bytes on the wire and well over a hundred in memory. Measured
307/// against that estimate, a budget reporting 119 MiB held a process at 5.6 GiB.
308///
309/// Every arm therefore charges at least the node itself, and a container
310/// charges for the slack its backing allocation carries. The result
311/// over-estimates on every adversarial shape and costs honest traffic a couple
312/// of percent, which is the direction a bound has to err in.
313pub(crate) fn approx_bytes(entry: &RecordEntry) -> usize {
314 2 * std::mem::size_of::<RecordEntry>()
315 + entry.uri.len()
316 + entry.cid.as_ref().map_or(0, String::len)
317 + json_bytes(&entry.value)
318}
319
320/// What a parsed JSON value retains, without measuring the heap.
321fn json_bytes(v: &serde_json::Value) -> usize {
322 /// Every value, of every kind, occupies one of these wherever it sits.
323 const NODE: usize = std::mem::size_of::<serde_json::Value>();
324 /// Two nodes per value: the slot it occupies, and the slack the container
325 /// holding it carries — a `Vec` grows by doubling, so up to one spare slot
326 /// per live one.
327 const SLOT: usize = 2 * NODE;
328 /// A map entry is a tree node of its own, with links and a key beside the
329 /// value. Rounded up rather than derived, since the layout is not ours.
330 const MAP_ENTRY: usize = 104;
331 /// A map's backing node, allocated whole.
332 ///
333 /// **Empirical, and not derived from anything the compiler checks.** Unlike
334 /// [`NODE`], which is a `size_of`, this and `MAP_ENTRY` come from measuring
335 /// `std`'s `BTreeMap` layout — B = 6, so eleven pairs to a leaf — under the
336 /// `serde_json` in this lockfile. A toolchain that changes that layout, or a
337 /// `serde_json` that swaps the map type, moves the real cost without moving
338 /// these. The known-answer test below is the tripwire, and it is only as
339 /// good as the day its figures were taken.
340 ///
341 /// `serde_json::Map` is a `BTreeMap` here — no `preserve_order` in the
342 /// lock — and its leaf carries room for eleven pairs whether or not they
343 /// are used, measured at ~632 bytes. So a one-key object costs what an
344 /// eleven-key one does, and a chain of them costs that per level. Charging
345 /// a container's minimum the way an array does under-reports this by about
346 /// half, which is the same failure as the version this replaces, two orders
347 /// of magnitude smaller.
348 const MAP_NODE: usize = 512;
349 match v {
350 // The `4 * NODE` is the container's own minimum allocation; each child
351 // then charges for itself, recursively. Dropping that recursion is what
352 // made an array of empty arrays look free.
353 serde_json::Value::Array(a) => 4 * NODE + a.iter().map(json_bytes).sum::<usize>(),
354 serde_json::Value::Object(o) => {
355 MAP_NODE
356 + o.iter()
357 .map(|(k, v)| MAP_ENTRY + k.len().max(NODE / 2) + SLOT + json_bytes(v))
358 .sum::<usize>()
359 }
360 serde_json::Value::String(s) => SLOT + s.len(),
361 // Null, bool and number are all the node and nothing else.
362 _ => SLOT,
363 }
364}
365
366/// Running byte accounting for one walk.
367pub(crate) struct ByteBudget {
368 used: usize,
369 max: usize,
370}
371
372impl ByteBudget {
373 pub(crate) fn new(max: usize) -> Self {
374 Self { used: 0, max }
375 }
376
377 /// Charge a page. `false` when the walk must stop; a refused page is NOT
378 /// charged, so `used` always describes what the caller actually kept.
379 pub(crate) fn admit(&mut self, page: &[RecordEntry]) -> bool {
380 let cost: usize = page.iter().map(approx_bytes).sum();
381 match self.used.checked_add(cost) {
382 Some(total) if total <= self.max => {
383 self.used = total;
384 true
385 }
386 _ => false,
387 }
388 }
389
390 /// Charge `bytes` that no record accounts for. Same contract as
391 /// [`Self::admit`]: `false` means stop, and a refused charge is not taken.
392 pub(crate) fn charge(&mut self, bytes: usize) -> bool {
393 match self.used.checked_add(bytes) {
394 Some(total) if total <= self.max => {
395 self.used = total;
396 true
397 }
398 _ => false,
399 }
400 }
401
402 pub(crate) fn used(&self) -> usize {
403 self.used
404 }
405
406 /// The ceiling this budget was built with.
407 pub(crate) fn max(&self) -> usize {
408 self.max
409 }
410
411 /// What is left. A transient page has to fit in this, not in the ceiling —
412 /// otherwise a walk that has already retained most of its budget can still
413 /// hold a full budget's worth of page on top of it.
414 pub(crate) fn remaining(&self) -> usize {
415 self.max.saturating_sub(self.used)
416 }
417}
418
419/// Append a page, refusing to exceed `max`.
420///
421/// **An error, never a truncation.** The caller of the live walk is
422/// `web::resolve_subscriptions`, whose result reaches `store::replace_sub_refs`
423/// — a `DELETE` followed by reinserting exactly what it was handed. A short
424/// list there is not a short list, it is revoked access to whatever fell off
425/// the end. Returning `Err` lets `resolve_subscriptions` take its documented
426/// fail-closed branch and serve the last-known projection instead.
427///
428/// `out` is left untouched on refusal, so a partial page cannot survive.
429pub(crate) fn extend_bounded(
430 out: &mut Vec<RecordEntry>,
431 page: Vec<RecordEntry>,
432 max: usize,
433 collection: &str,
434) -> Result<()> {
435 if out.len() + page.len() > max {
436 anyhow::bail!(
437 "listRecords for {collection} exceeded the {max}-record cap \
438 ({} held, {} more offered) — refusing to accumulate further",
439 out.len(),
440 page.len(),
441 );
442 }
443 out.extend(page);
444 Ok(())
445}
446
447/// Errors from the atproto identity + PDS layer.
448///
449/// Wraps the transport, the atproto XRPC error envelope (`{"error","message"}`),
450/// and the identity-resolution failure modes so callers can distinguish "the
451/// network broke" from "the PDS said no" from "this handle doesn't resolve".
452#[derive(Debug, thiserror::Error)]
453pub enum AtProtoError {
454 /// The underlying HTTP transport failed (DNS, TLS, timeout, connect).
455 #[error("atproto transport error: {0}")]
456 Transport(#[from] reqwest::Error),
457
458 /// The XRPC endpoint returned a non-2xx status with an atproto error
459 /// envelope (or an opaque body). `error` is the atproto error name (e.g.
460 /// `RecordNotFound`, `AuthMissing`), `message` the human string.
461 #[error("atproto XRPC error {status}: {error}{}", .message.as_deref().map(|m| format!(" — {m}")).unwrap_or_default())]
462 Xrpc {
463 /// The HTTP status code.
464 status: StatusCode,
465 /// The atproto error name (the `error` field), or `"Unknown"`.
466 error: String,
467 /// The optional human-readable `message` field.
468 message: Option<String>,
469 },
470
471 /// A handle could not be resolved to a DID.
472 #[error("could not resolve handle {handle:?} to a DID")]
473 HandleResolution {
474 /// The handle that failed to resolve.
475 handle: String,
476 },
477
478 /// A DID document could not be resolved, or lacks a usable PDS service
479 /// endpoint (`#atproto_pds`).
480 #[error("could not resolve DID {did:?} to a PDS endpoint: {reason}")]
481 DidResolution {
482 /// The DID that failed to resolve.
483 did: String,
484 /// Why resolution failed.
485 reason: String,
486 /// The same, as a value a caller can branch on without reading `reason`.
487 cause: DidResolutionCause,
488 },
489}
490
491/// Why a DID did not resolve to a PDS, structured — so a poller can file a
492/// deleted account under "the server answered" rather than "the network broke".
493#[derive(Debug, Clone, Copy, PartialEq, Eq)]
494#[non_exhaustive]
495pub enum DidResolutionCause {
496 /// A DID method this reader does not resolve.
497 UnsupportedMethod,
498 /// The DID document fetch got an answer, and it was not a success (a
499 /// tombstoned or unknown DID is a 404 from the PLC directory).
500 Status,
501 /// The DID document has no `#atproto_pds` service.
502 NoPdsEndpoint,
503 /// The PDS endpoint it names is refused by the SSRF guard.
504 NotAPublicTarget,
505}
506
507/// Whether a write failed because its `swapRecord` no longer matched — the
508/// record moved between the caller's read and its write (#149).
509///
510/// **Matched on the structured rejection, by error name.** Every client
511/// surfaces a PDS refusal as [`AtProtoError::Xrpc`]: the direct client at the
512/// root, the Rust OAuth client under a context, the sidecar client with the
513/// PDS's name carried through `/internal/repo`. `InvalidSwap` is the reference
514/// PDS's own name for a compare-and-swap mismatch and is answered 400; the
515/// name is the signal rather than the status, because the name is what says
516/// "someone else wrote this" and a 400 alone says nothing of the kind. A
517/// transport failure, or a message that merely contains the word, is not one.
518///
519/// The cause of an [`ApplyWritesIncomplete`] is walked explicitly, as
520/// `readstate::may_be_existence_mismatch` does: that wrapper's `source()`
521/// continues from its cause's SOURCE, so on the sidecar client — whose cause
522/// IS the `AtProtoError` — `err.chain()` alone would step over it.
523pub fn is_invalid_swap(err: &anyhow::Error) -> bool {
524 let wrapped = ApplyWritesIncomplete::of(err).map(|p| p.cause().chain());
525 err.chain()
526 .chain(wrapped.into_iter().flatten())
527 .any(|cause| {
528 matches!(
529 cause.downcast_ref::<AtProtoError>(),
530 Some(AtProtoError::Xrpc { error, .. }) if error == "InvalidSwap"
531 )
532 })
533}
534
535impl AtProtoError {
536 /// True when the XRPC error is a "record not found" — handy for upsert paths
537 /// that treat a missing record as "create instead of update".
538 pub fn is_record_not_found(&self) -> bool {
539 matches!(
540 self,
541 AtProtoError::Xrpc { error, .. } if error == "RecordNotFound"
542 )
543 }
544}
545
546// ---------------------------------------------------------------------------
547// Auth — the direct-PDS path (dev / tests)
548// ---------------------------------------------------------------------------
549
550/// A source of atproto access tokens.
551///
552/// This trait abstracts over token acquisition for the direct [`PdsClient`]
553/// (used by local runs and tests). A [`PdsClient`] can hold a `dyn TokenSource`
554/// instead of a static [`Auth`] without any call-site change, so a token source
555/// that refreshes out of band can be dropped in later.
556///
557/// It is async + `Send + Sync` so a background refresh can live behind it.
558#[allow(async_fn_in_trait)]
559pub trait TokenSource: Send + Sync {
560 /// Return the current bearer access token to send as `Authorization`.
561 async fn access_token(&self) -> Result<String>;
562}
563
564/// The auth material a [`PdsClient`] carries.
565///
566/// A small enum rather than a bare string, so the match stays exhaustive if a
567/// second direct-auth mechanism is added alongside app-password sessions.
568#[derive(Clone)]
569pub enum Auth {
570 /// A bearer access token from a `com.atproto.server.createSession`
571 /// (app-password) session. This is the direct-PDS auth used by local runs
572 /// and tests; the live web path authenticates via the OAuth sidecar instead
573 /// (see [`SidecarClient`]).
574 Session(SessionAuth),
575
576 /// The atproto OAuth confidential-client path is handled entirely by the
577 /// `@atproto/oauth-client` sidecar ([`SidecarClient`]), which mints, DPoP-binds,
578 /// and refreshes tokens. The direct [`PdsClient`] does not carry OAuth tokens;
579 /// this variant is a placeholder so the `Auth` enum documents that the OAuth
580 /// path lives elsewhere.
581 Oauth(OauthPlaceholder),
582
583 /// **No credentials at all** — an unauthenticated public read of a repo the
584 /// caller does not own. `com.atproto.repo.listRecords` is public on a
585 /// standard PDS, so a stranger's `community.lexicon.rss.*` records can be
586 /// read with no session; this variant makes that expressible without
587 /// inventing a fake token.
588 ///
589 /// A client holding it is **read-only**: [`Auth::bearer`] returns an error,
590 /// so every write path (`create_record` / `put_record` / `delete_record` /
591 /// `apply_writes`, all of which go through
592 /// `authed_headers`) fails closed. Construct one
593 /// via [`PdsClient::anonymous`].
594 Anonymous,
595}
596
597impl Auth {
598 /// The bearer access token to present on `com.atproto.repo.*` calls.
599 ///
600 /// Only [`Auth::Session`] carries a token (the session's `accessJwt`).
601 /// [`Auth::Oauth`] carries none — the sidecar owns the OAuth path — so it
602 /// returns an error pointing callers at [`SidecarClient`]. [`Auth::Anonymous`]
603 /// carries none by construction, which is what makes an anonymous client
604 /// read-only.
605 pub fn bearer(&self) -> Result<&str> {
606 match self {
607 Auth::Session(s) => Ok(&s.access_jwt),
608 Auth::Oauth(_) => anyhow::bail!(
609 "the direct PdsClient does not carry OAuth tokens — atproto OAuth is \
610 handled by the @atproto/oauth-client sidecar (SidecarClient); \
611 use Auth::Session (app-password) for the direct-PDS path"
612 ),
613 Auth::Anonymous => anyhow::bail!(
614 "this PdsClient is anonymous (unauthenticated public read) and carries no \
615 bearer token — authenticated repo writes require Auth::Session or the \
616 SidecarClient"
617 ),
618 }
619 }
620}
621
622/// A session obtained from `com.atproto.server.createSession` (interim
623/// app-password auth). Holds the DID + tokens + handle the server returned.
624#[derive(Clone, Debug, Deserialize)]
625pub struct SessionAuth {
626 /// The account DID this session authenticates.
627 pub did: String,
628 /// The account handle at session-creation time.
629 #[serde(default)]
630 pub handle: Option<String>,
631 /// The bearer access token presented on authed XRPC calls.
632 #[serde(rename = "accessJwt")]
633 pub access_jwt: String,
634 /// The refresh token, exchanged via `com.atproto.server.refreshSession`.
635 /// The direct-PDS refresh flow is not implemented here; the live web path
636 /// refreshes via the OAuth sidecar instead.
637 #[serde(rename = "refreshJwt", default)]
638 pub refresh_jwt: Option<String>,
639}
640
641/// Placeholder for the OAuth variant of [`Auth`].
642///
643/// Intentionally empty: the OAuth session material (DPoP key handle, token
644/// references) is held entirely by the sidecar, not by the direct [`PdsClient`].
645/// This type exists only so [`Auth::Oauth`] is a real variant and the split is
646/// visible in the type system.
647#[derive(Clone, Debug, Default)]
648#[non_exhaustive]
649pub struct OauthPlaceholder {}
650
651// ---------------------------------------------------------------------------
652// Identity resolution
653// ---------------------------------------------------------------------------
654
655/// Resolve an atproto handle to its DID.
656///
657/// Uses `com.atproto.identity.resolveHandle` against `resolver_base` (any host
658/// that implements it; [`DEFAULT_RESOLVER_HOST`] is a safe bootstrap). A fuller
659/// implementation would also try the DNS `_atproto` TXT record and the
660/// `https://<handle>/.well-known/atproto-did` fallback; the XRPC path is the
661/// common case and the one implemented here.
662pub async fn resolve_handle(client: &Client, resolver_base: &str, handle: &str) -> Result<String> {
663 // Build the query manually rather than via reqwest's `.query()` so we don't
664 // depend on the optional `query`/`url` reqwest feature (the declared feature
665 // set is rustls + gzip + json only).
666 let url = format!(
667 "{}/xrpc/com.atproto.identity.resolveHandle?handle={}",
668 resolver_base.trim_end_matches('/'),
669 urlencode(handle)
670 );
671
672 #[derive(Deserialize)]
673 struct ResolveHandleOut {
674 did: String,
675 }
676
677 // Route through the SSRF guard: `resolver_base` can be a user-influenced PDS
678 // host (from a prior DID-doc resolution), so a hostile endpoint must not be
679 // able to target loopback / link-local / metadata. Feed-privacy is NOT
680 // applied here (this is a legitimate atproto XRPC call, not a feed fetch).
681 let resp = crate::net::guarded_get_no_privacy(client, &url, &[]).await?;
682 if !resp.status().is_success() {
683 // Surface the XRPC envelope but map the common "not found" to the typed
684 // handle-resolution error so callers get a clean signal.
685 let err = xrpc_error_from(resp).await;
686 if let AtProtoError::Xrpc { status, .. } = &err {
687 if *status == StatusCode::BAD_REQUEST || *status == StatusCode::NOT_FOUND {
688 return Err(AtProtoError::HandleResolution {
689 handle: handle.to_string(),
690 }
691 .into());
692 }
693 }
694 return Err(err.into());
695 }
696
697 // Capped: `resolver_base` can be a user-influenced PDS host, as the comment
698 // above this function's guard already says.
699 let raw = crate::net::read_capped(resp).await?;
700 let out: ResolveHandleOut =
701 serde_json::from_slice(&raw).context("parsing resolveHandle response")?;
702 Ok(out.did)
703}
704
705/// Resolve a DID to its PDS service endpoint by fetching + parsing its DID
706/// document.
707///
708/// * `did:plc:*` → the PLC directory (`{plc_directory}/{did}`).
709/// * `did:web:host` → `https://host/.well-known/did.json`.
710///
711/// The PDS endpoint is the service in the DID doc whose `id` ends with
712/// `#atproto_pds` (type `AtprotoPersonalDataServer`); its `serviceEndpoint` is
713/// the base URL for all `com.atproto.repo.*` calls.
714pub async fn resolve_did_to_pds(client: &Client, plc_directory: &str, did: &str) -> Result<String> {
715 let doc_url = if let Some(rest) = did.strip_prefix("did:web:") {
716 // did:web host may itself be percent-encoded / contain a path; the
717 // common case is a bare host.
718 let host = rest.replace(':', "/");
719 format!("https://{host}/.well-known/did.json")
720 } else if did.starts_with("did:plc:") {
721 format!("{}/{}", plc_directory.trim_end_matches('/'), did)
722 } else {
723 return Err(AtProtoError::DidResolution {
724 did: did.to_string(),
725 reason: "unsupported DID method (only did:plc and did:web are handled)".to_string(),
726 cause: DidResolutionCause::UnsupportedMethod,
727 }
728 .into());
729 };
730
731 // SSRF guard: `doc_url` is attacker-controllable for `did:web:<host>` (the
732 // host comes straight from the DID) — a hostile `did:web:169.254.169.254`
733 // or `did:web:localhost` would otherwise make the server fetch an internal
734 // target and reflect its body. Route through the IP/scheme guard (no
735 // feed-privacy layer — this is a DID document, not a feed).
736 let resp = crate::net::guarded_get_no_privacy(client, &doc_url, &[]).await?;
737 if !resp.status().is_success() {
738 return Err(AtProtoError::DidResolution {
739 did: did.to_string(),
740 reason: format!("DID document fetch returned {}", resp.status()),
741 cause: DidResolutionCause::Status,
742 }
743 .into());
744 }
745
746 // **Capped, and this is the most remote-controlled body of the lot.** For a
747 // `did:web:` the host is taken straight out of the DID, so whoever supplies
748 // the DID chooses the server — and the SSRF guard only proves the address is
749 // public, not that the body is finite.
750 let raw = crate::net::read_capped(resp).await?;
751 let doc: DidDocument = serde_json::from_slice(&raw).context("parsing DID document")?;
752 let endpoint = doc
753 .pds_endpoint()
754 .ok_or_else(|| AtProtoError::DidResolution {
755 did: did.to_string(),
756 reason: "DID document has no #atproto_pds service endpoint".to_string(),
757 cause: DidResolutionCause::NoPdsEndpoint,
758 })?;
759
760 // SSRF guard on the RESOLVED endpoint: the `serviceEndpoint` is fully
761 // attacker-controlled (it's whatever the DID document says) and is handed to
762 // XRPC clients that fetch it directly. Reject a private/loopback/metadata
763 // target here so a hostile DID doc can't point the PDS at an internal host.
764 crate::net::assert_public_target(&endpoint)
765 .await
766 .map_err(|e| AtProtoError::DidResolution {
767 did: did.to_string(),
768 reason: format!("PDS serviceEndpoint is not a public target: {e}"),
769 cause: DidResolutionCause::NotAPublicTarget,
770 })?;
771 Ok(endpoint)
772}
773
774/// The subset of a DID document FeatherReader needs: its services, so it can
775/// find the `#atproto_pds` endpoint.
776#[derive(Debug, Clone, Deserialize)]
777pub struct DidDocument {
778 /// The document subject (the DID itself).
779 #[serde(default)]
780 pub id: String,
781 /// The declared services; the PDS is the one whose `id` ends `#atproto_pds`.
782 #[serde(default)]
783 pub service: Vec<DidService>,
784}
785
786/// One service entry in a [`DidDocument`].
787#[derive(Debug, Clone, Deserialize)]
788pub struct DidService {
789 /// The service id fragment (e.g. `#atproto_pds`).
790 pub id: String,
791 /// The service type (e.g. `AtprotoPersonalDataServer`).
792 #[serde(rename = "type", default)]
793 pub r#type: String,
794 /// The service base URL.
795 #[serde(rename = "serviceEndpoint")]
796 pub service_endpoint: String,
797}
798
799impl DidDocument {
800 /// The `#atproto_pds` service endpoint, if present.
801 pub fn pds_endpoint(&self) -> Option<String> {
802 self.service
803 .iter()
804 .find(|s| s.id.ends_with("#atproto_pds"))
805 .map(|s| s.service_endpoint.trim_end_matches('/').to_string())
806 }
807}
808
809// ---------------------------------------------------------------------------
810// Direct-PDS auth: app-password session
811// ---------------------------------------------------------------------------
812
813/// Create a session with an **app password** via
814/// `com.atproto.server.createSession`.
815///
816/// This is the direct-PDS path that makes [`PdsClient`] usable without the OAuth
817/// sidecar (local runs and tests). `pds_base` is the account's PDS (resolve it
818/// first with [`resolve_handle`] + [`resolve_did_to_pds`], or pass the entryway
819/// like `https://bsky.social`, which will service-proxy). `identifier` is a
820/// handle or DID; `app_password` is an app-password (never the main password).
821///
822/// The POST goes through [`crate::net::guarded_post_json`]. This is the single
823/// most credential-dense request in the crate — the app password travels in the
824/// JSON **body**, where reqwest's cross-origin header sanitisation cannot help
825/// it — so it gets the scheme/IP allow-list, the connect pin (no second DNS
826/// resolution to rebind), and a hard refusal to follow a redirect that would
827/// re-send that body to another host.
828pub async fn login_with_app_password(
829 client: &Client,
830 pds_base: &str,
831 identifier: &str,
832 app_password: &str,
833) -> Result<SessionAuth> {
834 let url = format!(
835 "{}/xrpc/com.atproto.server.createSession",
836 pds_base.trim_end_matches('/')
837 );
838 let body = serde_json::to_vec(&json!({ "identifier": identifier, "password": app_password }))
839 .context("serializing createSession request")?;
840 let resp = crate::net::guarded_post_json(client, &url, &[], body).await?;
841 if !resp.status().is_success() {
842 return Err(xrpc_error_from(resp).await.into());
843 }
844 let raw = crate::net::read_capped(resp).await?;
845 serde_json::from_slice(&raw).context("parsing createSession response")
846}
847
848// ---------------------------------------------------------------------------
849// The PDS client
850// ---------------------------------------------------------------------------
851
852/// A lightweight client for one user's PDS repo.
853///
854/// Holds the user's DID (the repo to read/write), the PDS base URL (resolved
855/// from the DID doc), the shared `reqwest::Client`, and the [`Auth`] token.
856/// All the `com.atproto.repo.*` methods below act on `self.did`'s repo.
857///
858/// The client may also be **anonymous** ([`PdsClient::anonymous`]), in which case
859/// it is read-only: it sends no `Authorization` header and every write path
860/// errors out of [`Auth::bearer`].
861///
862/// Cheap to clone (`Arc` internals); one is held per logged-in session.
863#[derive(Clone)]
864pub struct PdsClient {
865 http: Client,
866 /// The PDS base URL, e.g. `https://pds.example.com` (no trailing slash).
867 pds_base: Arc<str>,
868 /// The repo DID all calls target.
869 did: Arc<str>,
870 /// The auth material (an app-password session bearer for the direct path, or
871 /// [`Auth::Anonymous`] for a read-only public read of a stranger's repo).
872 auth: Auth,
873}
874
875/// A single record as returned in a `listRecords` / `getRecord` response.
876///
877/// `value` is the raw record body (with its `$type`); typed wrappers
878/// deserialize it into the matching [`crate::lexicon`] struct.
879#[derive(Debug, Clone, Deserialize)]
880pub struct RecordEntry {
881 /// The `at://did/collection/rkey` strong ref to this record.
882 pub uri: String,
883 /// The record CID (content hash).
884 #[serde(default)]
885 pub cid: Option<String>,
886 /// The raw record body.
887 pub value: Value,
888}
889
890impl RecordEntry {
891 /// The record key (the last `/`-segment of the `at://` URI).
892 pub fn rkey(&self) -> Option<&str> {
893 self.uri.rsplit('/').next()
894 }
895
896 /// Deserialize this record's `value` into a typed lexicon record.
897 pub fn parse<T: DeserializeOwned>(&self) -> Result<T> {
898 serde_json::from_value(self.value.clone())
899 .with_context(|| format!("deserializing record {}", self.uri))
900 }
901}
902
903/// The `com.atproto.repo.listRecords` response envelope.
904#[derive(Debug, Clone, Deserialize)]
905pub struct ListRecordsResponse {
906 /// The page of records.
907 #[serde(default)]
908 pub records: Vec<RecordEntry>,
909 /// The opaque pagination cursor for the next page, if any.
910 #[serde(default)]
911 pub cursor: Option<String>,
912 /// Records on this page whose envelope was malformed and were left out of
913 /// `records` (#177): one bad record no longer fails the page it is on.
914 #[serde(skip)]
915 pub malformed: usize,
916 /// The page's size on the wire, where the client knows it (0 otherwise).
917 /// A walk that SKIPS malformed records charges this to its budget, since
918 /// the skipped records are invisible to the per-record accounting.
919 #[serde(skip)]
920 pub wire_bytes: usize,
921}
922
923/// **A reader's own repo holds records this server cannot read** (#177).
924///
925/// Returned by every walk whose result is written through
926/// `store::replace_sub_refs`. Skipping a record there would silently drop it
927/// from the reader's subscriptions, so the walk refuses instead, and the web
928/// layer recognises this error by type to tell the reader why.
929#[derive(Debug, Clone, PartialEq, Eq)]
930pub struct MalformedRecords {
931 /// The collection being listed.
932 pub collection: String,
933 /// How many records on the refused page were malformed.
934 pub count: usize,
935}
936
937impl std::fmt::Display for MalformedRecords {
938 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
939 write!(
940 f,
941 "{} record(s) in {} have a malformed envelope; refusing the listing rather than \
942 dropping them",
943 self.count, self.collection
944 )
945 }
946}
947
948impl std::error::Error for MalformedRecords {}
949
950/// The `com.atproto.repo.createRecord` / `putRecord` response (a strong ref to
951/// the written record).
952#[derive(Debug, Clone, Deserialize)]
953pub struct WriteResult {
954 /// The `at://` URI of the written record.
955 pub uri: String,
956 /// The record CID after the write.
957 #[serde(default)]
958 pub cid: Option<String>,
959}
960
961impl WriteResult {
962 /// The record key — the last `/`-segment of the `at://` URI.
963 ///
964 /// The reader-facing `add_*` wrappers return this so the web layer can
965 /// address the freshly-created record (delete/rename) without a re-list.
966 pub fn rkey(&self) -> Option<&str> {
967 self.uri.rsplit('/').next()
968 }
969
970 /// The record key as an owned `String`, or the empty string if the URI is
971 /// somehow segment-less (never in practice — a PDS always returns an
972 /// `at://did/collection/rkey`). Convenience for the `-> rkey` wrappers.
973 pub fn into_rkey(self) -> String {
974 self.rkey().unwrap_or_default().to_string()
975 }
976}
977
978impl PdsClient {
979 /// Construct a client against an already-resolved PDS base + DID + auth.
980 pub fn new(
981 http: Client,
982 pds_base: impl Into<String>,
983 did: impl Into<String>,
984 auth: Auth,
985 ) -> Self {
986 Self {
987 http,
988 pds_base: Arc::from(pds_base.into().trim_end_matches('/')),
989 did: Arc::from(did.into()),
990 auth,
991 }
992 }
993
994 /// Construct a **read-only, unauthenticated** client for a public repo the
995 /// caller does not own — `com.atproto.repo.listRecords` is public on a
996 /// standard PDS, so a stranger's records need no credentials.
997 ///
998 /// Every `com.atproto.repo.*` **write** returns an error (there is no bearer;
999 /// see [`Auth::Anonymous`]). Callers are expected to have obtained `pds_base`
1000 /// from [`resolve_did_to_pds`], which already runs
1001 /// [`crate::net::assert_public_target`] on the resolved `serviceEndpoint` —
1002 /// but that is not what makes the fetch safe: every read is **re-vetted at
1003 /// fetch time** by [`crate::net::guarded_get_no_privacy`], which closes the
1004 /// DNS-rebinding window between resolve and connect. This constructor is
1005 /// deliberately synchronous and does no validation of its own, so the
1006 /// authoritative check is not duplicated (or, worse, mistaken for sufficient).
1007 pub fn anonymous(http: Client, pds_base: impl Into<String>, did: impl Into<String>) -> Self {
1008 Self::new(http, pds_base, did, Auth::Anonymous)
1009 }
1010
1011 /// Resolve `handle` → DID → PDS, obtain an app-password session, and build a
1012 /// ready-to-use client. A convenience constructor for the direct-PDS path
1013 /// that exercises the whole stack end-to-end.
1014 ///
1015 /// `resolver_base` / `plc_directory` default to [`DEFAULT_RESOLVER_HOST`] /
1016 /// [`DEFAULT_PLC_DIRECTORY`] when passed `None`.
1017 pub async fn login(
1018 http: Client,
1019 handle: &str,
1020 app_password: &str,
1021 resolver_base: Option<&str>,
1022 plc_directory: Option<&str>,
1023 ) -> Result<Self> {
1024 let resolver = resolver_base.unwrap_or(DEFAULT_RESOLVER_HOST);
1025 let plc = plc_directory.unwrap_or(DEFAULT_PLC_DIRECTORY);
1026
1027 let did = resolve_handle(&http, resolver, handle).await?;
1028 let pds_base = resolve_did_to_pds(&http, plc, &did).await?;
1029 let session = login_with_app_password(&http, &pds_base, &did, app_password).await?;
1030
1031 Ok(Self::new(
1032 http,
1033 pds_base,
1034 session.did.clone(),
1035 Auth::Session(session),
1036 ))
1037 }
1038
1039 /// The repo DID this client targets.
1040 pub fn did(&self) -> &str {
1041 &self.did
1042 }
1043
1044 /// The PDS base URL this client talks to.
1045 pub fn pds_base(&self) -> &str {
1046 &self.pds_base
1047 }
1048
1049 /// Build the `Authorization: Bearer …` header pair for an authed write.
1050 ///
1051 /// A `Vec` of pairs rather than a [`HeaderMap`] because every write now goes
1052 /// through [`crate::net::guarded_post_json`], which takes header pairs and
1053 /// sets `Content-Type: application/json` itself. Fails closed on
1054 /// [`Auth::Anonymous`] (there is no bearer), which is what makes an anonymous
1055 /// client read-only.
1056 fn authed_headers(&self) -> Result<Vec<(HeaderName, HeaderValue)>> {
1057 let bearer = self.auth.bearer()?;
1058 let mut value = HeaderValue::from_str(&format!("Bearer {bearer}"))
1059 .context("building Authorization header")?;
1060 value.set_sensitive(true);
1061 Ok(vec![(AUTHORIZATION, value)])
1062 }
1063
1064 fn xrpc_url(&self, method: &str) -> String {
1065 format!("{}/xrpc/{}", self.pds_base, method)
1066 }
1067
1068 // -- com.atproto.repo.* --------------------------------------------------
1069
1070 /// `com.atproto.repo.listRecords` — one page of a collection's records.
1071 ///
1072 /// `cursor` continues a previous page; `limit` caps the page (atproto's max
1073 /// is 100). Use [`list_all_records`](Self::list_all_records) to page fully.
1074 ///
1075 /// The fetch is routed through [`crate::net::guarded_get_no_privacy`] — the
1076 /// same per-hop scheme/IP allow-list and connect-pinning the feed poller and
1077 /// the identity-resolution paths use. `pds_base` was vetted by
1078 /// [`crate::net::assert_public_target`] at resolve time, but that is a
1079 /// *separate* DNS resolution from this fetch; routing the request through the
1080 /// guard closes the rebinding window, which matters as soon as the repo (and
1081 /// therefore the host) is chosen by a stranger. The response body is read via
1082 /// [`crate::net::read_capped`] so a hostile PDS cannot stream an unbounded
1083 /// body at a 512 MB box.
1084 pub async fn list_records(
1085 &self,
1086 collection: &str,
1087 limit: Option<u32>,
1088 cursor: Option<&str>,
1089 ) -> Result<ListRecordsResponse> {
1090 // Build the query manually (see `resolve_handle`): no reqwest `query`
1091 // feature dependency.
1092 let mut url = format!(
1093 "{}?repo={}&collection={}",
1094 self.xrpc_url("com.atproto.repo.listRecords"),
1095 urlencode(&self.did),
1096 urlencode(collection),
1097 );
1098 if let Some(limit) = limit {
1099 url.push_str(&format!("&limit={limit}"));
1100 }
1101 if let Some(cursor) = cursor {
1102 url.push_str(&format!("&cursor={}", urlencode(cursor)));
1103 }
1104
1105 // listRecords is public/unauthenticated on most PDSes, but we send the
1106 // bearer when we have a session one so private repos work too. An
1107 // Auth::Oauth / Auth::Anonymous client sends no Authorization header at
1108 // all. The guard drops the header if a redirect leaves this PDS's origin.
1109 let mut headers: Vec<(reqwest::header::HeaderName, HeaderValue)> = Vec::new();
1110 if let Auth::Session(s) = &self.auth {
1111 let mut value = HeaderValue::from_str(&format!("Bearer {}", s.access_jwt))
1112 .context("building Authorization header")?;
1113 value.set_sensitive(true);
1114 headers.push((AUTHORIZATION, value));
1115 }
1116 let resp = crate::net::guarded_get_no_privacy(&self.http, &url, &headers).await?;
1117 if !resp.status().is_success() {
1118 return Err(xrpc_error_from(resp).await.into());
1119 }
1120 let body = crate::net::read_capped(resp).await?;
1121 parse_list_records(&body)
1122 }
1123
1124 /// Page through **all** records in a collection, following the cursor until
1125 /// exhausted. Convenience over [`list_records`](Self::list_records) for the
1126 /// login-time "load the whole follow-list" read.
1127 ///
1128 /// Bounded by `MAX_LIST_PAGES` and by cursor-repetition detection, because
1129 /// `pds_base` may be a host we did not choose (see [`PdsClient::anonymous`]).
1130 pub async fn list_all_records(&self, collection: &str) -> Result<Vec<RecordEntry>> {
1131 self.list_all_records_within(collection, &mut ByteBudget::new(MAX_LIST_BYTES))
1132 .await
1133 }
1134
1135 /// [`list_all_records`](Self::list_all_records) against a caller's budget.
1136 ///
1137 /// **Shared, not per-walk.** Two walks that nest — a publication read runs a
1138 /// second walk while still holding the first's records — each had their own
1139 /// ceiling, so the process could hold twice it. Passing one budget in makes
1140 /// the bound a property of the caller's whole read, which is the thing that
1141 /// has to fit in the box, and the type enforces it where a comment would not.
1142 pub(crate) async fn list_all_records_within(
1143 &self,
1144 collection: &str,
1145 budget: &mut ByteBudget,
1146 ) -> Result<Vec<RecordEntry>> {
1147 self.walk_all_within(collection, budget, OnMalformed::Refuse)
1148 .await
1149 .map(|(records, _)| records)
1150 }
1151
1152 /// [`list_all_records_within`](Self::list_all_records_within) for a
1153 /// **stranger's** repo: a malformed record is skipped and counted instead of
1154 /// refusing the walk (#177). Never for a walk that reaches
1155 /// `replace_sub_refs`, where a skipped record is a dropped subscription.
1156 pub(crate) async fn list_all_records_skipping_within(
1157 &self,
1158 collection: &str,
1159 budget: &mut ByteBudget,
1160 ) -> Result<(Vec<RecordEntry>, usize)> {
1161 self.walk_all_within(collection, budget, OnMalformed::Skip)
1162 .await
1163 }
1164
1165 async fn walk_all_within(
1166 &self,
1167 collection: &str,
1168 budget: &mut ByteBudget,
1169 on_malformed: OnMalformed,
1170 ) -> Result<(Vec<RecordEntry>, usize)> {
1171 let mut out = Vec::new();
1172 let max_bytes = budget.max();
1173 let mut cursor: Option<String> = None;
1174 let mut more_offered = false;
1175 let mut malformed = 0usize;
1176 for _ in 0..MAX_LIST_PAGES {
1177 let page = self
1178 .list_records(collection, Some(100), cursor.as_deref())
1179 .await?;
1180 if on_malformed == OnMalformed::Refuse {
1181 refuse_malformed(&page, collection)?;
1182 }
1183 // **Skipped records still cost what they cost.** The per-record
1184 // accounting below never sees them, so without this a repo serving
1185 // pages of junk walks every page MAX_LIST_PAGES allows — gigabytes —
1186 // under a budget meant to stop it (found in review). The whole page
1187 // is charged: conservative, and only on pages that skipped something.
1188 // Charged ONCE, at the larger of what crossed the wire and what the
1189 // kept records retain: a parsed record can hold up to 42x its wire
1190 // size, so the wire alone under-charges (third review of #224).
1191 if page.malformed > 0
1192 && !budget.charge(
1193 page.wire_bytes
1194 .max(page.records.iter().map(approx_bytes).sum()),
1195 )
1196 {
1197 anyhow::bail!(
1198 "listRecords for {collection} exceeded the {max_bytes}-byte cap on pages \
1199 of malformed records — refusing to read further"
1200 );
1201 }
1202 malformed += page.malformed;
1203 // Skipped records count toward "the page had something", so a page
1204 // of nothing BUT malformed records does not end the walk early.
1205 let got = page.records.len() + page.malformed;
1206 // **Refused, not truncated**, for the reason `extend_bounded`
1207 // gives: this walk feeds `replace_sub_refs`, where a short list is
1208 // revoked access.
1209 // A page that skipped records was already charged its whole wire
1210 // size above, which covers its good records too; charging them again
1211 // failed walks that fit (found in review).
1212 if page.malformed == 0 && !budget.admit(&page.records) {
1213 anyhow::bail!(
1214 "listRecords for {collection} exceeded the {max_bytes}-byte cap \
1215 ({} held, {} bytes charged) — refusing to accumulate further",
1216 out.len(),
1217 budget.used(),
1218 );
1219 }
1220 extend_bounded(&mut out, page.records, MAX_LIST_RECORDS, collection)?;
1221 match page.cursor {
1222 // Guard against a PDS that echoes a cursor with an empty page,
1223 // or that hands back the SAME cursor forever (an infinite walk
1224 // that would otherwise re-count the same page every pass).
1225 Some(next) if got > 0 && Some(&next) != cursor.as_ref() => {
1226 cursor = Some(next);
1227 more_offered = true;
1228 }
1229 _ => {
1230 more_offered = false;
1231 break;
1232 }
1233 }
1234 }
1235 // **Running out of pages is a refusal, not a short answer.** Falling out
1236 // of the loop used to return `Ok(out)`, so a repo bigger than the page
1237 // budget produced a truncated list indistinguishable from a complete
1238 // one — and `resolve_subscriptions` needs an `Err` for its fail-closed
1239 // branch. Given `Ok`, it hands the short list to `replace_sub_refs`,
1240 // which DELETEs the reader's whole `sub_ref` projection and reinserts
1241 // only what it was given. `extend_bounded` cannot catch this either:
1242 // `MAX_LIST_PAGES` x the 100 we request is `MAX_LIST_RECORDS`, so the
1243 // page budget runs out first.
1244 //
1245 // **The cap is on REQUESTS, so where it bites in RECORDS is the server's
1246 // choice and not ours.** We ask for 100 a page; a PDS MAY answer with
1247 // fewer, and only one that honours the limit puts the boundary anywhere
1248 // near `MAX_LIST_PAGES` x 100. Halve the page size and the same budget
1249 // reaches half as many records; a server that returns MORE than asked
1250 // trips `extend_bounded` first, which is the case the sentence above does
1251 // not cover. Said this way because an earlier version of this comment
1252 // named a fixed record window as though our own constants decided it.
1253 //
1254 // **And at the boundary the refusal is a FALSE one.** Terminating costs
1255 // one extra request, because a short page can still carry a cursor — this
1256 // project's own PDS does exactly that — so a walk that fills its last
1257 // allowed page is holding every record it was ever going to hold and
1258 // refuses anyway, on the strength of a cursor it never followed. With
1259 // `limit=100` honoured that window is a repo of roughly 19 901 to 20 000
1260 // records. The direction is safe and the alternative is deleting feeds,
1261 // but it is a false refusal and not a clean boundary.
1262 if more_offered {
1263 anyhow::bail!(
1264 "listRecords for {collection} did not finish within {MAX_LIST_PAGES} pages \
1265 ({} held, and the PDS still offered more) — refusing a short list",
1266 out.len(),
1267 );
1268 }
1269 Ok((out, malformed))
1270 }
1271
1272 /// See [`RecordWalk`].
1273 /// The most recent records of a collection that the caller **keeps**,
1274 /// truncating rather than refusing.
1275 ///
1276 /// **The cap counts kept records, not walked ones.** Applying it to the
1277 /// raw collection starves a caller whose filter is selective: a quiet
1278 /// standard.site publication in a repo whose busy sibling fills the
1279 /// window returns nothing at all, permanently, and worse with every post
1280 /// the sibling makes. `MAX_LIST_PAGES` still bounds the request count, so
1281 /// a filter that matches nothing costs a fixed number of round trips.
1282 ///
1283 /// Truncating, not refusing, because this is an additive read: see
1284 /// `extend_truncating` for why the `extend_bounded` refusal would be
1285 /// strictly worse here.
1286 ///
1287 /// `page_size` is the caller's, because the right page depends on how big
1288 /// the records are: [`crate::net::read_capped`] bounds a response at 8 MB,
1289 /// so 100 long-form articles per page can exceed it and fail the whole
1290 /// walk.
1291 ///
1292 /// **Ordering is the PDS's**: `listRecords` is descending by *rkey*, which
1293 /// is newest-first only when rkeys are TIDs. For a publisher using slug
1294 /// rkeys the truncation keeps a lexicographic subset rather than a recent
1295 /// one — acceptable because the cap is now per-publication rather than
1296 /// per-repo, so reaching it at all means an archive larger than this
1297 /// reader stores.
1298 pub async fn list_recent_matching(
1299 &self,
1300 collection: &str,
1301 max_records: usize,
1302 page_size: u32,
1303 keep: impl FnMut(&RecordEntry) -> bool,
1304 ) -> Result<RecordWalk> {
1305 self.list_recent_matching_within(
1306 collection,
1307 max_records,
1308 &mut ByteBudget::new(MAX_LIST_BYTES),
1309 page_size,
1310 keep,
1311 )
1312 .await
1313 }
1314
1315 /// [`list_recent_matching`](Self::list_recent_matching) against a caller's
1316 /// budget. See [`list_all_records_within`](Self::list_all_records_within) for
1317 /// why it is the caller's and not the walk's.
1318 pub(crate) async fn list_recent_matching_within(
1319 &self,
1320 collection: &str,
1321 max_records: usize,
1322 budget: &mut ByteBudget,
1323 page_size: u32,
1324 mut keep: impl FnMut(&RecordEntry) -> bool,
1325 ) -> Result<RecordWalk> {
1326 let mut out = Vec::new();
1327 let mut cursor: Option<String> = None;
1328 let mut malformed = 0usize;
1329 // Every exit carries the skipped count, so none can forget it.
1330 let walk = |mut w: RecordWalk, malformed: usize| {
1331 w.malformed = malformed;
1332 w
1333 };
1334 for _ in 0..MAX_LIST_PAGES {
1335 let page = self
1336 .list_records(collection, Some(page_size), cursor.as_deref())
1337 .await?;
1338 // **Skipped, not refused**: this reads a stranger's collection, and
1339 // one bad record must not stall everything beside it (#177). Counted
1340 // in `got` too, so a page whose only records were malformed is not
1341 // mistaken for the end of the collection.
1342 // As in `walk_all_within`: skipped records are invisible to the
1343 // per-record charge, so their page pays for them here.
1344 let wire_charged = page.malformed > 0;
1345 malformed += page.malformed;
1346 let got = page.records.len() + page.malformed;
1347 // Is there a next page that is actually new? (A PDS may echo a
1348 // cursor with an empty page, or hand back the same one forever.)
1349 let more =
1350 matches!(&page.cursor, Some(next) if got > 0 && Some(next) != cursor.as_ref());
1351 // **The transient page needs its own bound.** The running total
1352 // charges what is KEPT, because that is what the walk retains and a
1353 // page is dropped after the filter. But transient is not free, and
1354 // `net::read_capped`'s 8 MB bounds the WIRE — the whole point of
1355 // this budget is that wire size and retained size are not the same
1356 // number. A page of records the filter rejects entirely charges
1357 // nothing against the total and can still hold hundreds of
1358 // megabytes, so no single page may exceed the walk's budget alone.
1359 // On a page that skipped records, the transient is the larger of
1360 // its wire size and its parsed records: the skipped ones are
1361 // invisible to the per-record sum.
1362 let mut page_cost: usize = page.records.iter().map(approx_bytes).sum();
1363 if wire_charged {
1364 page_cost = page_cost.max(page.wire_bytes);
1365 }
1366 if page_cost > budget.remaining() {
1367 return Ok(walk(RecordWalk::partial(out), malformed));
1368 }
1369 let kept: Vec<RecordEntry> = page.records.into_iter().filter(|r| keep(r)).collect();
1370 // Charged on what is KEPT, which is what this walk retains. **This
1371 // cannot refuse**, and saying so matters: the page check above
1372 // already proved the whole page fits in the remainder, and `kept` is
1373 // a subset of it. Written as `if !admit(…) { return }` it reads as a
1374 // second stopping rule, and a reader would look for the case that
1375 // trips it. There isn't one — the call is the bookkeeping.
1376 // A page that skipped records also pays for the skipped bytes,
1377 // once: the larger of its wire size and what it keeps.
1378 let charged = if wire_charged {
1379 budget.charge(page.wire_bytes.max(kept.iter().map(approx_bytes).sum()))
1380 } else {
1381 budget.admit(&kept)
1382 };
1383 debug_assert!(charged, "the page charge already proved this fits");
1384 if extend_truncating(&mut out, kept, max_records) {
1385 return Ok(walk(RecordWalk::partial(out), malformed));
1386 }
1387 if out.len() >= max_records {
1388 // Landing exactly on the cap is only a truncation if the
1389 // collection had more to give — `extend_truncating` cannot see
1390 // that, so the caller's "incomplete" signal is decided here.
1391 return Ok(RecordWalk {
1392 complete: !more,
1393 records: out,
1394 malformed,
1395 });
1396 }
1397 if !more {
1398 return Ok(walk(RecordWalk::complete(out), malformed));
1399 }
1400 cursor = page.cursor;
1401 }
1402 // **The page budget ran out with the collection still going.** Silence
1403 // here reintroduces the starvation this function exists to prevent, one
1404 // order of magnitude further out: a quiet publication in a repo whose
1405 // busy sibling has more records than MAX_LIST_PAGES × page_size can
1406 // reach returns nothing at all, forever, having spent every round trip
1407 // to find out. The caller is told so it can say which feed.
1408 Ok(walk(RecordWalk::partial(out), malformed))
1409 }
1410
1411 /// `com.atproto.repo.createRecord` — create a new record (server assigns the
1412 /// rkey, `key: tid`). Returns the written record's strong ref.
1413 /// **Private, not `pub` — and not `pub(crate)`.** This is generic over
1414 /// `T: Serialize`, so it will happily write a raw `lexicon::Subscription`:
1415 /// the general case of the hole `create_subscriptions_batch` was one
1416 /// instance of. The vetted wrappers in this `impl` are the sanctioned entry
1417 /// points. `pub(crate)` was tried first and stops nothing that matters — a
1418 /// handler in `web.rs` is in this crate. Private is what makes the wrappers
1419 /// a fact rather than a convention, and it costs nothing: nothing outside
1420 /// this module ever called it.
1421 async fn create_record<T: Serialize>(
1422 &self,
1423 collection: &str,
1424 record: &T,
1425 ) -> Result<WriteResult> {
1426 let body = json!({
1427 "repo": self.did.as_ref(),
1428 "collection": collection,
1429 "record": record,
1430 });
1431 self.repo_write("com.atproto.repo.createRecord", body).await
1432 }
1433
1434 /// `com.atproto.repo.putRecord` — upsert a record at a **known** rkey
1435 /// (`key: any`). This is the `readState` upsert primitive: a feed-derived
1436 /// rkey makes the write idempotent (one record per feed).
1437 /// **Private, not `pub` — and not `pub(crate)`.** This is generic over
1438 /// `T: Serialize`, so it will happily write a raw `lexicon::Subscription`:
1439 /// the general case of the hole `create_subscriptions_batch` was one
1440 /// instance of. The vetted wrappers in this `impl` are the sanctioned entry
1441 /// points. `pub(crate)` was tried first and stops nothing that matters — a
1442 /// handler in `web.rs` is in this crate. Private is what makes the wrappers
1443 /// a fact rather than a convention, and it costs nothing: nothing outside
1444 /// this module ever called it.
1445 ///
1446 /// `swap_record` is the CID the caller read the record at, sent as
1447 /// `swapRecord`: the PDS then refuses the write with `InvalidSwap` (see
1448 /// [`is_invalid_swap`]) if the record has moved since, rather than
1449 /// silently overwriting another client's change (#149). `None` omits the
1450 /// field — an unconditional write, which is what a write that read nothing
1451 /// means.
1452 async fn put_record<T: Serialize>(
1453 &self,
1454 collection: &str,
1455 rkey: &str,
1456 record: &T,
1457 swap_record: Option<&str>,
1458 ) -> Result<WriteResult> {
1459 let mut body = json!({
1460 "repo": self.did.as_ref(),
1461 "collection": collection,
1462 "rkey": rkey,
1463 "record": record,
1464 });
1465 if let Some(cid) = swap_record {
1466 body["swapRecord"] = json!(cid);
1467 }
1468 self.repo_write("com.atproto.repo.putRecord", body).await
1469 }
1470
1471 /// The single outbound path for every authenticated `com.atproto.repo.*`
1472 /// **write**, routed through [`crate::net::guarded_post_json`].
1473 ///
1474 /// Reads were hardened first (see [`list_records`](Self::list_records)), but
1475 /// the argument applies with more force here: `pds_base` is vetted by
1476 /// [`crate::net::assert_public_target`] at *resolve* time, and the write is a
1477 /// *separate* DNS resolution — the rebinding window `net.rs` exists to close.
1478 /// A write also carries the session bearer and, in
1479 /// [`login_with_app_password`], the app password itself, so the guard's
1480 /// refusal to follow redirects (a `307` re-sends the body verbatim to the new
1481 /// host) is doing real work and not just symmetry.
1482 async fn guarded_post(&self, url: &str, body: &Value) -> Result<reqwest::Response> {
1483 let headers = self.authed_headers()?;
1484 let payload = serde_json::to_vec(body).context("serializing XRPC request body")?;
1485 crate::net::guarded_post_json(&self.http, url, &headers, payload).await
1486 }
1487
1488 /// `com.atproto.repo.deleteRecord` — delete a record by collection + rkey
1489 /// (e.g. unsubscribe → delete the subscription record).
1490 pub async fn delete_record(&self, collection: &str, rkey: &str) -> Result<()> {
1491 let url = self.xrpc_url("com.atproto.repo.deleteRecord");
1492 let body = json!({
1493 "repo": self.did.as_ref(),
1494 "collection": collection,
1495 "rkey": rkey,
1496 });
1497 let resp = self.guarded_post(&url, &body).await?;
1498 if !resp.status().is_success() {
1499 return Err(xrpc_error_from(resp).await.into());
1500 }
1501 Ok(())
1502 }
1503
1504 /// `com.atproto.repo.applyWrites` — a **batch** of create/update/delete
1505 /// operations in one atomic-per-repo round-trip.
1506 ///
1507 /// This is the read-state flusher's workhorse: dozens of dirty per-feed
1508 /// [`ReadState`] cursors coalesce into one call rather than one `putRecord`
1509 /// each. See [`flush_read_states`](Self::flush_read_states).
1510 /// **Private, not `pub` — and not `pub(crate)`.** This is generic over
1511 /// `T: Serialize`, so it will happily write a raw `lexicon::Subscription`:
1512 /// the general case of the hole `create_subscriptions_batch` was one
1513 /// instance of. The vetted wrappers in this `impl` are the sanctioned entry
1514 /// points. `pub(crate)` was tried first and stops nothing that matters — a
1515 /// handler in `web.rs` is in this crate. Private is what makes the wrappers
1516 /// a fact rather than a convention, and it costs nothing: nothing outside
1517 /// this module ever called it.
1518 ///
1519 /// Sent in chunks within the PDS's limits — see [`apply_writes_chunked`]
1520 /// for what a failure part-way means. This used to send any batch as one
1521 /// call, including an empty one; an empty batch now sends nothing, as the
1522 /// other two clients already did.
1523 async fn apply_writes(&self, writes: &[WriteOp]) -> Result<()> {
1524 apply_writes_chunked(writes, |chunk| self.apply_writes_once(chunk)).await
1525 }
1526
1527 /// One `applyWrites` call, unchunked. Reached only through
1528 /// [`apply_writes`](Self::apply_writes).
1529 async fn apply_writes_once(&self, writes: &[WriteOp]) -> Result<()> {
1530 let url = self.xrpc_url("com.atproto.repo.applyWrites");
1531 let ops: Vec<Value> = writes.iter().map(WriteOp::to_json).collect();
1532 let body = json!({
1533 "repo": self.did.as_ref(),
1534 "writes": ops,
1535 });
1536 let resp = self.guarded_post(&url, &body).await?;
1537 if !resp.status().is_success() {
1538 return Err(xrpc_error_from(resp).await.into());
1539 }
1540 Ok(())
1541 }
1542
1543 /// Shared create/put path (both return a `{uri,cid}` strong ref).
1544 async fn repo_write(&self, method: &str, body: Value) -> Result<WriteResult> {
1545 let url = self.xrpc_url(method);
1546 let resp = self.guarded_post(&url, &body).await?;
1547 if !resp.status().is_success() {
1548 return Err(xrpc_error_from(resp).await.into());
1549 }
1550 // `read_capped` rather than `resp.json()`: a hostile PDS must not be able
1551 // to stream an unbounded body at a 512 MB box (same rule as the reads).
1552 let raw = crate::net::read_capped(resp).await?;
1553 serde_json::from_slice(&raw).with_context(|| format!("parsing {method} response"))
1554 }
1555
1556 // -- typed lexicon wrappers ---------------------------------------------
1557
1558 /// List every [`Subscription`] record in the user's repo (paged fully). The
1559 /// login-time "what does this user follow?" read.
1560 pub async fn list_subscriptions(&self) -> Result<Vec<(String, Subscription)>> {
1561 self.list_typed(lexicon::nsid::SUBSCRIPTION).await
1562 }
1563
1564 /// Every [`Subscription`] with the CID it was listed at (#149).
1565 pub async fn list_subscriptions_with_cids(
1566 &self,
1567 ) -> Result<Vec<(String, Option<String>, Subscription)>> {
1568 self.list_typed_with_cids(lexicon::nsid::SUBSCRIPTION).await
1569 }
1570
1571 /// Create a [`Subscription`] record (subscribe to a feed).
1572 pub async fn create_subscription(
1573 &self,
1574 sub: &crate::vetted::VettedSubscription,
1575 ) -> Result<WriteResult> {
1576 self.create_record(lexicon::nsid::SUBSCRIPTION, sub).await
1577 }
1578
1579 /// List every [`Folder`] record in the user's repo.
1580 pub async fn list_folders(&self) -> Result<Vec<(String, Folder)>> {
1581 self.list_typed(lexicon::nsid::FOLDER).await
1582 }
1583
1584 /// Every [`Folder`] with the CID it was listed at (#268).
1585 pub async fn list_folders_with_cids(&self) -> Result<Vec<(String, Option<String>, Folder)>> {
1586 self.list_typed_with_cids(lexicon::nsid::FOLDER).await
1587 }
1588
1589 /// Create a [`Folder`] record.
1590 pub async fn create_folder(&self, folder: &Folder) -> Result<WriteResult> {
1591 self.create_record(lexicon::nsid::FOLDER, folder).await
1592 }
1593
1594 /// List every [`Saved`] (starred) record in the user's repo.
1595 pub async fn list_saved(&self) -> Result<Vec<(String, Saved)>> {
1596 self.list_typed(lexicon::nsid::SAVED).await
1597 }
1598
1599 /// Create a [`Saved`] record (star an article).
1600 pub async fn create_saved(&self, saved: &crate::vetted::VettedSaved) -> Result<WriteResult> {
1601 self.create_record(lexicon::nsid::SAVED, saved).await
1602 }
1603
1604 /// List every [`ReadState`] cursor in the user's repo (the read side a
1605 /// login-time read-state merge would consume).
1606 pub async fn list_read_states(&self) -> Result<Vec<(String, ReadState)>> {
1607 self.list_typed(lexicon::nsid::READ_STATE).await
1608 }
1609
1610 /// Upsert a single [`ReadState`] cursor at its feed-derived rkey. For a
1611 /// batch of dirty cursors prefer [`flush_read_states`](Self::flush_read_states).
1612 pub async fn put_read_state(&self, rkey: &str, state: &ReadState) -> Result<WriteResult> {
1613 self.put_record(lexicon::nsid::READ_STATE, rkey, state, None)
1614 .await
1615 }
1616
1617 /// Batch-flush many dirty [`ReadState`] cursors via `applyWrites` (chunked) —
1618 /// the debounced read-state flusher's coalesced write.
1619 ///
1620 /// Each `(rkey, state, pds_created)` becomes a `create` op at the feed-derived
1621 /// rkey when the record does not yet exist, and an `update` when it does — so a
1622 /// feed's FIRST flush succeeds (an `#update` on a missing record errors, and
1623 /// `applyWrites` is atomic per-repo). Both kinds ride the same batch.
1624 pub async fn flush_read_states(&self, cursors: &[(String, ReadState, bool)]) -> Result<()> {
1625 if cursors.is_empty() {
1626 return Ok(());
1627 }
1628 let writes = read_state_write_ops(cursors)?;
1629 self.apply_writes(&writes).await
1630 }
1631
1632 /// List a collection and parse each record's value into `T`, pairing it with
1633 /// its rkey. Records that fail to deserialize are skipped with a warning
1634 /// (forward-compat: a future writer's extra fields shouldn't break login).
1635 async fn list_typed<T: DeserializeOwned>(&self, collection: &str) -> Result<Vec<(String, T)>> {
1636 Ok(self
1637 .list_typed_with_cids(collection)
1638 .await?
1639 .into_iter()
1640 .map(|(rkey, _cid, value)| (rkey, value))
1641 .collect())
1642 }
1643
1644 /// [`list_typed`](Self::list_typed), keeping each record's CID (#149).
1645 async fn list_typed_with_cids<T: DeserializeOwned>(
1646 &self,
1647 collection: &str,
1648 ) -> Result<Vec<(String, Option<String>, T)>> {
1649 let records = self.list_all_records(collection).await?;
1650 let mut out = Vec::with_capacity(records.len());
1651 for rec in records {
1652 let rkey = rec.rkey().unwrap_or_default().to_string();
1653 match rec.parse::<T>() {
1654 Ok(value) => out.push((rkey, rec.cid, value)),
1655 Err(e) => tracing::warn!(
1656 collection,
1657 uri = %rec.uri,
1658 error = %e,
1659 "skipping unparseable record in collection"
1660 ),
1661 }
1662 }
1663 Ok(out)
1664 }
1665}
1666
1667// ---------------------------------------------------------------------------
1668// The OAuth sidecar client — the LIVE com.atproto.repo.* path
1669// ---------------------------------------------------------------------------
1670
1671/// A client for the atproto OAuth sidecar's **internal** API.
1672///
1673/// This is the live path for every authed repo operation. Rather than the Rust
1674/// server holding PDS tokens, it POSTs `{did, action, …}` to the sidecar's
1675/// `/internal/repo` endpoint (gated by the shared `X-Internal-Secret`); the
1676/// sidecar `restore(did)`s the OAuth session — transparent DPoP + token refresh —
1677/// and runs the matching XRPC call via `@atproto/api`. The `did` (plus the shared
1678/// secret) is what authorizes the call; there is no bearer token on the Rust side.
1679///
1680/// It also fronts `/internal/session/:id`, the one-shot handoff the Rust callback
1681/// uses to turn a `session_id` (from the sidecar's browser redirect) into the
1682/// `{did, handle}` it keys its own signed cookie by.
1683///
1684/// Cheap to clone (shared `reqwest::Client` + `Arc`'d config).
1685#[derive(Clone)]
1686pub struct SidecarClient {
1687 http: Client,
1688 public_url: Arc<str>,
1689 internal_url: Arc<str>,
1690 internal_secret: Arc<str>,
1691}
1692
1693/// The `{did, handle}` a session-id resolves to (the sidecar's
1694/// `/internal/session/:id` body).
1695#[derive(Debug, Clone, Deserialize)]
1696pub struct SidecarSession {
1697 /// The account DID that logged in.
1698 pub did: String,
1699 /// The account handle at login time.
1700 #[serde(default)]
1701 pub handle: Option<String>,
1702}
1703
1704/// The sidecar's `/internal/revoke` response body:
1705/// `{ ok:true, did, revoked, hadSession }`.
1706#[derive(Debug, Clone, Deserialize)]
1707pub struct RevokeResult {
1708 /// The DID that was revoked.
1709 #[serde(default)]
1710 pub did: String,
1711 /// Whether the OAuth token revocation at the PDS succeeded. `false` means
1712 /// the local rows were still purged (best-effort), but the PDS-side tokens
1713 /// may not have been invalidated (network failure).
1714 #[serde(default)]
1715 pub revoked: bool,
1716 /// Whether the sidecar actually had a stored session for the DID.
1717 #[serde(default, rename = "hadSession")]
1718 pub had_session: bool,
1719}
1720
1721/// The action verbs the sidecar's `/internal/repo` endpoint dispatches on.
1722#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1723pub enum RepoAction {
1724 /// `com.atproto.repo.listRecords`.
1725 List,
1726 /// `com.atproto.repo.createRecord`.
1727 Create,
1728 /// `com.atproto.repo.putRecord`.
1729 Put,
1730 /// `com.atproto.repo.deleteRecord`.
1731 Delete,
1732 /// `com.atproto.repo.applyWrites` (batch).
1733 ApplyWrites,
1734}
1735
1736impl RepoAction {
1737 fn as_str(self) -> &'static str {
1738 match self {
1739 RepoAction::List => "list",
1740 RepoAction::Create => "create",
1741 RepoAction::Put => "put",
1742 RepoAction::Delete => "delete",
1743 RepoAction::ApplyWrites => "applyWrites",
1744 }
1745 }
1746}
1747
1748/// The `/internal/repo` success envelope: `{ ok:true, data:<raw XRPC JSON> }`.
1749#[derive(Debug, Deserialize)]
1750struct RepoOk {
1751 #[serde(default)]
1752 data: Value,
1753}
1754
1755/// `/internal/repo`'s ok envelope for a **listing**, typed all the way down.
1756///
1757/// `data` absent is not `data` empty, the same distinction `records` carries: it
1758/// is what a proxy makes of an unexpected upstream body.
1759#[derive(Debug, Deserialize)]
1760struct RepoOkList {
1761 /// The sidecar's own `ok`, which it can set false on a 200.
1762 #[serde(default)]
1763 ok: Option<bool>,
1764 /// And its own `error` — a different envelope from the PDS's, one layer out.
1765 #[serde(default)]
1766 error: Option<Value>,
1767 #[serde(default)]
1768 data: Option<ListRecordsBody>,
1769}
1770
1771/// The `/internal/repo` error envelope: `{ ok:false, error, message, status? }`.
1772#[derive(Debug, Deserialize)]
1773struct RepoErr {
1774 #[serde(default)]
1775 error: Option<String>,
1776 #[serde(default)]
1777 message: Option<String>,
1778 #[serde(default)]
1779 status: Option<u16>,
1780}
1781
1782impl SidecarClient {
1783 /// Build a sidecar client from the shared `reqwest::Client` and the
1784 /// resolved public + internal base URLs + internal secret (from
1785 /// [`crate::config::SidecarConfig`]). `public_url` anchors the browser
1786 /// `/login` redirect; `internal_url` is the loopback base for the `/internal/*`
1787 /// API (they collapse to the same value in single-URL local dev).
1788 pub fn new(
1789 http: Client,
1790 public_url: impl Into<String>,
1791 internal_url: impl Into<String>,
1792 internal_secret: impl Into<String>,
1793 ) -> Self {
1794 Self {
1795 http,
1796 public_url: Arc::from(public_url.into().trim_end_matches('/')),
1797 internal_url: Arc::from(internal_url.into().trim_end_matches('/')),
1798 internal_secret: Arc::from(internal_secret.into()),
1799 }
1800 }
1801
1802 /// The sidecar's public `/login` URL for a handle, round-tripping an opaque
1803 /// `return` value through OAuth state (used to bounce the browser back to a
1804 /// specific place after login). The browser is redirected here.
1805 pub fn login_url(&self, handle: &str, return_to: Option<&str>) -> String {
1806 let mut url = format!("{}/login?handle={}", self.public_url, urlencode(handle));
1807 if let Some(r) = return_to {
1808 url.push_str(&format!("&return={}", urlencode(r)));
1809 }
1810 url
1811 }
1812
1813 /// Resolve a one-shot `session_id` (from the sidecar's post-OAuth redirect)
1814 /// to the `{did, handle}` that logged in. `Ok(None)` on `404 SessionNotFound`.
1815 pub async fn resolve_session(&self, session_id: &str) -> Result<Option<SidecarSession>> {
1816 let url = format!(
1817 "{}/internal/session/{}",
1818 self.internal_url,
1819 urlencode(session_id)
1820 );
1821 let resp = self
1822 .http
1823 .get(&url)
1824 .header("X-Internal-Secret", self.internal_secret.as_ref())
1825 .send()
1826 .await?;
1827 if resp.status() == StatusCode::NOT_FOUND {
1828 return Ok(None);
1829 }
1830 if !resp.status().is_success() {
1831 return Err(xrpc_error_from(resp).await.into());
1832 }
1833 let raw = crate::net::read_capped(resp).await?;
1834 let session: SidecarSession =
1835 serde_json::from_slice(&raw).context("parsing /internal/session response")?;
1836 Ok(Some(session))
1837 }
1838
1839 /// Revoke a DID's OAuth session at the sidecar: `POST /internal/revoke`.
1840 ///
1841 /// This revokes the refresh + access tokens at the PDS **and** purges the
1842 /// sidecar's stored `oauth_session` + `app_session` rows for the DID. It is
1843 /// idempotent — revoking a DID with no live session returns
1844 /// `had_session: false`. Called on `/logout` (so the cookie clear isn't the
1845 /// only thing that ends the session) and on `/account/delete`.
1846 pub async fn revoke_session(&self, did: &str) -> Result<RevokeResult> {
1847 let url = format!("{}/internal/revoke", self.internal_url);
1848 let resp = self
1849 .http
1850 .post(&url)
1851 .header("X-Internal-Secret", self.internal_secret.as_ref())
1852 .json(&json!({ "did": did }))
1853 .send()
1854 .await?;
1855 if !resp.status().is_success() {
1856 return Err(xrpc_error_from(resp).await.into());
1857 }
1858 let raw = crate::net::read_capped(resp).await?;
1859 let result: RevokeResult =
1860 serde_json::from_slice(&raw).context("parsing /internal/revoke response")?;
1861 Ok(result)
1862 }
1863
1864 /// POST one op to `/internal/repo` and return the raw XRPC `data` payload.
1865 ///
1866 /// `body` must already carry `did` + `action` + the action's required fields
1867 /// (the typed wrappers below build these). Maps the sidecar's error envelope
1868 /// to [`AtProtoError`]: `404 SessionNotFound` → `Xrpc{error:"SessionNotFound"}`
1869 /// so callers can treat it as "re-login required".
1870 async fn repo(&self, body: Value) -> Result<Value> {
1871 let raw = self.repo_bytes(body).await?;
1872 // **The same guard as the listing path, twelve lines below.** This is
1873 // the response path for every write — create, put, delete, applyWrites —
1874 // and `RepoOk.data` is an unbounded `Value`. Measured without it: the
1875 // identical 8 MB attack retains 786 MB. The listing had the guard and
1876 // this did not, which is the drift a shared helper exists to prevent.
1877 refuse_a_structure_explosion(&raw, "the /internal/repo body")?;
1878 let ok: RepoOk = serde_json::from_slice(&raw).context("parsing /internal/repo ok body")?;
1879 Ok(ok.data)
1880 }
1881
1882 /// [`repo`](Self::repo) without the `Value`.
1883 ///
1884 /// The listing path needs the bytes: a `serde_json::Map` resolves a repeated
1885 /// key last-wins, so a body carrying a second, empty `records` array read as
1886 /// a successful empty page — and an empty page here is `replace_sub_refs`
1887 /// deleting every `sub_ref` the reader has. serde refuses a duplicated field
1888 /// outright, but only if it sees the bytes.
1889 async fn repo_bytes(&self, body: Value) -> Result<Vec<u8>> {
1890 let url = format!("{}/internal/repo", self.internal_url);
1891 let resp = self
1892 .http
1893 .post(&url)
1894 .header("X-Internal-Secret", self.internal_secret.as_ref())
1895 .json(&body)
1896 .send()
1897 .await?;
1898 let status = resp.status();
1899 // **Capped, like every other body this codebase reads.** `resp.json()`
1900 // buffers whatever arrives; `/internal/repo` proxies the account's PDS,
1901 // so that length is chosen by a host the reader picked and we did not.
1902 // The 8 MB ceiling that bounds the direct client did not exist here,
1903 // and the sidecar is the default backend — so the one path with no byte
1904 // bound at all was the one most deployments run.
1905 let raw = crate::net::read_capped(resp).await;
1906 if status.is_success() {
1907 return raw;
1908 }
1909 // Error path: parse the sidecar's `{ok:false,error,message,status}` shape.
1910 //
1911 // **A body we could not read must not cost us the status.** Reading
1912 // before the branch was the obvious shape and it swallowed the HTTP
1913 // status on an over-cap or truncated error body, turning a `404
1914 // SessionNotFound` into a bare "body exceeded the cap". `xrpc_error_from`
1915 // already makes the opposite choice deliberately, for the same reason.
1916 let err: RepoErr = raw
1917 .ok()
1918 .and_then(|body| serde_json::from_slice(&body).ok())
1919 .unwrap_or(RepoErr {
1920 error: None,
1921 message: None,
1922 status: None,
1923 });
1924 let mapped = err
1925 .status
1926 .and_then(|s| StatusCode::from_u16(s).ok())
1927 .unwrap_or(status);
1928 Err(AtProtoError::Xrpc {
1929 status: mapped,
1930 error: err.error.unwrap_or_else(|| "Unknown".to_string()),
1931 message: err.message,
1932 }
1933 .into())
1934 }
1935
1936 // -- raw com.atproto.repo.* over the sidecar -----------------------------
1937
1938 /// `list` — one page of a collection's records for `did`.
1939 pub async fn list_records(
1940 &self,
1941 did: &str,
1942 collection: &str,
1943 limit: Option<u32>,
1944 cursor: Option<&str>,
1945 ) -> Result<ListRecordsResponse> {
1946 let mut body = json!({
1947 "did": did,
1948 "action": RepoAction::List.as_str(),
1949 "collection": collection,
1950 });
1951 if let Some(limit) = limit {
1952 body["limit"] = json!(limit);
1953 }
1954 if let Some(cursor) = cursor {
1955 body["cursor"] = json!(cursor);
1956 }
1957 // Straight into the shared wire struct, one parse, no `Value` between —
1958 // so this client gets the same guards as the other two, including the
1959 // duplicated-key refusal that only serde can make.
1960 let raw = self.repo_bytes(body).await?;
1961 refuse_a_structure_explosion(&raw, "the sidecar listRecords body")?;
1962 let envelope: RepoOkList =
1963 serde_json::from_slice(&raw).context("parsing sidecar listRecords data")?;
1964 // **Two envelope layers here, not one.** `page_from_body` guards the
1965 // PDS's, which arrives inside `data`; this is the sidecar's own, and it
1966 // can say `{"ok":false,"error":"ExpiredToken"}` on a 200 while still
1967 // carrying a `data` that reads as a perfectly good empty page.
1968 if envelope.ok == Some(false) {
1969 let name = envelope
1970 .error
1971 .as_ref()
1972 .and_then(envelope_error_name)
1973 .unwrap_or_else(|| "unspecified".to_string());
1974 anyhow::bail!("the sidecar answered 2xx with ok:false ({name})");
1975 }
1976 if let Some(name) = envelope.error.as_ref().and_then(envelope_error_name) {
1977 anyhow::bail!("the sidecar answered 2xx with an error envelope: {name}");
1978 }
1979 let Some(data) = envelope.data else {
1980 anyhow::bail!("listRecords returned no records field (empty or unexpected body)");
1981 };
1982 page_from_body(data).context("parsing sidecar listRecords data")
1983 }
1984
1985 /// Page through **all** records in a collection for `did`.
1986 ///
1987 /// Bounded by `MAX_LIST_PAGES` and cursor-repetition detection, same as
1988 /// [`PdsClient::list_all_records`] — the sidecar proxies to the account's
1989 /// PDS, so the page count is ultimately remote-controlled here too.
1990 pub async fn list_all_records(&self, did: &str, collection: &str) -> Result<Vec<RecordEntry>> {
1991 self.list_all_records_within(did, collection, &mut ByteBudget::new(MAX_LIST_BYTES))
1992 .await
1993 }
1994
1995 /// [`list_all_records`](Self::list_all_records) against a caller's budget.
1996 pub(crate) async fn list_all_records_within(
1997 &self,
1998 did: &str,
1999 collection: &str,
2000 budget: &mut ByteBudget,
2001 ) -> Result<Vec<RecordEntry>> {
2002 let mut out = Vec::new();
2003 let max_bytes = budget.max();
2004 let mut cursor: Option<String> = None;
2005 let mut more_offered = false;
2006 for _ in 0..MAX_LIST_PAGES {
2007 let page = self
2008 .list_records(did, collection, Some(100), cursor.as_deref())
2009 .await?;
2010 refuse_malformed(&page, collection)?;
2011 let got = page.records.len();
2012 // The sidecar proxies the account's PDS, so this walk's size is as
2013 // remote-controlled as the direct client's. It carried no budget at
2014 // all until a review noticed it was the default backend.
2015 if !budget.admit(&page.records) {
2016 anyhow::bail!(
2017 "listRecords for {collection} exceeded the {max_bytes}-byte cap \
2018 ({} held, {} bytes charged) — refusing to accumulate further",
2019 out.len(),
2020 budget.used(),
2021 );
2022 }
2023 extend_bounded(&mut out, page.records, MAX_LIST_RECORDS, collection)?;
2024 match page.cursor {
2025 Some(next) if got > 0 && Some(&next) != cursor.as_ref() => {
2026 cursor = Some(next);
2027 more_offered = true;
2028 }
2029 _ => {
2030 more_offered = false;
2031 break;
2032 }
2033 }
2034 }
2035 // **Running out of pages is a refusal, not a short answer.** Falling out
2036 // of the loop used to return `Ok(out)`, so a repo bigger than the page
2037 // budget produced a truncated list indistinguishable from a complete
2038 // one — and `resolve_subscriptions` needs an `Err` for its fail-closed
2039 // branch. Given `Ok`, it hands the short list to `replace_sub_refs`,
2040 // which DELETEs the reader's whole `sub_ref` projection and reinserts
2041 // only what it was given. `extend_bounded` cannot catch this either:
2042 // `MAX_LIST_PAGES` x the 100 we request is `MAX_LIST_RECORDS`, so the
2043 // page budget runs out first.
2044 //
2045 // **The cap is on REQUESTS, so where it bites in RECORDS is the server's
2046 // choice and not ours.** We ask for 100 a page; a PDS MAY answer with
2047 // fewer, and only one that honours the limit puts the boundary anywhere
2048 // near `MAX_LIST_PAGES` x 100. Halve the page size and the same budget
2049 // reaches half as many records; a server that returns MORE than asked
2050 // trips `extend_bounded` first, which is the case the sentence above does
2051 // not cover. Said this way because an earlier version of this comment
2052 // named a fixed record window as though our own constants decided it.
2053 //
2054 // **And at the boundary the refusal is a FALSE one.** Terminating costs
2055 // one extra request, because a short page can still carry a cursor — this
2056 // project's own PDS does exactly that — so a walk that fills its last
2057 // allowed page is holding every record it was ever going to hold and
2058 // refuses anyway, on the strength of a cursor it never followed. With
2059 // `limit=100` honoured that window is a repo of roughly 19 901 to 20 000
2060 // records. The direction is safe and the alternative is deleting feeds,
2061 // but it is a false refusal and not a clean boundary.
2062 if more_offered {
2063 anyhow::bail!(
2064 "listRecords for {collection} did not finish within {MAX_LIST_PAGES} pages \
2065 ({} held, and the PDS still offered more) — refusing a short list",
2066 out.len(),
2067 );
2068 }
2069 Ok(out)
2070 }
2071
2072 /// `create` — create a record (server-assigned rkey). Returns its strong ref.
2073 /// **Private, not `pub` — and not `pub(crate)`.** This is generic over
2074 /// `T: Serialize`, so it will happily write a raw `lexicon::Subscription`:
2075 /// the general case of the hole `create_subscriptions_batch` was one
2076 /// instance of. The vetted wrappers in this `impl` are the sanctioned entry
2077 /// points. `pub(crate)` was tried first and stops nothing that matters — a
2078 /// handler in `web.rs` is in this crate. Private is what makes the wrappers
2079 /// a fact rather than a convention, and it costs nothing: nothing outside
2080 /// this module ever called it.
2081 async fn create_record<T: Serialize>(
2082 &self,
2083 did: &str,
2084 collection: &str,
2085 record: &T,
2086 ) -> Result<WriteResult> {
2087 let body = json!({
2088 "did": did,
2089 "action": RepoAction::Create.as_str(),
2090 "collection": collection,
2091 "record": record,
2092 });
2093 let data = self.repo(body).await?;
2094 serde_json::from_value(data).context("parsing sidecar createRecord data")
2095 }
2096
2097 /// `put` — upsert a record at a known rkey. Returns its strong ref.
2098 /// **Private, not `pub` — and not `pub(crate)`.** This is generic over
2099 /// `T: Serialize`, so it will happily write a raw `lexicon::Subscription`:
2100 /// the general case of the hole `create_subscriptions_batch` was one
2101 /// instance of. The vetted wrappers in this `impl` are the sanctioned entry
2102 /// points. `pub(crate)` was tried first and stops nothing that matters — a
2103 /// handler in `web.rs` is in this crate. Private is what makes the wrappers
2104 /// a fact rather than a convention, and it costs nothing: nothing outside
2105 /// this module ever called it.
2106 async fn put_record<T: Serialize>(
2107 &self,
2108 did: &str,
2109 collection: &str,
2110 rkey: &str,
2111 record: &T,
2112 swap_record: Option<&str>,
2113 ) -> Result<WriteResult> {
2114 let mut body = json!({
2115 "did": did,
2116 "action": RepoAction::Put.as_str(),
2117 "collection": collection,
2118 "rkey": rkey,
2119 "record": record,
2120 });
2121 // The sidecar validates it as a CID and passes it to `putRecord`; its
2122 // `InvalidSwap` comes back through `repo`'s error envelope with the
2123 // PDS's status and name intact (#149).
2124 if let Some(cid) = swap_record {
2125 body["swapRecord"] = json!(cid);
2126 }
2127 let data = self.repo(body).await?;
2128 serde_json::from_value(data).context("parsing sidecar putRecord data")
2129 }
2130
2131 /// `delete` — delete a record by collection + rkey.
2132 pub async fn delete_record(&self, did: &str, collection: &str, rkey: &str) -> Result<()> {
2133 let body = json!({
2134 "did": did,
2135 "action": RepoAction::Delete.as_str(),
2136 "collection": collection,
2137 "rkey": rkey,
2138 });
2139 // A 200 carrying an error envelope is not a delete: this reported
2140 // success while the record stayed in the reader's repo, and the UI
2141 // showed them unsubscribed from a feed they still had.
2142 self.repo(body)
2143 .await
2144 .and_then(|data| reject_error_envelope(&data))?;
2145 Ok(())
2146 }
2147
2148 /// `applyWrites` — a batch of create/update/delete ops in one round-trip.
2149 /// **Private, not `pub` — and not `pub(crate)`.** This is generic over
2150 /// `T: Serialize`, so it will happily write a raw `lexicon::Subscription`:
2151 /// the general case of the hole `create_subscriptions_batch` was one
2152 /// instance of. The vetted wrappers in this `impl` are the sanctioned entry
2153 /// points. `pub(crate)` was tried first and stops nothing that matters — a
2154 /// handler in `web.rs` is in this crate. Private is what makes the wrappers
2155 /// a fact rather than a convention, and it costs nothing: nothing outside
2156 /// this module ever called it.
2157 ///
2158 /// Sent in chunks within the PDS's limits — see [`apply_writes_chunked`]
2159 /// for what a failure part-way means. **Chunked here, not in the sidecar**,
2160 /// although the sidecar is what calls the PDS: it answers one
2161 /// `/internal/repo` request with one result, which has no way to say that
2162 /// half a batch landed, and this client is its only caller. Chunking here
2163 /// also keeps the hop itself under the sidecar's 1 MiB Fastify body limit.
2164 async fn apply_writes(&self, did: &str, writes: &[WriteOp]) -> Result<()> {
2165 apply_writes_chunked(writes, |chunk| self.apply_writes_once(did, chunk)).await
2166 }
2167
2168 /// One `/internal/repo` `applyWrites`, unchunked. Reached only through
2169 /// [`apply_writes`](Self::apply_writes).
2170 async fn apply_writes_once(&self, did: &str, writes: &[WriteOp]) -> Result<()> {
2171 let ops: Vec<Value> = writes.iter().map(WriteOp::to_sidecar_json).collect();
2172 let body = json!({
2173 "did": did,
2174 "action": RepoAction::ApplyWrites.as_str(),
2175 "writes": ops,
2176 });
2177 self.repo(body)
2178 .await
2179 .and_then(|data| reject_error_envelope(&data))?;
2180 Ok(())
2181 }
2182
2183 // -- typed lexicon wrappers (mirror the old PdsClient surface) ------------
2184
2185 /// List every [`Subscription`] record in `did`'s repo (paged fully).
2186 pub async fn list_subscriptions(&self, did: &str) -> Result<Vec<(String, Subscription)>> {
2187 self.list_typed(did, lexicon::nsid::SUBSCRIPTION).await
2188 }
2189
2190 /// Every [`Subscription`] with the CID it was listed at, unsorted — the
2191 /// read half of a read-modify-write that puts with `swapRecord` (#149).
2192 pub async fn list_subscriptions_with_cids(
2193 &self,
2194 did: &str,
2195 ) -> Result<Vec<(String, Option<String>, Subscription)>> {
2196 self.list_typed_with_cids(did, lexicon::nsid::SUBSCRIPTION)
2197 .await
2198 }
2199
2200 /// Create a [`Subscription`] record (subscribe to a feed).
2201 pub async fn create_subscription(
2202 &self,
2203 did: &str,
2204 sub: &crate::vetted::VettedSubscription,
2205 ) -> Result<WriteResult> {
2206 self.create_record(did, lexicon::nsid::SUBSCRIPTION, sub)
2207 .await
2208 }
2209
2210 /// Delete a [`Subscription`] record by rkey (unsubscribe).
2211 pub async fn delete_subscription(&self, did: &str, rkey: &str) -> Result<()> {
2212 self.delete_record(did, lexicon::nsid::SUBSCRIPTION, rkey)
2213 .await
2214 }
2215
2216 /// List every [`Folder`] record in `did`'s repo.
2217 pub async fn list_folders(&self, did: &str) -> Result<Vec<(String, Folder)>> {
2218 self.list_typed(did, lexicon::nsid::FOLDER).await
2219 }
2220
2221 /// Every [`Folder`] with the CID it was listed at, unsorted (#268).
2222 pub async fn list_folders_with_cids(
2223 &self,
2224 did: &str,
2225 ) -> Result<Vec<(String, Option<String>, Folder)>> {
2226 self.list_typed_with_cids(did, lexicon::nsid::FOLDER).await
2227 }
2228
2229 /// List every [`Saved`] record in `did`'s repo.
2230 pub async fn list_saved(&self, did: &str) -> Result<Vec<(String, Saved)>> {
2231 self.list_typed(did, lexicon::nsid::SAVED).await
2232 }
2233
2234 /// List every [`ReadState`] cursor in `did`'s repo (the read side a
2235 /// login-time read-state merge would consume).
2236 pub async fn list_read_states(&self, did: &str) -> Result<Vec<(String, ReadState)>> {
2237 self.list_typed(did, lexicon::nsid::READ_STATE).await
2238 }
2239
2240 /// Upsert a single [`ReadState`] cursor at its feed-derived rkey.
2241 pub async fn put_read_state(
2242 &self,
2243 did: &str,
2244 rkey: &str,
2245 state: &ReadState,
2246 ) -> Result<WriteResult> {
2247 self.put_record(did, lexicon::nsid::READ_STATE, rkey, state, None)
2248 .await
2249 }
2250
2251 /// Batch-flush many dirty [`ReadState`] cursors via `applyWrites` (chunked).
2252 ///
2253 /// Each `(rkey, state, pds_created)` becomes a `create` op at the feed-derived
2254 /// rkey when the record does NOT yet exist (`pds_created == false`), and an
2255 /// `update` op when it does. This is what makes the FIRST flush of a feed
2256 /// succeed: `applyWrites#update` errors on a record that does not pre-exist,
2257 /// and `applyWrites` is atomic per-repo, so a single not-yet-created cursor
2258 /// would otherwise drop the whole DID batch. Both kinds ride the SAME
2259 /// `applyWrites` batch so batching is preserved.
2260 pub async fn flush_read_states(
2261 &self,
2262 did: &str,
2263 cursors: &[(String, ReadState, bool)],
2264 ) -> Result<()> {
2265 if cursors.is_empty() {
2266 return Ok(());
2267 }
2268 let writes = read_state_write_ops(cursors)?;
2269 self.apply_writes(did, &writes).await
2270 }
2271
2272 // -- reader-facing record CRUD (the surface the web layer calls) ----------
2273 //
2274 // These are the typed convenience methods `web.rs` uses to manage a user's
2275 // feeds/folders/saved items *as records in their PDS*. They mirror the
2276 // create/list surface above but use the reader vocabulary
2277 // (add/remove/rename) and, for the `add_*` verbs, return the server-assigned
2278 // rkey so the caller can address the new record without a re-list. Ordering
2279 // is made deterministic where it matters (see [`list_subscriptions_sorted`]
2280 // etc.) so the server-rendered HTML is stable between reads.
2281
2282 // -- subscriptions -------------------------------------------------------
2283
2284 /// Add a subscription (subscribe to a feed) — `createRecord`, server-assigned
2285 /// `tid` rkey. Returns the new record's **rkey** so the web layer can offer
2286 /// unsubscribe/rename immediately.
2287 pub async fn add_subscription(
2288 &self,
2289 did: &str,
2290 sub: &crate::vetted::VettedSubscription,
2291 ) -> Result<String> {
2292 Ok(self.create_subscription(did, sub).await?.into_rkey())
2293 }
2294
2295 /// Remove a subscription (unsubscribe) by rkey — `deleteRecord`. Alias of
2296 /// [`delete_subscription`](Self::delete_subscription) in the reader vocabulary.
2297 pub async fn remove_subscription(&self, did: &str, rkey: &str) -> Result<()> {
2298 self.delete_subscription(did, rkey).await
2299 }
2300
2301 /// Update / rename a subscription in place at a known rkey — `putRecord`.
2302 ///
2303 /// The whole record is replaced (retitle, move to a folder, change the
2304 /// fetch hint …). Upsert semantics: it also creates the record if the rkey
2305 /// is somehow absent, so it is safe as a general "write this exact record".
2306 pub async fn update_subscription(
2307 &self,
2308 did: &str,
2309 rkey: &str,
2310 sub: &crate::vetted::VettedSubscription,
2311 swap_record: Option<&str>,
2312 ) -> Result<WriteResult> {
2313 self.put_record(did, lexicon::nsid::SUBSCRIPTION, rkey, sub, swap_record)
2314 .await
2315 }
2316
2317 /// List every subscription, **sorted deterministically** — by display title
2318 /// (case-insensitive), then feed URL, then rkey as the final tiebreaker — so
2319 /// the rendered feed list is stable across reads regardless of PDS return
2320 /// order. Untitled feeds sort by their URL.
2321 pub async fn list_subscriptions_sorted(
2322 &self,
2323 did: &str,
2324 ) -> Result<Vec<(String, Subscription)>> {
2325 let mut subs = self.list_subscriptions(did).await?;
2326 // The comparator is SHARED with the Rust-native client so the two
2327 // cannot order the list differently across the cutover.
2328 subs.sort_by(lexicon::sort::subscriptions);
2329 Ok(subs)
2330 }
2331
2332 /// Batch-add many subscriptions via `applyWrites` (chunked) — the OPML-import path.
2333 ///
2334 /// Each feed becomes one `create` op. Client-side monotonic `tid`
2335 /// rkeys are assigned so the batch is deterministic and the imported feeds
2336 /// keep OPML order (server-assigned tids would also be monotonic, but pinning
2337 /// them here makes the whole import reproducible and testable offline).
2338 /// Returns the assigned rkeys in input order.
2339 ///
2340 /// More than [`APPLY_WRITES_MAX_OPS`] feeds is more than one call, so the
2341 /// import can part-land: on an error, [`ApplyWritesIncomplete::of`] gives
2342 /// `landed`, and the first `landed` of `subs` are in the repo.
2343 pub async fn add_subscriptions_bulk(
2344 &self,
2345 did: &str,
2346 subs: &[crate::vetted::VettedSubscription],
2347 ) -> Result<Vec<String>> {
2348 let mut gen = TidGenerator::new();
2349 let mut rkeys = Vec::with_capacity(subs.len());
2350 let mut writes = Vec::with_capacity(subs.len());
2351 for sub in subs {
2352 let rkey = gen.next();
2353 writes.push(WriteOp::Create {
2354 collection: lexicon::nsid::SUBSCRIPTION.to_string(),
2355 rkey: Some(rkey.clone()),
2356 value: serde_json::to_value(sub)?,
2357 });
2358 rkeys.push(rkey);
2359 }
2360 self.apply_writes(did, &writes).await?;
2361 Ok(rkeys)
2362 }
2363
2364 // -- folders -------------------------------------------------------------
2365
2366 /// Add a folder — `createRecord`, server-assigned `tid` rkey. Returns the
2367 /// new folder's rkey (subscriptions reference it by its `at://` URI).
2368 pub async fn add_folder(&self, did: &str, folder: &Folder) -> Result<String> {
2369 Ok(self
2370 .create_record(did, lexicon::nsid::FOLDER, folder)
2371 .await?
2372 .into_rkey())
2373 }
2374
2375 /// Remove a folder by rkey — `deleteRecord`. (Subscriptions referencing it
2376 /// are left untouched; a dangling `folder` ref reads as "unfiled".)
2377 pub async fn remove_folder(&self, did: &str, rkey: &str) -> Result<()> {
2378 self.delete_record(did, lexicon::nsid::FOLDER, rkey).await
2379 }
2380
2381 /// Rename / update a folder in place at a known rkey — `putRecord`
2382 /// (rename, or change its `position` sort hint).
2383 ///
2384 /// Replaces the WHOLE record, so `folder` must be the record as read with
2385 /// only the intended change. `swap_record` is the CID it was read at: the
2386 /// PDS refuses the write with `InvalidSwap` if the record has moved since
2387 /// (#268). `None` writes unconditionally.
2388 pub async fn rename_folder(
2389 &self,
2390 did: &str,
2391 rkey: &str,
2392 folder: &Folder,
2393 swap_record: Option<&str>,
2394 ) -> Result<WriteResult> {
2395 self.put_record(did, lexicon::nsid::FOLDER, rkey, folder, swap_record)
2396 .await
2397 }
2398
2399 /// List every folder, **sorted deterministically** — by `position` (the
2400 /// lexicon's sort hint; unset sorts last), then name (case-insensitive),
2401 /// then rkey — so the sidebar order is stable.
2402 pub async fn list_folders_sorted(&self, did: &str) -> Result<Vec<(String, Folder)>> {
2403 let mut folders = self.list_folders(did).await?;
2404 folders.sort_by(lexicon::sort::folders);
2405 Ok(folders)
2406 }
2407
2408 // -- saved / starred -----------------------------------------------------
2409
2410 /// Add a saved (starred / save-for-later) entry — `createRecord`,
2411 /// server-assigned `tid` rkey. Returns the new record's rkey.
2412 pub async fn add_saved(&self, did: &str, saved: &crate::vetted::VettedSaved) -> Result<String> {
2413 Ok(self
2414 .create_record(did, lexicon::nsid::SAVED, saved)
2415 .await?
2416 .into_rkey())
2417 }
2418
2419 /// Remove a saved entry by rkey — `deleteRecord` (un-star).
2420 pub async fn remove_saved(&self, did: &str, rkey: &str) -> Result<()> {
2421 self.delete_record(did, lexicon::nsid::SAVED, rkey).await
2422 }
2423
2424 /// List every saved entry, **sorted deterministically** — newest first by
2425 /// `createdAt` (RFC-3339 sorts lexicographically), then rkey — so the
2426 /// "saved for later" list reads most-recent-first and is stable.
2427 pub async fn list_saved_sorted(&self, did: &str) -> Result<Vec<(String, Saved)>> {
2428 let mut saved = self.list_saved(did).await?;
2429 saved.sort_by(lexicon::sort::saved);
2430 Ok(saved)
2431 }
2432
2433 /// List a collection for `did` and parse each record's value into `T`,
2434 /// pairing it with its rkey. Unparseable records are skipped with a warning
2435 /// (forward-compat).
2436 async fn list_typed<T: DeserializeOwned>(
2437 &self,
2438 did: &str,
2439 collection: &str,
2440 ) -> Result<Vec<(String, T)>> {
2441 Ok(self
2442 .list_typed_with_cids(did, collection)
2443 .await?
2444 .into_iter()
2445 .map(|(rkey, _cid, value)| (rkey, value))
2446 .collect())
2447 }
2448
2449 /// [`list_typed`](Self::list_typed), keeping each record's CID (#149).
2450 async fn list_typed_with_cids<T: DeserializeOwned>(
2451 &self,
2452 did: &str,
2453 collection: &str,
2454 ) -> Result<Vec<(String, Option<String>, T)>> {
2455 let records = self.list_all_records(did, collection).await?;
2456 let mut out = Vec::with_capacity(records.len());
2457 for rec in records {
2458 let rkey = rec.rkey().unwrap_or_default().to_string();
2459 match rec.parse::<T>() {
2460 Ok(value) => out.push((rkey, rec.cid, value)),
2461 Err(e) => tracing::warn!(
2462 collection,
2463 uri = %rec.uri,
2464 error = %e,
2465 "skipping unparseable record in collection"
2466 ),
2467 }
2468 }
2469 Ok(out)
2470 }
2471}
2472
2473// ---------------------------------------------------------------------------
2474// applyWrites operations
2475// ---------------------------------------------------------------------------
2476
2477/// Build the `applyWrites` ops for a batch of dirty read-state cursors.
2478///
2479/// Each `(rkey, state, pds_created)` becomes a `#create` op (at the stable
2480/// feed-derived rkey) when the PDS record does NOT yet exist, and a `#update`
2481/// when it does. This is the crux of the first-flush fix: an `#update` on a
2482/// missing record errors, and `applyWrites` is atomic per-repo, so a single
2483/// not-yet-created cursor in the batch would drop the whole DID's flush. Emitting
2484/// a `create` for those makes a feed's first flush succeed while keeping every
2485/// op in ONE batch. Shared by both the sidecar and direct-PDS flush paths.
2486pub(crate) fn read_state_write_ops(cursors: &[(String, ReadState, bool)]) -> Result<Vec<WriteOp>> {
2487 cursors
2488 .iter()
2489 .map(|(rkey, state, pds_created)| {
2490 let value = serde_json::to_value(state)?;
2491 Ok(if *pds_created {
2492 WriteOp::Update {
2493 collection: lexicon::nsid::READ_STATE.to_string(),
2494 rkey: rkey.clone(),
2495 value,
2496 }
2497 } else {
2498 WriteOp::Create {
2499 collection: lexicon::nsid::READ_STATE.to_string(),
2500 rkey: Some(rkey.clone()),
2501 value,
2502 }
2503 })
2504 })
2505 .collect()
2506}
2507
2508/// Most writes one `com.atproto.repo.applyWrites` call may carry.
2509///
2510/// **The limit is the reference PDS's, not the lexicon's.** The lexicon's
2511/// `writes` array has no `maxLength` (checked against
2512/// `lexicons/com/atproto/repo/applyWrites.json` on bluesky-social/atproto
2513/// `main`, and against its history back to 2024-02); the cap is enforced by the
2514/// handler, `packages/pds/src/api/com/atproto/repo/applyWrites.ts`
2515/// (`if (writes.length > 200) throw new InvalidRequestError('Too many writes.
2516/// Max: 200')`, unchanged at a0c49d9). vlpds documents the same figure for its
2517/// commit coalescing. A PDS that allows more loses nothing by being sent 200.
2518pub const APPLY_WRITES_MAX_OPS: usize = 200;
2519
2520/// Most bytes of serialized writes one `applyWrites` call may carry.
2521///
2522/// **Sized to the smallest limit a deployed PDS is known to apply, not the
2523/// largest.** The reference PDS took `applyWrites` bodies up to its server-wide
2524/// `jsonLimit` of 150 KiB (`150 * 1024` in `packages/pds/src/index.ts`) until
2525/// atproto#4989 (2026-05-21) raised the record methods to `1_000_000` bytes.
2526/// Self-hosted PDSes run older releases for months, so the 150 KiB figure is
2527/// the live one for some readers. The other limits on this path are all
2528/// larger: the newer reference PDS's 1,000,000 bytes, the sidecar hop's Fastify
2529/// default `bodyLimit` of 1 MiB, and vlpds's coalesced commit of "up to 200
2530/// operations or 1 MB of record bytes".
2531///
2532/// 128 KiB leaves 22 KiB under 150 KiB for what this does not count — the
2533/// `{"repo": …, "writes": [ ]}` envelope, around a hundred bytes — and is
2534/// measured on the PDS-shaped op (`WriteOp::to_json`), which is also what the
2535/// sidecar forwards and is larger than the sidecar's own request shape.
2536///
2537/// What it costs, measured: 200 subscriptions with a title and site URL each
2538/// are ~76 KB, so an ordinary OPML import is still one call per 200 feeds. A
2539/// read-state cursor with a full 1,000-id set is ~10 KB with the store's
2540/// integer entry ids (~20 KB with both sets full), so a flush crosses the
2541/// bound at around a dozen full cursors — rare, since a cursor is compacted at
2542/// half the cap — and a refused body is worse than an extra round trip.
2543///
2544/// A single op larger than this is still sent, alone: it cannot be split, and
2545/// the PDS is the one to judge it.
2546pub const APPLY_WRITES_MAX_BYTES: usize = 128 * 1024;
2547
2548/// Split a batch into the consecutive ranges [`apply_writes_chunked`] sends,
2549/// each within [`APPLY_WRITES_MAX_OPS`] and [`APPLY_WRITES_MAX_BYTES`].
2550///
2551/// Ranges rather than slices so a caller can say WHICH writes a chunk held.
2552/// Order is preserved, every op is in exactly one range, and no range is empty;
2553/// an empty batch yields no ranges.
2554pub(crate) fn chunk_writes(writes: &[WriteOp]) -> Vec<std::ops::Range<usize>> {
2555 let mut chunks = Vec::new();
2556 let mut start = 0;
2557 let mut bytes = 0;
2558 for (i, op) in writes.iter().enumerate() {
2559 // +1 for the comma between array elements.
2560 let size = op.to_json().to_string().len() + 1;
2561 let held = i - start;
2562 if held > 0 && (held == APPLY_WRITES_MAX_OPS || bytes + size > APPLY_WRITES_MAX_BYTES) {
2563 chunks.push(start..i);
2564 start = i;
2565 bytes = 0;
2566 }
2567 bytes += size;
2568 }
2569 if start < writes.len() {
2570 chunks.push(start..writes.len());
2571 }
2572 chunks
2573}
2574
2575/// Send `writes` as consecutive `applyWrites` calls within the PDS's limits,
2576/// stopping at the first that fails.
2577///
2578/// **Every client's `apply_writes` goes through this**, so no caller can send
2579/// an oversized call: the OAuth client ([`crate::oauth::xrpc::Repo`]), the
2580/// sidecar client ([`SidecarClient`]) and the direct client ([`PdsClient`]).
2581/// `send` is that client's single-call primitive.
2582///
2583/// **The batch is no longer atomic.** `applyWrites` is atomic per CALL, so a
2584/// split batch can half-land. The contract a caller gets instead:
2585///
2586/// * chunks go in input order, one at a time, and a failure stops the run —
2587/// so what landed is always a PREFIX of `writes`;
2588/// * on failure the error carries an [`ApplyWritesIncomplete`] saying how long
2589/// that prefix is, how many writes after it are in doubt (the failed chunk:
2590/// atomic, so all or none, but a timeout cannot say which), and that the
2591/// rest were never sent.
2592///
2593/// Stopping rather than carrying on is what keeps that a prefix: sending chunk
2594/// 3 after chunk 2 failed would leave a gap no caller could describe in one
2595/// number, and a gap in an OPML import is feeds silently missing from the
2596/// middle of the list.
2597pub(crate) async fn apply_writes_chunked<'a, F, Fut>(
2598 writes: &'a [WriteOp],
2599 mut send: F,
2600) -> Result<()>
2601where
2602 F: FnMut(&'a [WriteOp]) -> Fut,
2603 Fut: std::future::Future<Output = Result<()>>,
2604{
2605 let chunks = chunk_writes(writes);
2606 let count = chunks.len();
2607 for (i, range) in chunks.into_iter().enumerate() {
2608 let (landed, in_doubt) = (range.start, range.len());
2609 if let Err(cause) = send(&writes[range]).await {
2610 return Err(ApplyWritesIncomplete {
2611 landed,
2612 in_doubt,
2613 total: writes.len(),
2614 chunk: i + 1,
2615 chunks: count,
2616 cause,
2617 }
2618 .into());
2619 }
2620 }
2621 Ok(())
2622}
2623
2624/// How far a chunked `applyWrites` got before a chunk failed — carried by the
2625/// error from every client's `apply_writes`, and so from `flush_read_states`
2626/// and `add_subscriptions_bulk`.
2627///
2628/// Read it with [`ApplyWritesIncomplete::of`]. In terms of the caller's own
2629/// input, in order:
2630///
2631/// * `writes[..landed]` were committed (each chunk was acknowledged);
2632/// * `writes[landed..landed + in_doubt]` were the failed call — usually not
2633/// committed, but a timeout or a lost response cannot rule it out;
2634/// * everything after was never sent.
2635///
2636/// Display is the underlying failure's message, with the progress appended
2637/// when the batch had more than one chunk, so a log line still names the
2638/// PDS's reason. The failure's own causes stay reachable through
2639/// [`anyhow::Error::chain`].
2640#[derive(Debug)]
2641pub struct ApplyWritesIncomplete {
2642 /// Writes committed, counted from the start of the input.
2643 pub landed: usize,
2644 /// Writes in the failed call, starting at `landed`.
2645 pub in_doubt: usize,
2646 /// Writes in the whole batch.
2647 pub total: usize,
2648 chunk: usize,
2649 chunks: usize,
2650 cause: anyhow::Error,
2651}
2652
2653impl ApplyWritesIncomplete {
2654 /// The progress record an `apply_writes` error carries, if it has one.
2655 pub fn of(err: &anyhow::Error) -> Option<&Self> {
2656 err.chain().find_map(|e| e.downcast_ref::<Self>())
2657 }
2658
2659 /// The failure that stopped the run, as the client reported it.
2660 pub fn cause(&self) -> &anyhow::Error {
2661 &self.cause
2662 }
2663}
2664
2665impl std::fmt::Display for ApplyWritesIncomplete {
2666 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
2667 write!(f, "{}", self.cause)?;
2668 if self.chunks > 1 {
2669 write!(
2670 f,
2671 " (applyWrites call {} of {}; {} of {} writes had landed)",
2672 self.chunk, self.chunks, self.landed, self.total
2673 )?;
2674 }
2675 Ok(())
2676 }
2677}
2678
2679impl std::error::Error for ApplyWritesIncomplete {
2680 // The cause's own Display is already in ours, so the chain continues from
2681 // ITS source — `{:#}` would otherwise print the PDS's message twice.
2682 fn source(&self) -> Option<&(dyn std::error::Error + 'static)> {
2683 self.cause.chain().nth(1)
2684 }
2685}
2686
2687/// One operation in a `PdsClient::apply_writes` batch.
2688///
2689/// Maps to the `com.atproto.repo.applyWrites` union of
2690/// `#create` / `#update` / `#delete`.
2691#[derive(Debug, Clone)]
2692pub enum WriteOp {
2693 /// Create a record (server-assigned rkey unless `rkey` is given).
2694 Create {
2695 /// The collection NSID.
2696 collection: String,
2697 /// Optional explicit rkey (`None` → server assigns a tid).
2698 rkey: Option<String>,
2699 /// The record body.
2700 value: Value,
2701 },
2702 /// Upsert a record at a known rkey (the read-state cursor case).
2703 Update {
2704 /// The collection NSID.
2705 collection: String,
2706 /// The rkey to write at.
2707 rkey: String,
2708 /// The record body.
2709 value: Value,
2710 },
2711 /// Delete a record by collection + rkey.
2712 Delete {
2713 /// The collection NSID.
2714 collection: String,
2715 /// The rkey to delete.
2716 rkey: String,
2717 },
2718}
2719
2720impl WriteOp {
2721 /// Render this op as the tagged JSON `com.atproto.repo.applyWrites` expects.
2722 ///
2723 /// `pub(crate)` so [`crate::oauth::xrpc`] can build the same batch body.
2724 /// Sharing the rendering rather than reimplementing it is what keeps the two
2725 /// clients wire-identical across the cutover.
2726 pub(crate) fn to_json(&self) -> Value {
2727 match self {
2728 WriteOp::Create {
2729 collection,
2730 rkey,
2731 value,
2732 } => {
2733 let mut op = json!({
2734 "$type": "com.atproto.repo.applyWrites#create",
2735 "collection": collection,
2736 "value": value,
2737 });
2738 if let Some(rkey) = rkey {
2739 op["rkey"] = json!(rkey);
2740 }
2741 op
2742 }
2743 WriteOp::Update {
2744 collection,
2745 rkey,
2746 value,
2747 } => json!({
2748 "$type": "com.atproto.repo.applyWrites#update",
2749 "collection": collection,
2750 "rkey": rkey,
2751 "value": value,
2752 }),
2753 WriteOp::Delete { collection, rkey } => json!({
2754 "$type": "com.atproto.repo.applyWrites#delete",
2755 "collection": collection,
2756 "rkey": rkey,
2757 }),
2758 }
2759 }
2760
2761 /// Render this op in the shape the OAuth sidecar's `/internal/repo`
2762 /// `applyWrites` expects: `{action, collection, rkey?, value?}` (the sidecar
2763 /// maps `action` → the `com.atproto.repo.applyWrites#<kind>` union member).
2764 fn to_sidecar_json(&self) -> Value {
2765 match self {
2766 WriteOp::Create {
2767 collection,
2768 rkey,
2769 value,
2770 } => {
2771 let mut op = json!({
2772 "action": "create",
2773 "collection": collection,
2774 "value": value,
2775 });
2776 if let Some(rkey) = rkey {
2777 op["rkey"] = json!(rkey);
2778 }
2779 op
2780 }
2781 WriteOp::Update {
2782 collection,
2783 rkey,
2784 value,
2785 } => json!({
2786 "action": "update",
2787 "collection": collection,
2788 "rkey": rkey,
2789 "value": value,
2790 }),
2791 WriteOp::Delete { collection, rkey } => json!({
2792 "action": "delete",
2793 "collection": collection,
2794 "rkey": rkey,
2795 }),
2796 }
2797 }
2798}
2799
2800// ---------------------------------------------------------------------------
2801// TID rkeys (client-assigned, sortable, deterministic within a batch)
2802// ---------------------------------------------------------------------------
2803
2804/// The atproto base32-sortable alphabet (`s32`) — the digits/letters, minus the
2805/// ambiguous set, in **ascending** order so a bytewise string compare of two
2806/// TIDs matches their timestamp order.
2807const S32_ALPHABET: &[u8; 32] = b"234567abcdefghijklmnopqrstuvwxyz";
2808
2809/// A monotonic generator of atproto **TID** record keys.
2810///
2811/// A TID is a 13-char `s32`-encoded 64-bit integer: a 53-bit microsecond
2812/// timestamp in the high bits and a 10-bit "clock id" in the low bits (the top
2813/// bit is always 0). Encoded in the ascending `s32` alphabet, TIDs sort
2814/// lexicographically in creation order — which is exactly what we want for a
2815/// batched OPML import: assigning the rkeys ourselves keeps the imported feeds
2816/// in input order and makes [`add_subscriptions_bulk`](SidecarClient::add_subscriptions_bulk)
2817/// fully reproducible/testable without a live PDS.
2818///
2819/// Monotonicity within one generator is guaranteed by tracking the last value
2820/// and bumping to `last + 1` if the clock hasn't advanced — so a burst of
2821/// same-microsecond calls still yields strictly increasing, ordered rkeys.
2822pub(crate) struct TidGenerator {
2823 /// The last raw 64-bit TID value emitted (0 = none yet).
2824 last: u64,
2825 /// The low-10-bit clock id, randomized once per generator to avoid
2826 /// cross-instance collisions on the same microsecond.
2827 clock_id: u64,
2828}
2829
2830impl TidGenerator {
2831 /// A fresh generator with a per-instance clock id derived from the current
2832 /// nanosecond clock (no extra deps; uniqueness only needs to hold within a
2833 /// single import batch, and the timestamp bits carry the ordering).
2834 pub(crate) fn new() -> Self {
2835 let nanos = std::time::SystemTime::now()
2836 .duration_since(std::time::UNIX_EPOCH)
2837 .map(|d| d.subsec_nanos() as u64)
2838 .unwrap_or(0);
2839 Self {
2840 last: 0,
2841 clock_id: nanos & 0x3ff,
2842 }
2843 }
2844
2845 /// The next monotonic TID rkey (13 `s32` chars).
2846 pub(crate) fn next(&mut self) -> String {
2847 let micros = std::time::SystemTime::now()
2848 .duration_since(std::time::UNIX_EPOCH)
2849 .map(|d| d.as_micros() as u64)
2850 .unwrap_or(0);
2851 // Timestamp in bits 63..10 (top bit stays 0), clock id in bits 9..0.
2852 let mut raw = ((micros & 0x001f_ffff_ffff_ffff) << 10) | self.clock_id;
2853 if raw <= self.last {
2854 raw = self.last + 1;
2855 }
2856 self.last = raw;
2857 encode_s32_tid(raw)
2858 }
2859}
2860
2861/// Encode a 64-bit TID value as a 13-char big-endian `s32` string.
2862fn encode_s32_tid(mut v: u64) -> String {
2863 let mut buf = [0u8; 13];
2864 for slot in buf.iter_mut().rev() {
2865 *slot = S32_ALPHABET[(v & 0x1f) as usize];
2866 v >>= 5;
2867 }
2868 // 13 * 5 = 65 bits cover the 64-bit value; the leading char carries bits
2869 // 64..60, and bit 64 does not exist in a `u64` while bit 63 is always 0 in
2870 // a real TID, so the leading char is always one of the alphabet's first
2871 // eight symbols. Between 2005-09-05 and 2041-05-10 it is the second one,
2872 // which is why real TIDs all begin with `3`.
2873 String::from_utf8(buf.to_vec()).unwrap_or_default()
2874}
2875
2876/// The earliest instant a real TID can encode: 2020-01-01T00:00:00Z, in
2877/// microseconds.
2878///
2879/// atproto did not exist before this, so a "TID" decoding to earlier is a record
2880/// key that merely *looks* like one.
2881///
2882/// **This bound catches only the slugs that fall outside the window, and that
2883/// is a minority of them.** 13 lowercase alphanumerics is an ordinary slug
2884/// shape and also a valid `s32` value, and one beginning `3` decodes into the
2885/// last few years as readily as a real record key does: `3hoursinparis` reads
2886/// as 2020-11-24, `3ideasforjune` as 2021-08-12. Nothing in the string
2887/// distinguishes them — telling a slug from a TID would mean asking the PDS
2888/// when the record was written, which the listing does not report.
2889///
2890/// What the window does buy is that a mis-read date is always an ordinary past
2891/// instant rather than an unsweepable future one. That is worth having and it
2892/// is *not* harmless: a slug reading as 2020 is older than any realistic
2893/// retention window, so the row is swept, re-listed on the next poll, and
2894/// arrives unread again — the cycle this dating work narrows but does not
2895/// close. Refusing to insert what is already past the floor is what closes it,
2896/// for a mis-read slug and a genuine archive alike, and that belongs with the
2897/// retention floor rather than here.
2898const TID_FLOOR_MICROS: i64 = 1_577_836_800_000_000;
2899
2900/// How far ahead of our own clock a timestamp someone else authored may be and
2901/// still be believed.
2902///
2903/// A PDS a second or two fast would otherwise leave a brand-new document
2904/// undated until the following poll, and an undated row is the least visible
2905/// one in the reading list. Well under any interval that matters to retention
2906/// or the per-feed cap.
2907///
2908/// **Both date sources use it.** It began as a TID-only allowance, which left a
2909/// stated `publishedAt` judged against a bare `now` while the record key two
2910/// lines below got five minutes — the same clock, two different answers, for no
2911/// reason either comment could give.
2912pub(crate) const CLOCK_SKEW_GRACE_SECS: i64 = 300;
2913
2914/// Decode a 13-char `s32` TID rkey back to its raw 64-bit value.
2915///
2916/// The exact inverse of [`encode_s32_tid`] over the values a TID can hold.
2917///
2918/// `None` for anything that is not a 13-character `s32` value: wrong length, a
2919/// character outside the alphabet, or a value whose top bit is set. That last
2920/// rejection is stricter than the TID syntax regex, which admits leading `c`
2921/// through `j`; the spec's separate rule that the high bit is always 0 is the
2922/// one enforced here, and it keeps every decoded value inside the range
2923/// [`tid_timestamp`] can shift without loss.
2924///
2925/// **This does not decide whether the string is a TID**, only whether it is a
2926/// number. Thirteen lowercase alphanumerics is also an ordinary slug, and a
2927/// slug decodes as readily as a record key does. Refusing an implausible
2928/// instant is [`tid_timestamp`]'s job, and it is where that case is caught.
2929pub(crate) fn decode_s32_tid(rkey: &str) -> Option<u64> {
2930 if rkey.len() != 13 {
2931 return None;
2932 }
2933 let mut v: u64 = 0;
2934 for b in rkey.bytes() {
2935 let digit = S32_ALPHABET.iter().position(|c| *c == b)? as u64;
2936 // `checked_*` rather than shifting: 13 chars carry 65 bits, so the
2937 // largest 13-char string overflows a `u64` and must read as "not a
2938 // TID" instead of wrapping to a plausible-looking value.
2939 v = v.checked_mul(32)?.checked_add(digit)?;
2940 }
2941 (v >> 63 == 0).then_some(v)
2942}
2943
2944/// The instant a TID rkey encodes, or `None` if the rkey is not a plausible
2945/// TID.
2946///
2947/// **Bounded at both ends on purpose.** A TID's timestamp is minted from the
2948/// writer's clock, so one decoding far into the future is either a broken clock
2949/// or a slug that happens to be 13 `s32` characters; one decoding to before
2950/// [`TID_FLOOR_MICROS`] predates atproto. Neither is a date worth trusting, and
2951/// the caller's fallback for "no date" is safer than a wrong one.
2952///
2953/// The bounds are not a slug detector — see [`TID_FLOOR_MICROS`] for why they
2954/// cannot be, and for what they do guarantee instead.
2955pub(crate) fn tid_timestamp(rkey: &str) -> Option<chrono::DateTime<chrono::Utc>> {
2956 // The low 10 bits are the clock id; the rest is microseconds since the
2957 // epoch, and clearing bit 63 above bounds it well inside `i64`.
2958 let micros = i64::try_from(decode_s32_tid(rkey)? >> 10).ok()?;
2959 if micros < TID_FLOOR_MICROS {
2960 return None;
2961 }
2962 let at = chrono::DateTime::from_timestamp_micros(micros)?;
2963 let ceiling = chrono::Utc::now() + chrono::Duration::seconds(CLOCK_SKEW_GRACE_SECS);
2964 (at <= ceiling).then_some(at)
2965}
2966
2967// ---------------------------------------------------------------------------
2968// XRPC error helper
2969// ---------------------------------------------------------------------------
2970
2971/// Minimal percent-encoding for a query-string component.
2972///
2973/// Encodes everything outside the RFC 3986 unreserved set, which covers the
2974/// values FeatherReader passes (DIDs like `did:plc:…`, NSIDs, opaque cursors,
2975/// handles) without pulling in the optional reqwest `url`/`query` feature.
2976///
2977/// `pub(crate)` so [`crate::network`] builds its relay query strings the same
2978/// way rather than keeping a second copy of the escape table.
2979pub(crate) fn urlencode(s: &str) -> String {
2980 let mut out = String::with_capacity(s.len());
2981 for b in s.bytes() {
2982 match b {
2983 b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => {
2984 out.push(b as char)
2985 }
2986 _ => out.push_str(&format!("%{b:02X}")),
2987 }
2988 }
2989 out
2990}
2991
2992/// The most nodes a `listRecords` body may ask us to build.
2993///
2994/// **A bound on the parse, checked before the parse.** Every other limit here is
2995/// consulted after `serde_json` has already materialised the page, which cannot
2996/// prevent the allocation it exists to prevent: one 8 MB response of `{"":0}`
2997/// objects was measured retaining 824 MB, a 98x wire-to-heap amplification, on a
2998/// 512 MB box. A record cap does not see it — the page holds one record. A page
2999/// cap does not see it — there is one request. A byte budget does not see it
3000/// until the memory is already spent.
3001///
3002/// **640 000, measured — not two million, which this file's own arithmetic
3003/// already contradicted.** An earlier version reasoned "32 bytes a node plus
3004/// slack, so two million is about 128 MB". That is the model `json_bytes` two
3005/// hundred lines above explicitly rejects: it charges `MAP_NODE` + `MAP_ENTRY` +
3006/// `SLOT` = 680 bytes for a single-entry object, which costs three counted
3007/// characters. Measured against a counting allocator, the worst shape reaches
3008/// **210 bytes per counted character**, so two million admitted **400 MB** — a
3009/// bound that let through more than the attack it was written to stop, and that
3010/// the walk's own byte budget then refused a step later.
3011///
3012/// 640 000 x 210 B is about 128 MiB, which is the figure the walk budget uses
3013/// and the one this claims.
3014///
3015/// **Re-measured when a review put the worst shape at 221 B; it does not
3016/// reproduce.** Sweeping nesting depths 10, 50, 100 and 120 against the same
3017/// counting allocator, the worst is 210.6 B per counted character (at depth 120)
3018/// and peak equals retained — `serde_json` overshoots by nothing measurable
3019/// while it builds. Depth cannot be pushed further to raise the ratio, either:
3020/// `serde_json`'s own recursion limit of 128 refuses a deeper body outright,
3021/// before this guard would even matter.
3022///
3023/// **The floor is real traffic, not comfort.** The densest legitimate page is a
3024/// full `readState` listing — 100 records each carrying two arrays of
3025/// [`crate::lexicon::ReadState::MAX_IDS`] ids — which counts 403 003. So the cap
3026/// sits above the densest page the lexicons permit, and a test holds it there.
3027///
3028/// Ordinary traffic is nowhere near either number: a page of 100 real-sized
3029/// standard.site documents (seven fields, a 15 kB `textContent`, 1.5 MB on the
3030/// wire) counts **4 003**. An earlier version of this line said 40 000, which was
3031/// wrong by an order of magnitude in the direction that makes the cap look tighter
3032/// than it is; the test that was supposed to hold it served a single record and
3033/// would have passed with the cap set to 1 000.
3034pub(crate) const MAX_LIST_STRUCTURAL_CHARS: usize = 640_000;
3035
3036/// How many nodes `body` would parse into, to within one, without parsing it.
3037///
3038/// **A lower bound, despite what an earlier name said.** `[1,2,3]` counts three
3039/// — one `[` and two commas — and builds four values. The deficit is never more
3040/// than one (verified exhaustively over every body of length 1-5 from a JSON
3041/// alphabet), because every node but the outermost is introduced by one of the
3042/// characters counted here. That is the direction a guard needs: it can
3043/// under-count by one and still refuse everything it must.
3044///
3045/// Counts the structural characters that introduce a value — `{`, `[`, `,`, `:`
3046/// — **outside strings**, which is what makes this sound: a node cannot appear
3047/// without one, and a string's contents cannot invent one. Skipping strings is
3048/// the whole difficulty; counting naively would refuse a legitimate article that
3049/// happens to contain a million commas.
3050pub(crate) fn count_structural_chars(body: &[u8]) -> usize {
3051 let mut nodes = 0usize;
3052 let mut in_string = false;
3053 let mut escaped = false;
3054 for &b in body {
3055 if in_string {
3056 // `\"` stays inside the string; `\\` does not escape the quote that
3057 // follows it. Getting this pair wrong makes the scan count a whole
3058 // document as structure, or none of it.
3059 if escaped {
3060 escaped = false;
3061 } else if b == b'\\' {
3062 escaped = true;
3063 } else if b == b'"' {
3064 in_string = false;
3065 }
3066 continue;
3067 }
3068 match b {
3069 // A string is a node, and everything inside it is not.
3070 b'"' => {
3071 in_string = true;
3072 nodes += 1;
3073 }
3074 b'{' | b'[' | b',' | b':' => nodes += 1,
3075 _ => {}
3076 }
3077 }
3078 nodes
3079}
3080
3081/// Refuse a body carrying more JSON structure than
3082/// [`MAX_LIST_STRUCTURAL_CHARS`].
3083pub(crate) fn refuse_a_structure_explosion(body: &[u8], what: &str) -> Result<()> {
3084 let counted = count_structural_chars(body);
3085 anyhow::ensure!(
3086 counted <= MAX_LIST_STRUCTURAL_CHARS,
3087 "{what} counts at least {counted} structural characters, over the \
3088 {MAX_LIST_STRUCTURAL_CHARS} cap — refusing before parsing it"
3089 );
3090 Ok(())
3091}
3092
3093/// Parse a `listRecords` body, refusing an error envelope that arrived on a 2xx.
3094///
3095/// Some PDS implementations answer 200 for application failures, and the status
3096/// check in the caller cannot see those. Without the guard, `{"error","message"}`
3097/// deserialises as a page with no records — so a walk over a stranger's
3098/// collection returns a healthy, empty result in place of an error, and for the
3099/// walk that feeds `replace_sub_refs` that is revoked access rather than an empty
3100/// repo. The guard itself lives in [`page_from_body`], which every client shares.
3101pub(crate) fn parse_list_records(body: &[u8]) -> Result<ListRecordsResponse> {
3102 // **An empty body is the "unexpected body" case, not a parse error.** Reading
3103 // bytes reaches it as "EOF while parsing", where the OAuth client used to
3104 // reach it as "no records field" (its `send` mapped an empty 2xx to
3105 // `Value::Null`) and the direct client reached it as "EOF" too. Refused
3106 // either way, so this is a unification rather than a preservation — nothing
3107 // outside the tests matches on the text, and `resolve_subscriptions` fails
3108 // closed on any `Err`. It is for whoever reads the log.
3109 if body.is_empty() {
3110 anyhow::bail!("listRecords returned no records field (empty or unexpected body)");
3111 }
3112 refuse_a_structure_explosion(body, "the listRecords body")?;
3113 let parsed: ListRecordsBody =
3114 serde_json::from_slice(body).context("parsing listRecords response")?;
3115 let mut page = page_from_body(parsed)?;
3116 page.wire_bytes = body.len();
3117 Ok(page)
3118}
3119
3120/// Apply both invariants to an already-deserialised body.
3121///
3122/// **The one place the guards live, for all three clients.** They were added a
3123/// client at a time twice over, which is the whole reason a shared function
3124/// exists; splitting the sidecar onto a different route would have started that
3125/// again, so it deserialises into this same struct.
3126fn page_from_body(parsed: ListRecordsBody) -> Result<ListRecordsResponse> {
3127 if let Some(error) = parsed.error.as_ref().and_then(envelope_error_name) {
3128 let message = parsed
3129 .message
3130 .as_ref()
3131 .and_then(Value::as_str)
3132 .map(|m| format!(" — {}", truncate_for_message(m)))
3133 .unwrap_or_default();
3134 anyhow::bail!("PDS answered 2xx with an error envelope: {error}{message}");
3135 }
3136 let entries = parsed.records.ok_or_else(|| {
3137 anyhow::anyhow!("listRecords returned no records field (empty or unexpected body)")
3138 })?;
3139 let mut records = Vec::with_capacity(entries.len());
3140 let mut malformed = 0;
3141 for entry in entries {
3142 match entry {
3143 MaybeRecord::Record(r) => records.push(r),
3144 MaybeRecord::Malformed(_) => malformed += 1,
3145 }
3146 }
3147 Ok(ListRecordsResponse {
3148 records,
3149 cursor: parsed.cursor,
3150 malformed,
3151 wire_bytes: 0,
3152 })
3153}
3154
3155/// The wire shape of a `listRecords` body, read in **one** pass.
3156///
3157/// **Parsing to `Value` and then into the struct materialises the page twice.**
3158/// `serde_json::from_value` rebuilds rather than moves, so an 8 MB response was
3159/// measured holding both copies at once — a peak of roughly double the retained
3160/// size, reached before any accounting the caller does, which is why no budget
3161/// charged after the parse can cover it.
3162///
3163/// The two invariants that used to live on a `Value` are
3164/// expressed here as fields instead of lookups, and mean exactly what they did:
3165/// an `error` present on a 2xx is a failure, not an empty page, and `records`
3166/// ABSENT is not `records` empty.
3167#[derive(Debug, Default)]
3168struct ListRecordsBody {
3169 /// A `Value`, not a `String`. Typing it as a string made
3170 /// `{"error":404,"records":[]}` fail as "invalid type: integer" rather than
3171 /// as an envelope — the wrong reason for the exact shape the guard exists
3172 /// for, and the guard's whole point is that this distinction is load-bearing.
3173 error: Option<Value>,
3174 /// Likewise, and for a duller reason: `message` carries no security role,
3175 /// and typing it as a string made a PDS that stamps a non-string one onto an
3176 /// otherwise good page of a thousand records fail the entire listing.
3177 message: Option<Value>,
3178 /// `None` means the field was absent — what a proxy makes of an empty or
3179 /// unexpected upstream body. `Some(vec![])` is a genuine empty page.
3180 records: Option<Vec<MaybeRecord>>,
3181 cursor: Option<String>,
3182}
3183
3184/// One element of a `listRecords` page: a record, or something that is not one.
3185///
3186/// **Parsed per record, so one malformed envelope costs that record and not
3187/// the page** (#177). `RecordEntry.uri` is required, and the page used to be
3188/// parsed in one `from_value`, so a single `{"cid":…,"value":{}}` failed every
3189/// record beside it. Whoever reads the page decides what a skipped record
3190/// means: a stranger's publication skips it, a reader's own repo refuses.
3191#[derive(Debug, Deserialize)]
3192#[serde(untagged)]
3193enum MaybeRecord {
3194 Record(RecordEntry),
3195 Malformed(serde::de::IgnoredAny),
3196}
3197
3198/// Refuse a page that skipped records, for a walk that must not drop any.
3199///
3200/// Every walk whose result reaches `store::replace_sub_refs` calls this: a
3201/// record left out there is a subscription silently removed.
3202/// What a walk does with a record whose envelope is malformed (#177).
3203#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3204enum OnMalformed {
3205 /// Refuse the walk with [`MalformedRecords`]: the result reaches
3206 /// `replace_sub_refs`, where a skipped record is a dropped subscription.
3207 Refuse,
3208 /// Skip and count it: a stranger's repo, where one bad record must not
3209 /// stall everything beside it.
3210 Skip,
3211}
3212
3213pub(crate) fn refuse_malformed(page: &ListRecordsResponse, collection: &str) -> Result<()> {
3214 if page.malformed > 0 {
3215 return Err(MalformedRecords {
3216 collection: collection.to_string(),
3217 count: page.malformed,
3218 }
3219 .into());
3220 }
3221 Ok(())
3222}
3223
3224/// **Hand-written, because the derive accepts a listing that is not an object.**
3225///
3226/// serde's derived `Deserialize` takes a struct in POSITIONAL form as well as
3227/// map form, so with every field defaulted the fourteen bytes `[null,null,[]]`
3228/// bound `records` to an empty vector and read as a healthy page — on all three
3229/// clients, and the `Value` route this replaced refused it, because
3230/// `Value::Array::get("records")` is always `None`. Neither the envelope guard
3231/// nor the duplicated-key refusal can fire on a body with no keys at all, so one
3232/// short array defeated every protection here at once and reached
3233/// `replace_sub_refs`, which deletes the reader's whole subscription projection.
3234///
3235/// **Unknown fields are read, not skipped.** `IgnoredAny` does not validate what
3236/// it skips, so `{"records":[],"x":"<invalid utf-8>"}` — not valid JSON at all —
3237/// also read as a healthy empty page where the `Value` route refused it. Reading
3238/// the value into a `Value` and dropping it costs an allocation on a field nobody
3239/// wants, and buys back the validation.
3240impl<'de> Deserialize<'de> for ListRecordsBody {
3241 fn deserialize<D: serde::Deserializer<'de>>(d: D) -> std::result::Result<Self, D::Error> {
3242 struct AsMap;
3243 impl<'de> serde::de::Visitor<'de> for AsMap {
3244 type Value = ListRecordsBody;
3245 fn expecting(&self, f: &mut std::fmt::Formatter) -> std::fmt::Result {
3246 f.write_str("a listRecords object")
3247 }
3248 fn visit_map<M: serde::de::MapAccess<'de>>(
3249 self,
3250 mut map: M,
3251 ) -> std::result::Result<ListRecordsBody, M::Error> {
3252 use serde::de::Error;
3253 let mut out = ListRecordsBody::default();
3254 let (mut error, mut message, mut records, mut cursor) =
3255 (false, false, false, false);
3256 while let Some(key) = map.next_key::<String>()? {
3257 let seen = match key.as_str() {
3258 "error" => std::mem::replace(&mut error, true),
3259 "message" => std::mem::replace(&mut message, true),
3260 "records" => std::mem::replace(&mut records, true),
3261 "cursor" => std::mem::replace(&mut cursor, true),
3262 _ => false,
3263 };
3264 if seen {
3265 // A repeated key is last-wins in a `Value`, which is how
3266 // a smuggled second, empty `records` array read as a
3267 // successful page. Refused here.
3268 return Err(M::Error::duplicate_field(match key.as_str() {
3269 "error" => "error",
3270 "message" => "message",
3271 "records" => "records",
3272 _ => "cursor",
3273 }));
3274 }
3275 match key.as_str() {
3276 "error" => out.error = Some(map.next_value()?),
3277 "message" => out.message = Some(map.next_value()?),
3278 "records" => out.records = Some(map.next_value()?),
3279 "cursor" => out.cursor = map.next_value()?,
3280 _ => {
3281 let _validated: Value = map.next_value()?;
3282 }
3283 }
3284 }
3285 Ok(out)
3286 }
3287 }
3288 d.deserialize_map(AsMap)
3289 }
3290}
3291
3292/// Refuse an atproto error envelope that arrived on a 2xx.
3293///
3294/// **No listing reaches this any more — [`page_from_body`] is the one every
3295/// client shares.** It survives for the WRITE paths, where the response is still
3296/// a `Value`: `deleteRecord` and `applyWrites` on both live clients.
3297///
3298/// The history is worth keeping, because it is why a shared function exists at
3299/// all. Each client used to take `records` off the JSON its own way — the live
3300/// one with `unwrap_or(Array([]))`, the sidecar through a defaulted `Value` — and
3301/// each turned `200 {"error": …}` into `Ok(empty)`. That is not the fail-closed
3302/// branch in `web::resolve_subscriptions`: `sync_sub_refs` wrote the empty set
3303/// and `replace_sub_refs` DELETEd the DID's entire `sub_ref` projection. One bad
3304/// response revoked a reader's access to every feed they had. The guard was added
3305/// to one client at a time, twice, which is the drift a single function prevents
3306/// — and why the listing guard now lives in exactly one place rather than here.
3307pub(crate) fn reject_error_envelope(value: &Value) -> Result<()> {
3308 let Some(error) = value.get("error").and_then(envelope_error_name) else {
3309 return Ok(());
3310 };
3311 let message = value
3312 .get("message")
3313 .and_then(Value::as_str)
3314 .map(|m| format!(" — {}", truncate_for_message(m)))
3315 .unwrap_or_default();
3316 anyhow::bail!("PDS answered 2xx with an error envelope: {error}{message}")
3317}
3318
3319/// The name in an `error` field, or `None` when the field does not denote one.
3320///
3321/// **Whatever its type.** Keying on `as_str` meant a PDS answering
3322/// `{"error":404,"records":[]}` — or `{}`, or `[]` — passed the guard and read as
3323/// a healthy empty page, the shape that makes `replace_sub_refs` delete every
3324/// `sub_ref` a reader has. A non-string `error` is not a well-formed envelope,
3325/// but it is certainly not a successful listing either.
3326///
3327/// **Except the four spellings of "no error".** Absent and `null` are what an
3328/// ordinary listing carries; `false` and `0` are a convention proxies use, and
3329/// treating those as envelopes turns a good page of a thousand records into a
3330/// hard refusal, which on these walks means the reader's sidebar degrades to a
3331/// stale projection on every request.
3332///
3333/// **Bounded.** The name reaches a `warn!` that also logs the DID, and the value
3334/// is attacker-chosen: a PDS answering with hundreds of kilobytes under `error`
3335/// would otherwise put all of it in the log and allocate another copy, in code
3336/// whose purpose is cutting peak allocation.
3337fn envelope_error_name(error: &Value) -> Option<String> {
3338 match error {
3339 Value::Null | Value::Bool(false) => None,
3340 // Integer zero only. `as_f64() == Some(0.0)` also matched `-0`, `0.0`
3341 // and anything that underflows, so `1e-400` was "no error".
3342 Value::Number(n) if n.as_i64() == Some(0) || n.as_u64() == Some(0) => None,
3343
3344 Value::String(s) => Some(truncate_for_message(s)),
3345 // **The type, not the value.** `to_string()` would serialise the whole
3346 // attacker-chosen subtree before truncating it, allocating a full extra
3347 // copy of up to the body cap — in code whose purpose is cutting peak
3348 // allocation. A non-string `error` is malformed, so its contents tell a
3349 // reader nothing its shape does not.
3350 Value::Bool(_) => Some("<non-string error: bool>".to_string()),
3351 Value::Number(_) => Some("<non-string error: number>".to_string()),
3352 Value::Array(_) => Some("<non-string error: array>".to_string()),
3353 Value::Object(_) => Some("<non-string error: object>".to_string()),
3354 }
3355}
3356
3357/// Cap a string destined for an error message at a readable length.
3358fn truncate_for_message(s: &str) -> String {
3359 /// **Bytes, not characters.** A log line is bytes, and counting characters
3360 /// let astral-plane code points render four times the intended bound.
3361 const MAX_BYTES: usize = 120;
3362 if s.len() <= MAX_BYTES {
3363 return s.to_string();
3364 }
3365 let cut = s
3366 .char_indices()
3367 .map(|(i, _)| i)
3368 .take_while(|i| *i <= MAX_BYTES)
3369 .last()
3370 .unwrap_or(0);
3371 format!("{}… ({} bytes)", &s[..cut], s.len())
3372}
3373
3374/// The atproto XRPC error envelope body: `{"error": "...", "message": "..."}`.
3375#[derive(Debug, Deserialize)]
3376struct XrpcErrorBody {
3377 #[serde(default)]
3378 error: Option<String>,
3379 #[serde(default)]
3380 message: Option<String>,
3381}
3382
3383/// Consume a non-2xx response into a typed [`AtProtoError::Xrpc`], parsing the
3384/// atproto error envelope when present (falling back to `"Unknown"`).
3385///
3386/// The body is read through [`crate::net::read_capped`], **not** `resp.json()`.
3387/// Every guarded call caps its success body; routing the error body through
3388/// `resp.json()` would have left a hole exactly where the hostile-PDS threat
3389/// model points — reqwest decompresses gzip before deserialising, so a `400`
3390/// carrying a decompression bomb was an unbounded allocation on a 512 MB box.
3391/// A body we cannot read (over-cap, transport error) degrades to `"Unknown"`,
3392/// which is the same fallback an unparseable envelope already took.
3393async fn xrpc_error_from(resp: reqwest::Response) -> AtProtoError {
3394 let status = resp.status();
3395 let (error, message) = match crate::net::read_capped(resp).await {
3396 Ok(raw) => match serde_json::from_slice::<XrpcErrorBody>(&raw) {
3397 Ok(body) => (
3398 body.error.unwrap_or_else(|| "Unknown".to_string()),
3399 body.message,
3400 ),
3401 Err(_) => ("Unknown".to_string(), None),
3402 },
3403 Err(_) => ("Unknown".to_string(), None),
3404 };
3405 AtProtoError::Xrpc {
3406 status,
3407 error,
3408 message,
3409 }
3410}
3411
3412// ---------------------------------------------------------------------------
3413// Tests — record (de)serialization against a repo listRecords response shape.
3414// No network.
3415// ---------------------------------------------------------------------------
3416
3417#[cfg(test)]
3418pub(crate) mod tests {
3419 use super::*;
3420
3421 /// **Regression (v0.2.8 review).** Every guarded call caps its *success*
3422 /// body via `read_capped`, but the non-2xx branch went through
3423 /// `resp.json::<XrpcErrorBody>()` — unbounded, and with reqwest's gzip
3424 /// decompression in front of it. That left a hole precisely where the
3425 /// module's own threat model points: a hostile or DNS-rebound PDS answers
3426 /// `400` with a decompression bomb and gets an unbounded allocation on a
3427 /// 512 MB box. Both this PR's review passes checked the success path and
3428 /// walked past the error path, so the cap is asserted here explicitly.
3429 ///
3430 /// Fetched directly rather than through the guard, which rightly refuses
3431 /// loopback — the same reason `net::tests::read_capped_rejects_over_cap_body`
3432 /// bypasses it. The stub answers 200; `xrpc_error_from` reads the status only
3433 /// to record it, so the body handling under test is identical.
3434 #[tokio::test]
3435 async fn xrpc_error_body_is_capped() {
3436 // A syntactically VALID envelope, one byte past the cap. If the body were
3437 // parsed unbounded this would deserialize and yield "TooBig"; capped, it
3438 // is refused unread and degrades to the "Unknown" fallback.
3439 let filler = "x".repeat(crate::net::MAX_BODY_BYTES);
3440 let big = format!(r#"{{"error":"TooBig","message":"{filler}"}}"#).into_bytes();
3441 assert!(big.len() > crate::net::MAX_BODY_BYTES);
3442
3443 let base = crate::net::tests::serve_body(big).await;
3444 let resp = reqwest::Client::builder()
3445 .build()
3446 .unwrap()
3447 .get(&base)
3448 .send()
3449 .await
3450 .unwrap();
3451
3452 match xrpc_error_from(resp).await {
3453 AtProtoError::Xrpc { error, message, .. } => {
3454 assert_eq!(error, "Unknown", "an over-cap error body must not parse");
3455 assert!(message.is_none());
3456 }
3457 other => panic!("expected Xrpc, got {other:?}"),
3458 }
3459 }
3460
3461 /// The other half: a normal-sized envelope still parses, so capping the
3462 /// error path did not cost the diagnostics it exists to provide.
3463 #[tokio::test]
3464 async fn xrpc_error_body_within_the_cap_still_parses() {
3465 let base = crate::net::tests::serve_body(
3466 br#"{"error":"InvalidRequest","message":"bad rkey"}"#.to_vec(),
3467 )
3468 .await;
3469 let resp = reqwest::Client::builder()
3470 .build()
3471 .unwrap()
3472 .get(&base)
3473 .send()
3474 .await
3475 .unwrap();
3476
3477 match xrpc_error_from(resp).await {
3478 AtProtoError::Xrpc { error, message, .. } => {
3479 assert_eq!(error, "InvalidRequest");
3480 assert_eq!(message.as_deref(), Some("bad rkey"));
3481 }
3482 other => panic!("expected Xrpc, got {other:?}"),
3483 }
3484 }
3485
3486 /// A realistic `com.atproto.repo.listRecords` response for the subscription
3487 /// collection, as a PDS returns it — the envelope wraps each record in
3488 /// `{uri, cid, value}` and the record `value` carries its `$type`.
3489 fn subscription_list_json() -> Value {
3490 json!({
3491 "records": [
3492 {
3493 "uri": "at://did:plc:abc123/community.lexicon.rss.subscription/3ksub0001",
3494 "cid": "bafyreisubone",
3495 "value": {
3496 "$type": "community.lexicon.rss.subscription",
3497 "url": "https://example.com/feed.xml",
3498 "title": "Example Blog",
3499 "siteUrl": "https://example.com/",
3500 "fetchHint": "hourly",
3501 "createdAt": "2026-07-12T00:00:00.000Z"
3502 }
3503 },
3504 {
3505 "uri": "at://did:plc:abc123/community.lexicon.rss.subscription/3ksub0002",
3506 "cid": "bafyreisubtwo",
3507 "value": {
3508 "$type": "community.lexicon.rss.subscription",
3509 "url": "https://blog.example.org/atom.xml",
3510 "createdAt": "2026-07-11T12:00:00.000Z"
3511 }
3512 }
3513 ],
3514 "cursor": "3ksub0002"
3515 })
3516 }
3517
3518 /// **A big archive is truncated, not refused.** `extend_bounded` bails on
3519 /// its cap, which is right for the `sub_ref` walk (a short list there is
3520 /// revoked access) and wrong for an additive read: a publication with more
3521 /// documents than the cap would return `Err` on every poll — permanently
3522 /// unreadable rather than partially read. 2 000 posts is an ordinary
3523 /// figure for a long-running blog.
3524 #[test]
3525 fn a_reading_walk_truncates_where_the_sub_ref_walk_refuses() {
3526 let page = |n: usize| -> Vec<RecordEntry> {
3527 (0..n)
3528 .map(|i| RecordEntry {
3529 uri: format!("at://did:plc:x/c/{i}"),
3530 cid: None,
3531 value: Value::Null,
3532 })
3533 .collect()
3534 };
3535 let mut out = Vec::new();
3536 assert!(!extend_truncating(&mut out, page(2), 3), "not full yet");
3537 assert_eq!(out.len(), 2);
3538 // The page that overshoots contributes what fits, and says "stop".
3539 assert!(extend_truncating(&mut out, page(5), 3), "must report full");
3540 assert_eq!(out.len(), 3, "a reading walk must keep what fits");
3541 // The same overshoot is a hard error on the fail-closed path.
3542 let mut refused = Vec::new();
3543 assert!(extend_bounded(&mut refused, page(5), 3, "c").is_err());
3544 assert!(refused.is_empty(), "a refusal must leave nothing behind");
3545 }
3546
3547 /// **The cap counts the records the caller KEEPS, not the ones the repo
3548 /// holds.** A repo-wide cap applied before the caller's filter starves a
3549 /// quiet publication whose busy sibling fills the window: poll it, walk
3550 /// the newest 2 000 documents, discard all of them as the sibling's,
3551 /// return nothing — permanently, and worse with every post the sibling
3552 /// makes. The walk pages on until it has `max` MATCHING records (still
3553 /// bounded by `MAX_LIST_PAGES` requests).
3554 #[tokio::test]
3555 async fn the_cap_counts_matching_records_not_walked_ones() {
3556 // Every page: 4 records, only the last of which the caller wants.
3557 let records: Vec<Value> = (0..4)
3558 .map(|i| {
3559 serde_json::json!({
3560 "uri": format!("at://did:plc:x/c/{i}"),
3561 "value": {"mine": i == 3}
3562 })
3563 })
3564 .collect();
3565 let body = serde_json::json!({ "records": records, "cursor": serde_json::Value::Null })
3566 .to_string();
3567 let base = crate::net::tests::serve_body(body.into_bytes()).await;
3568 let port: u16 = base
3569 .trim_end_matches('/')
3570 .rsplit(':')
3571 .next()
3572 .unwrap()
3573 .parse()
3574 .unwrap();
3575 crate::net::test_host_override(
3576 "matching-pds.test",
3577 std::net::SocketAddr::from(([127, 0, 0, 1], port)),
3578 );
3579 let client = PdsClient::anonymous(
3580 ssrf_test_client(),
3581 format!("http://matching-pds.test:{port}"),
3582 "did:plc:x",
3583 );
3584
3585 let kept = client
3586 .list_recent_matching("site.standard.document", 3, 100, |r| {
3587 r.value
3588 .get("mine")
3589 .and_then(Value::as_bool)
3590 .unwrap_or(false)
3591 })
3592 .await
3593 .expect("walk failed")
3594 .records;
3595 // One page, no cursor: one match survives. The point is that the three
3596 // non-matching records did NOT consume the cap.
3597 assert_eq!(kept.len(), 1, "the filter ran after the cap, not before it");
3598 }
3599
3600 /// **A walk that stopped early says so.** Landing exactly on the cap, or
3601 /// running out of page budget, returns the same short `Vec` as a small
3602 /// collection — and the caller cannot tell them apart afterwards. That
3603 /// silence is how the starvation this walk exists to prevent came back one
3604 /// order of magnitude further out: a quiet publication whose busy sibling
3605 /// fills every page returns nothing, forever, looking healthy.
3606 #[tokio::test]
3607 async fn a_walk_that_stops_early_reports_itself_incomplete() {
3608 let records: Vec<Value> = (0..2)
3609 .map(|i| serde_json::json!({"uri": format!("at://did:plc:x/c/{i}"), "value": {}}))
3610 .collect();
3611 // Every page is full AND advertises another — the shape that lands on
3612 // the cap with the collection still going.
3613 let body = serde_json::json!({ "records": records, "cursor": "next" }).to_string();
3614 let base = crate::net::tests::serve_body(body.into_bytes()).await;
3615 let port: u16 = base
3616 .trim_end_matches('/')
3617 .rsplit(':')
3618 .next()
3619 .unwrap()
3620 .parse()
3621 .unwrap();
3622 crate::net::test_host_override(
3623 "incomplete-pds.test",
3624 std::net::SocketAddr::from(([127, 0, 0, 1], port)),
3625 );
3626 let client = PdsClient::anonymous(
3627 ssrf_test_client(),
3628 format!("http://incomplete-pds.test:{port}"),
3629 "did:plc:x",
3630 );
3631
3632 let walk = client
3633 .list_recent_matching("c", 2, 100, |_| true)
3634 .await
3635 .expect("walk failed");
3636 assert_eq!(walk.records.len(), 2);
3637 assert!(
3638 !walk.complete,
3639 "a walk that filled its cap with pages still to come called itself complete"
3640 );
3641 }
3642
3643 /// The other side: a collection that runs out IS complete, so the caller
3644 /// does not warn about every ordinary small publication.
3645 #[tokio::test]
3646 async fn a_walk_that_exhausts_the_collection_reports_itself_complete() {
3647 let body = serde_json::json!({
3648 "records": [{"uri": "at://did:plc:x/c/1", "value": {}}]
3649 })
3650 .to_string();
3651 let base = crate::net::tests::serve_body(body.into_bytes()).await;
3652 let port: u16 = base
3653 .trim_end_matches('/')
3654 .rsplit(':')
3655 .next()
3656 .unwrap()
3657 .parse()
3658 .unwrap();
3659 crate::net::test_host_override(
3660 "complete-pds.test",
3661 std::net::SocketAddr::from(([127, 0, 0, 1], port)),
3662 );
3663 let client = PdsClient::anonymous(
3664 ssrf_test_client(),
3665 format!("http://complete-pds.test:{port}"),
3666 "did:plc:x",
3667 );
3668
3669 let walk = client
3670 .list_recent_matching("c", 100, 100, |_| true)
3671 .await
3672 .expect("walk failed");
3673 assert_eq!(walk.records.len(), 1);
3674 assert!(walk.complete, "an exhausted collection is a complete read");
3675 }
3676
3677 /// **`{}` is not a page of zero records.** The records-presence guard
3678 /// landed on the OAuth client first; a proxy answering
3679 /// `{"ok":true,"data":{}}` kept the same `sub_ref`-wipe open on the
3680 /// sidecar path, and `{}` from a stranger's PDS made an empty publication
3681 /// look healthy.
3682 #[test]
3683 fn a_body_without_a_records_field_is_not_an_empty_page() {
3684 let err =
3685 parse_list_records(br#"{}"#).expect_err("`{}` was read as a page of zero records");
3686 assert!(format!("{err:#}").contains("no records field"), "{err:#}");
3687 let err = parse_list_records(br#"{"cursor":"c"}"#)
3688 .expect_err("a cursor-only body was read as a page");
3689 assert!(format!("{err:#}").contains("no records field"), "{err:#}");
3690 let page = parse_list_records(br#"{"records":[]}"#).unwrap();
3691 assert!(page.records.is_empty());
3692 }
3693
3694 /// Serve one oversized-but-well-formed body and point `host` at it.
3695 async fn serve_oversized(host: &str, shape: &str) -> String {
3696 let filler = "x".repeat(crate::net::MAX_BODY_BYTES);
3697 let body = shape.replace("PAD", &filler);
3698 assert!(body.len() > crate::net::MAX_BODY_BYTES);
3699 let base = crate::net::tests::serve_body(body.into_bytes()).await;
3700 let port: u16 = base
3701 .trim_end_matches('/')
3702 .rsplit(':')
3703 .next()
3704 .unwrap()
3705 .parse()
3706 .unwrap();
3707 crate::net::test_host_override(host, std::net::SocketAddr::from(([127, 0, 0, 1], port)));
3708 format!("http://{host}:{port}")
3709 }
3710
3711 /// **The DID document is the most remote-controlled body of the lot.**
3712 ///
3713 /// For a `did:web:` the host comes straight out of the DID, so whoever
3714 /// supplies the DID chooses the server. The SSRF guard proves the address
3715 /// is public; it says nothing about the body being finite.
3716 #[tokio::test]
3717 async fn the_did_document_read_is_capped() {
3718 let base = serve_oversized(
3719 "did-doc-cap.test",
3720 r##"{"service":[{"id":"#atproto_pds","type":"AtprotoPersonalDataServer","serviceEndpoint":"https://pds.example"}],"pad":"PAD"}"##,
3721 )
3722 .await;
3723 let err = resolve_did_to_pds(
3724 &ssrf_test_client(),
3725 &base,
3726 "did:plc:ohutz6x5acjmpuulp3x7wxxc",
3727 )
3728 .await
3729 .expect_err("an oversized DID document was buffered whole");
3730 assert!(
3731 format!("{err:#}").contains("cap"),
3732 "failed for the wrong reason: {err:#}"
3733 );
3734 }
3735
3736 /// `resolver_base` is a user-influenced PDS host, as this function's own
3737 /// guard comment says.
3738 #[tokio::test]
3739 async fn the_resolve_handle_read_is_capped() {
3740 let base = serve_oversized(
3741 "resolve-handle-cap.test",
3742 r#"{"did":"did:plc:ohutz6x5acjmpuulp3x7wxxc","pad":"PAD"}"#,
3743 )
3744 .await;
3745 let err = resolve_handle(&ssrf_test_client(), &base, "alice.example.com")
3746 .await
3747 .expect_err("an oversized resolveHandle body was buffered whole");
3748 assert!(
3749 format!("{err:#}").contains("cap"),
3750 "failed for the wrong reason: {err:#}"
3751 );
3752 }
3753
3754 /// **The sidecar's body is capped like every other body we read.**
3755 ///
3756 /// `/internal/repo` proxies whatever the account's PDS returned, so its
3757 /// size is remote-controlled by a host the reader chose and we did not.
3758 /// Every other response in this codebase goes through
3759 /// [`crate::net::read_capped`]; this one buffered the whole thing with
3760 /// `resp.json()`, so the 8 MB ceiling that bounds the direct PDS client
3761 /// simply did not exist on the sidecar backend — which is the default.
3762 #[tokio::test]
3763 async fn the_sidecar_client_caps_the_body_it_will_buffer() {
3764 // Well-formed, and past the cap. The guard has to fire on size, not
3765 // on the shape being wrong.
3766 let filler = "x".repeat(crate::net::MAX_BODY_BYTES);
3767 let body = format!(r#"{{"ok":true,"data":{{"records":[],"pad":"{filler}"}}}}"#);
3768 assert!(body.len() > crate::net::MAX_BODY_BYTES);
3769 let base = crate::net::tests::serve_body(body.into_bytes()).await;
3770 let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
3771 let err = client
3772 .list_records(
3773 "did:plc:ewvi7nxzyoun6zhxrhs64oiz",
3774 "app.feather.subscription",
3775 None,
3776 None,
3777 )
3778 .await
3779 .expect_err("an oversized sidecar body was buffered whole");
3780 assert!(
3781 format!("{err:#}").contains("cap"),
3782 "failed for the wrong reason: {err:#}"
3783 );
3784 }
3785
3786 /// The sidecar path needs the records guard too, not only the envelope
3787 /// one: `{"ok":true,"data":{}}` is what a proxy makes of an empty or
3788 /// unexpected upstream body.
3789 #[tokio::test]
3790 async fn the_sidecar_client_refuses_a_data_object_without_records() {
3791 let base = crate::net::tests::serve_body(br#"{"ok":true,"data":{}}"#.to_vec()).await;
3792 let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
3793 let err = client
3794 .list_records(
3795 "did:plc:ewvi7nxzyoun6zhxrhs64oiz",
3796 "app.feather.subscription",
3797 None,
3798 None,
3799 )
3800 .await
3801 .expect_err("`data: {}` was read as an empty repo");
3802 assert!(format!("{err:#}").contains("no records field"), "{err:#}");
3803 }
3804
3805 /// An exactly-full final page dropped nothing, so it must not warn that it
3806 /// did: `>=` reported truncation whenever the last page landed flush.
3807 #[test]
3808 fn an_exactly_full_page_is_not_a_truncation() {
3809 let page = |n: usize| -> Vec<RecordEntry> {
3810 (0..n)
3811 .map(|i| RecordEntry {
3812 uri: format!("at://did:plc:x/c/{i}"),
3813 cid: None,
3814 value: Value::Null,
3815 })
3816 .collect()
3817 };
3818 let mut out = Vec::new();
3819 assert!(
3820 !extend_truncating(&mut out, page(3), 3),
3821 "a page that exactly fills the cap dropped nothing"
3822 );
3823 assert_eq!(out.len(), 3);
3824 assert!(
3825 extend_truncating(&mut out, page(1), 3),
3826 "one more IS a drop"
3827 );
3828 assert_eq!(out.len(), 3);
3829 }
3830
3831 /// **The reading walk USES the truncating accumulator.** The helper being
3832 /// correct is not the point — the previous round's bug was a guard that
3833 /// existed and was not called. Driven through a real server: one page of
3834 /// five records under a cap of three.
3835 #[tokio::test]
3836 async fn the_reading_walk_returns_a_truncated_archive_rather_than_an_error() {
3837 let records: Vec<Value> = (0..5)
3838 .map(|i| serde_json::json!({"uri": format!("at://did:plc:x/c/{i}"), "value": {}}))
3839 .collect();
3840 let body = serde_json::json!({ "records": records }).to_string();
3841 let base = crate::net::tests::serve_body(body.into_bytes()).await;
3842 let port: u16 = base
3843 .trim_end_matches('/')
3844 .rsplit(':')
3845 .next()
3846 .unwrap()
3847 .parse()
3848 .unwrap();
3849 crate::net::test_host_override(
3850 "truncating-pds.test",
3851 std::net::SocketAddr::from(([127, 0, 0, 1], port)),
3852 );
3853 let client = PdsClient::anonymous(
3854 ssrf_test_client(),
3855 format!("http://truncating-pds.test:{port}"),
3856 "did:plc:x",
3857 );
3858
3859 let walk = client
3860 .list_recent_matching("site.standard.document", 3, 100, |_| true)
3861 .await
3862 .expect("a big archive must be readable, not an error");
3863 assert_eq!(
3864 walk.records.len(),
3865 3,
3866 "the walk did not truncate to its cap"
3867 );
3868 assert!(
3869 !walk.complete,
3870 "a truncated walk must not report completeness"
3871 );
3872
3873 // The fail-closed walk still refuses the same overshoot.
3874 let err = client
3875 .list_all_records("community.lexicon.rss.subscription")
3876 .await;
3877 assert!(
3878 err.is_ok() || format!("{:#}", err.unwrap_err()).contains("cap"),
3879 "the sub_ref walk must keep its refusal"
3880 );
3881 }
3882
3883 /// **A write is not "succeeded" because the status was 200.** The sidecar's
3884 /// `delete_record` and `apply_writes` discard the body entirely, so a
3885 /// `200 {"error": …}` reported success: the UI showed a reader
3886 /// unsubscribed while the record was still in their repo, and a whole
3887 /// batch of writes vanished silently.
3888 #[tokio::test]
3889 async fn the_sidecar_client_refuses_a_200_error_envelope_on_writes() {
3890 let base = crate::net::tests::serve_body(
3891 br#"{"ok":true,"data":{"error":"InvalidRequest","message":"nope"}}"#.to_vec(),
3892 )
3893 .await;
3894 let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
3895 let did = "did:plc:ewvi7nxzyoun6zhxrhs64oiz";
3896 let err = client
3897 .delete_subscription(did, "rk1")
3898 .await
3899 .expect_err("a failed delete was reported as success");
3900 assert!(format!("{err:#}").contains("InvalidRequest"), "{err:#}");
3901
3902 let err = client
3903 .apply_writes(
3904 did,
3905 &[WriteOp::Delete {
3906 collection: lexicon::nsid::SUBSCRIPTION.to_string(),
3907 rkey: "rk1".to_string(),
3908 }],
3909 )
3910 .await
3911 .expect_err("a failed batch was reported as success");
3912 assert!(format!("{err:#}").contains("InvalidRequest"), "{err:#}");
3913 }
3914
3915 /// The sidecar proxies the PDS's body, so the same 2xx envelope arrives
3916 /// through `RepoOk.data` — a defaulted `Value` that deserialised into an
3917 /// empty page just as happily. Driven through the real client.
3918 #[tokio::test]
3919 async fn the_sidecar_client_refuses_a_200_error_envelope() {
3920 let base = crate::net::tests::serve_body(
3921 br#"{"ok":true,"data":{"error":"InvalidRequest","message":"nope"}}"#.to_vec(),
3922 )
3923 .await;
3924 let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
3925 let err = client
3926 .list_records(
3927 "did:plc:ewvi7nxzyoun6zhxrhs64oiz",
3928 "app.feather.subscription",
3929 None,
3930 None,
3931 )
3932 .await
3933 .expect_err("an error envelope was read as an empty page");
3934 assert!(
3935 format!("{err:#}").contains("InvalidRequest"),
3936 "failed for the wrong reason: {err:#}"
3937 );
3938 }
3939
3940 /// **Every listRecords caller refuses a 2xx error envelope, not just the
3941 /// anonymous one.** `oauth::xrpc::Repo` reads `records` off the JSON with
3942 /// `unwrap_or(Array([]))` and the sidecar's `RepoOk.data` is a defaulted
3943 /// `Value`, so a PDS answering 200 with an envelope reached
3944 /// `resolve_subscriptions` as `Ok(empty)` — which is not the fail-closed
3945 /// branch, so `sync_sub_refs` DELETEd the DID's whole `sub_ref` projection:
3946 /// one bad response revokes a reader's access to every feed they have.
3947 #[test]
3948 fn an_error_envelope_is_refused_whatever_shape_it_arrives_in() {
3949 let envelope = serde_json::json!({"error": "InvalidRequest", "message": "bad cursor"});
3950 let err = reject_error_envelope(&envelope).expect_err("an envelope passed as data");
3951 assert!(format!("{err:#}").contains("InvalidRequest"), "{err:#}");
3952 // A real page, and an empty real page, are both data.
3953 reject_error_envelope(&serde_json::json!({"records": []})).expect("an empty page is data");
3954 reject_error_envelope(&serde_json::json!({"records": [], "cursor": "c"})).unwrap();
3955 }
3956
3957 /// **A 200 carrying an error envelope is not an empty page.** `records` is
3958 /// `#[serde(default)]`, so `{"error": "...", "message": "..."}` on a 200
3959 /// deserialised as zero records — and a walk over a stranger's documents
3960 /// then returned a healthy, empty feed instead of an error. Some PDS
3961 /// implementations do answer 200 for application-level failures.
3962 #[test]
3963 fn a_200_with_an_error_envelope_is_not_an_empty_page() {
3964 let err = parse_list_records(br#"{"error":"InvalidRequest","message":"bad cursor"}"#)
3965 .expect_err("an error envelope parsed as a page");
3966 assert!(format!("{err:#}").contains("InvalidRequest"), "{err:#}");
3967 let page = parse_list_records(br#"{"records":[]}"#).expect("an empty page is a page");
3968 assert!(page.records.is_empty() && page.cursor.is_none());
3969 }
3970
3971 /// **Both invariants, now read out of the bytes rather than out of a
3972 /// `Value`.** Parsing once is the point of the change; parsing once while
3973 /// quietly dropping a guard would be a much worse trade, and these are the
3974 /// shapes those guards exist for.
3975 #[test]
3976 fn parsing_a_page_from_bytes_keeps_both_invariants() {
3977 // Each shape names the reason it must fail for. Accepting either
3978 // message would let the envelope guard be deleted without a test
3979 // noticing, because an error envelope also has no `records` field — so
3980 // it keeps failing, for a reason that stops applying the day a PDS
3981 // returns an envelope alongside a records array.
3982 for (label, body, because) in [
3983 (
3984 "an error envelope on a 2xx",
3985 &br#"{"error":"InvalidRequest","message":"bad cursor"}"#[..],
3986 "error envelope",
3987 ),
3988 (
3989 "an envelope that also carries records",
3990 &br#"{"error":"InvalidRequest","records":[]}"#[..],
3991 "error envelope",
3992 ),
3993 (
3994 "a body with no records field",
3995 &br#"{"cursor":"c"}"#[..],
3996 "no records",
3997 ),
3998 ("a proxy's empty object", &br#"{}"#[..], "no records"),
3999 ("an empty body", &b""[..], "no records"),
4000 ] {
4001 let err = parse_list_records(body)
4002 .map(|p| panic!("{label} was read as a page of {} records", p.records.len()))
4003 .unwrap_err();
4004 let msg = format!("{err:#}");
4005 assert!(
4006 msg.contains(because),
4007 "{label} should have failed on {because:?}, got: {msg}"
4008 );
4009 }
4010 let page = parse_list_records(br#"{"records":[],"cursor":"c"}"#)
4011 .expect("a genuinely empty page is still a page");
4012 assert!(page.records.is_empty());
4013 assert_eq!(page.cursor.as_deref(), Some("c"));
4014 }
4015
4016 #[test]
4017 fn list_records_envelope_deserializes() {
4018 let resp: ListRecordsResponse =
4019 serde_json::from_value(subscription_list_json()).expect("envelope");
4020 assert_eq!(resp.records.len(), 2);
4021 assert_eq!(resp.cursor.as_deref(), Some("3ksub0002"));
4022 assert_eq!(resp.records[0].cid.as_deref(), Some("bafyreisubone"));
4023 }
4024
4025 #[test]
4026 fn record_entry_rkey_is_last_uri_segment() {
4027 let resp: ListRecordsResponse =
4028 serde_json::from_value(subscription_list_json()).expect("envelope");
4029 assert_eq!(resp.records[0].rkey(), Some("3ksub0001"));
4030 assert_eq!(resp.records[1].rkey(), Some("3ksub0002"));
4031 }
4032
4033 #[test]
4034 fn record_value_parses_into_lexicon_subscription() {
4035 let resp: ListRecordsResponse =
4036 serde_json::from_value(subscription_list_json()).expect("envelope");
4037
4038 let full: Subscription = resp.records[0].parse().expect("parse full sub");
4039 assert_eq!(full.r#type, lexicon::nsid::SUBSCRIPTION);
4040 assert_eq!(full.url, "https://example.com/feed.xml");
4041 assert_eq!(full.title.as_deref(), Some("Example Blog"));
4042 assert_eq!(full.site_url.as_deref(), Some("https://example.com/"));
4043 assert_eq!(full.fetch_hint, Some(lexicon::FetchHint::Hourly));
4044
4045 let minimal: Subscription = resp.records[1].parse().expect("parse minimal sub");
4046 assert_eq!(minimal.url, "https://blog.example.org/atom.xml");
4047 assert!(minimal.title.is_none());
4048 }
4049
4050 fn ssrf_test_client() -> Client {
4051 Client::builder()
4052 .user_agent(crate::USER_AGENT)
4053 .build()
4054 .unwrap()
4055 }
4056
4057 /// A hostile `did:web` whose host is the cloud-metadata address must be
4058 /// REFUSED before any request leaves the box — the DID-document fetch now
4059 /// routes through the SSRF guard (`guarded_get_no_privacy`), which rejects
4060 /// link-local / metadata targets.
4061 #[tokio::test]
4062 async fn resolve_did_web_blocks_metadata_host() {
4063 let client = ssrf_test_client();
4064 let err = resolve_did_to_pds(&client, "https://plc.directory", "did:web:169.254.169.254")
4065 .await
4066 .unwrap_err()
4067 .to_string();
4068 assert!(
4069 err.contains("forbidden") || err.contains("internal"),
4070 "expected an SSRF refusal, got: {err}"
4071 );
4072 }
4073
4074 /// A `did:web` pointing at loopback is likewise blocked (internal service
4075 /// reflection).
4076 #[tokio::test]
4077 async fn resolve_did_web_blocks_loopback_host() {
4078 let client = ssrf_test_client();
4079 let err = resolve_did_to_pds(&client, "https://plc.directory", "did:web:127.0.0.1")
4080 .await
4081 .unwrap_err()
4082 .to_string();
4083 assert!(
4084 err.contains("forbidden") || err.contains("internal"),
4085 "expected an SSRF refusal, got: {err}"
4086 );
4087 }
4088
4089 /// `resolve_handle` against a metadata/loopback resolver base is also guarded
4090 /// (the base can come from a prior hostile DID-doc resolution).
4091 #[tokio::test]
4092 async fn resolve_handle_blocks_metadata_resolver_base() {
4093 let client = ssrf_test_client();
4094 let err = resolve_handle(&client, "http://169.254.169.254", "alice.example.com")
4095 .await
4096 .unwrap_err()
4097 .to_string();
4098 assert!(
4099 err.contains("forbidden") || err.contains("internal"),
4100 "expected an SSRF refusal, got: {err}"
4101 );
4102 }
4103
4104 /// A resolved `serviceEndpoint` that targets an internal host is rejected at
4105 /// resolve time via [`crate::net::assert_public_target`], so it can never be
4106 /// handed to a raw XRPC client.
4107 #[tokio::test]
4108 async fn service_endpoint_internal_target_rejected() {
4109 assert!(crate::net::assert_public_target("http://169.254.169.254/")
4110 .await
4111 .is_err());
4112 assert!(crate::net::assert_public_target("http://127.0.0.1:3000/")
4113 .await
4114 .is_err());
4115 // A public endpoint literal passes.
4116 assert!(crate::net::assert_public_target("https://1.1.1.1/")
4117 .await
4118 .is_ok());
4119 }
4120
4121 /// A `PdsClient` pointed at an internal `pds_base`, as an attacker-controlled
4122 /// DID document could arrange between the `assert_public_target` at resolve
4123 /// time and the request.
4124 fn internal_target_client(pds_base: &str) -> PdsClient {
4125 PdsClient::new(
4126 ssrf_test_client(),
4127 pds_base,
4128 "did:plc:victim",
4129 Auth::Session(SessionAuth {
4130 did: "did:plc:victim".to_string(),
4131 handle: None,
4132 access_jwt: "session-bearer-must-not-leak".to_string(),
4133 refresh_jwt: None,
4134 }),
4135 )
4136 }
4137
4138 /// **Regression (v0.2.8):** every `com.atproto.repo.*` WRITE must go through
4139 /// the SSRF guard, not the shared client. Before the fix only `list_records`
4140 /// was guarded, so `createRecord` / `putRecord` / `deleteRecord` /
4141 /// `applyWrites` would happily deliver the session bearer to
4142 /// `169.254.169.254` or loopback on a rebound host.
4143 #[tokio::test]
4144 async fn every_repo_write_is_refused_against_an_internal_pds() {
4145 for base in [
4146 "http://169.254.169.254",
4147 "http://127.0.0.1:9",
4148 "http://[::1]",
4149 ] {
4150 let client = internal_target_client(base);
4151 let sub = Subscription::new("https://example.com/feed.xml", "2026-08-13T00:00:00Z");
4152
4153 let mut errors = vec![
4154 client
4155 .create_record(lexicon::nsid::SUBSCRIPTION, &sub)
4156 .await
4157 .unwrap_err()
4158 .to_string(),
4159 client
4160 .put_record(lexicon::nsid::SUBSCRIPTION, "rkey", &sub, None)
4161 .await
4162 .unwrap_err()
4163 .to_string(),
4164 client
4165 .delete_record(lexicon::nsid::SUBSCRIPTION, "rkey")
4166 .await
4167 .unwrap_err()
4168 .to_string(),
4169 ];
4170 errors.push(
4171 client
4172 .apply_writes(&[WriteOp::Delete {
4173 collection: lexicon::nsid::SUBSCRIPTION.to_string(),
4174 rkey: "rkey".to_string(),
4175 }])
4176 .await
4177 .unwrap_err()
4178 .to_string(),
4179 );
4180
4181 for err in errors {
4182 assert!(
4183 err.contains("forbidden") || err.contains("internal"),
4184 "{base}: expected an SSRF refusal, got: {err}"
4185 );
4186 }
4187 }
4188 }
4189
4190 /// **Regression (v0.2.8):** the app password travels in the request BODY,
4191 /// where reqwest's cross-origin header sanitisation cannot protect it — so
4192 /// `createSession` is guarded too, and a rebound/internal `pds_base` never
4193 /// receives it.
4194 #[tokio::test]
4195 async fn app_password_login_is_refused_against_an_internal_pds() {
4196 let client = ssrf_test_client();
4197 for base in ["http://169.254.169.254", "http://127.0.0.1:9"] {
4198 let err = login_with_app_password(&client, base, "alice.example.com", "hunter2-app-pw")
4199 .await
4200 .unwrap_err()
4201 .to_string();
4202 assert!(
4203 err.contains("forbidden") || err.contains("internal"),
4204 "{base}: expected an SSRF refusal, got: {err}"
4205 );
4206 }
4207 }
4208
4209 /// An anonymous client is read-only: the write paths fail closed on
4210 /// [`Auth::bearer`] before any socket work, so `Auth::Anonymous` can never
4211 /// become a credential-less write primitive against a stranger's PDS.
4212 #[tokio::test]
4213 async fn anonymous_client_cannot_write() {
4214 let client = PdsClient::anonymous(
4215 ssrf_test_client(),
4216 "https://pds.example.com",
4217 "did:plc:stranger",
4218 );
4219 let err = client
4220 .delete_record(lexicon::nsid::SUBSCRIPTION, "rkey")
4221 .await
4222 .unwrap_err()
4223 .to_string();
4224 assert!(
4225 err.contains("no credentials") || err.contains("anonymous") || err.contains("bearer"),
4226 "expected a fail-closed auth error, got: {err}"
4227 );
4228 }
4229
4230 #[test]
4231 fn write_result_deserializes() {
4232 let wr: WriteResult = serde_json::from_value(json!({
4233 "uri": "at://did:plc:abc123/community.lexicon.rss.subscription/3ksubnew",
4234 "cid": "bafyreinew"
4235 }))
4236 .expect("write result");
4237 assert!(wr.uri.ends_with("3ksubnew"));
4238 assert_eq!(wr.cid.as_deref(), Some("bafyreinew"));
4239 }
4240
4241 #[test]
4242 fn did_document_finds_pds_endpoint() {
4243 let doc: DidDocument = serde_json::from_value(json!({
4244 "id": "did:plc:abc123",
4245 "service": [
4246 {
4247 "id": "#atproto_pds",
4248 "type": "AtprotoPersonalDataServer",
4249 "serviceEndpoint": "https://pds.example.com/"
4250 }
4251 ]
4252 }))
4253 .expect("did doc");
4254 assert_eq!(
4255 doc.pds_endpoint().as_deref(),
4256 Some("https://pds.example.com")
4257 );
4258 }
4259
4260 #[test]
4261 fn did_document_without_pds_yields_none() {
4262 let doc: DidDocument = serde_json::from_value(json!({
4263 "id": "did:plc:abc123",
4264 "service": []
4265 }))
4266 .expect("did doc");
4267 assert!(doc.pds_endpoint().is_none());
4268 }
4269
4270 #[test]
4271 fn session_auth_deserializes_create_session_shape() {
4272 let session: SessionAuth = serde_json::from_value(json!({
4273 "did": "did:plc:abc123",
4274 "handle": "alice.example.com",
4275 "accessJwt": "eyJh...access",
4276 "refreshJwt": "eyJh...refresh"
4277 }))
4278 .expect("session");
4279 assert_eq!(session.did, "did:plc:abc123");
4280 assert_eq!(session.handle.as_deref(), Some("alice.example.com"));
4281 let auth = Auth::Session(session);
4282 assert_eq!(auth.bearer().expect("bearer"), "eyJh...access");
4283 }
4284
4285 #[test]
4286 fn oauth_variant_carries_no_direct_bearer() {
4287 let auth = Auth::Oauth(OauthPlaceholder::default());
4288 assert!(
4289 auth.bearer().is_err(),
4290 "Auth::Oauth carries no direct bearer — the sidecar owns the OAuth path"
4291 );
4292 }
4293
4294 #[test]
4295 fn anonymous_variant_carries_no_bearer() {
4296 let err = Auth::Anonymous.bearer().unwrap_err().to_string();
4297 assert!(
4298 err.contains("anonymous"),
4299 "the anonymous refusal must name itself, got: {err}"
4300 );
4301 }
4302
4303 #[test]
4304 fn anonymous_client_targets_the_requested_repo() {
4305 let client = PdsClient::anonymous(
4306 ssrf_test_client(),
4307 "https://pds.example.com/",
4308 "did:plc:abc123",
4309 );
4310 // The trailing slash is trimmed so `xrpc_url` joins cleanly.
4311 assert_eq!(client.pds_base(), "https://pds.example.com");
4312 assert_eq!(client.did(), "did:plc:abc123");
4313 // …and it holds no credential.
4314 assert!(client.auth.bearer().is_err());
4315 }
4316
4317 /// The regression test for the defect this milestone fixes: `list_records`
4318 /// used to send on the shared client, bypassing the SSRF guard entirely. It
4319 /// now routes through `net::guarded_get_no_privacy`, so an internal
4320 /// `pds_base` is refused before a packet leaves the box. Hermetic — the hosts
4321 /// are IP literals, rejected without any DNS lookup or connect.
4322 #[tokio::test]
4323 async fn list_records_blocks_internal_pds_base() {
4324 for base in ["http://169.254.169.254", "http://127.0.0.1:1"] {
4325 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
4326 let err = client
4327 .list_records(lexicon::nsid::SUBSCRIPTION, Some(1), None)
4328 .await
4329 .unwrap_err()
4330 .to_string();
4331 assert!(
4332 err.contains("forbidden") || err.contains("internal"),
4333 "expected an SSRF refusal for {base}, got: {err}"
4334 );
4335 }
4336 }
4337
4338 /// The guard is not anonymous-only: an *authenticated* client reading a
4339 /// hostile PDS base is blocked identically. (That path was only ever safe by
4340 /// accident of usage.)
4341 #[tokio::test]
4342 async fn list_records_guard_applies_to_authed_clients_too() {
4343 let auth = Auth::Session(SessionAuth {
4344 did: "did:plc:x".to_string(),
4345 handle: None,
4346 access_jwt: "x".to_string(),
4347 refresh_jwt: None,
4348 });
4349 let client = PdsClient::new(
4350 ssrf_test_client(),
4351 "http://169.254.169.254",
4352 "did:plc:x",
4353 auth,
4354 );
4355 let err = client
4356 .list_records(lexicon::nsid::SUBSCRIPTION, Some(1), None)
4357 .await
4358 .unwrap_err()
4359 .to_string();
4360 assert!(
4361 err.contains("forbidden") || err.contains("internal"),
4362 "expected an SSRF refusal, got: {err}"
4363 );
4364 }
4365
4366 #[test]
4367 fn apply_writes_ops_render_tagged_union() {
4368 let create = WriteOp::Create {
4369 collection: lexicon::nsid::SUBSCRIPTION.to_string(),
4370 rkey: None,
4371 value: json!({"url": "https://example.com/feed.xml"}),
4372 };
4373 let update = WriteOp::Update {
4374 collection: lexicon::nsid::READ_STATE.to_string(),
4375 rkey: "feedhash01".to_string(),
4376 value: json!({"feedUrl": "https://example.com/feed.xml"}),
4377 };
4378 let delete = WriteOp::Delete {
4379 collection: lexicon::nsid::SAVED.to_string(),
4380 rkey: "3ksaved01".to_string(),
4381 };
4382
4383 assert_eq!(
4384 create.to_json()["$type"],
4385 json!("com.atproto.repo.applyWrites#create")
4386 );
4387 // A create with no explicit rkey omits the field (server assigns a tid).
4388 assert!(create.to_json().get("rkey").is_none());
4389
4390 assert_eq!(
4391 update.to_json()["$type"],
4392 json!("com.atproto.repo.applyWrites#update")
4393 );
4394 assert_eq!(update.to_json()["rkey"], json!("feedhash01"));
4395
4396 assert_eq!(
4397 delete.to_json()["$type"],
4398 json!("com.atproto.repo.applyWrites#delete")
4399 );
4400 assert_eq!(delete.to_json()["rkey"], json!("3ksaved01"));
4401 }
4402
4403 #[test]
4404 fn read_state_flush_creates_first_then_updates() {
4405 // A cursor whose PDS record does NOT yet exist (pds_created = false) must
4406 // become a CREATE op at its stable rkey — NOT a bare update, which would
4407 // error on the missing record and (applyWrites being atomic per-repo) drop
4408 // the whole batch on a feed's first flush.
4409 let fresh = (
4410 "rs-fresh".to_string(),
4411 ReadState::new("https://a.example/feed.xml", None, "2026-07-12T00:00:00Z"),
4412 false,
4413 );
4414 // An already-created cursor updates in place.
4415 let existing = (
4416 "rs-existing".to_string(),
4417 ReadState::new(
4418 "https://b.example/feed.xml",
4419 Some("2026-07-11T00:00:00Z".to_string()),
4420 "2026-07-12T00:00:00Z",
4421 ),
4422 true,
4423 );
4424
4425 let ops = read_state_write_ops(&[fresh, existing]).expect("build ops");
4426 assert_eq!(ops.len(), 2);
4427
4428 // First op: a create carrying the stable rkey (put/create, not update).
4429 let create = ops[0].to_json();
4430 assert_eq!(
4431 create["$type"],
4432 json!("com.atproto.repo.applyWrites#create"),
4433 "first flush of a new feed must CREATE its readState record"
4434 );
4435 assert_eq!(create["rkey"], json!("rs-fresh"));
4436 // The created record omits readThrough (F1): backlog not implicitly read.
4437 assert!(create["value"].get("readThrough").is_none());
4438
4439 // Second op: an update for the already-created record.
4440 let update = ops[1].to_json();
4441 assert_eq!(
4442 update["$type"],
4443 json!("com.atproto.repo.applyWrites#update")
4444 );
4445 assert_eq!(update["rkey"], json!("rs-existing"));
4446
4447 // Both ride the SAME batch — batching is preserved.
4448 assert_eq!(ops.len(), 2);
4449 }
4450
4451 #[test]
4452 fn urlencode_escapes_did_colons_and_keeps_unreserved() {
4453 assert_eq!(urlencode("did:plc:abc123"), "did%3Aplc%3Aabc123");
4454 assert_eq!(
4455 urlencode("community.lexicon.rss.subscription"),
4456 "community.lexicon.rss.subscription"
4457 );
4458 assert_eq!(urlencode("a b&c"), "a%20b%26c");
4459 }
4460
4461 #[test]
4462 fn xrpc_record_not_found_is_detected() {
4463 let err = AtProtoError::Xrpc {
4464 status: StatusCode::BAD_REQUEST,
4465 error: "RecordNotFound".to_string(),
4466 message: Some("Could not locate record".to_string()),
4467 };
4468 assert!(err.is_record_not_found());
4469 }
4470
4471 // -- reader-facing CRUD: rkey extraction --------------------------------
4472
4473 #[test]
4474 fn write_result_extracts_rkey_from_uri() {
4475 let wr: WriteResult = serde_json::from_value(json!({
4476 "uri": "at://did:plc:abc123/community.lexicon.rss.subscription/3ksubnew",
4477 "cid": "bafyreinew"
4478 }))
4479 .expect("write result");
4480 assert_eq!(wr.rkey(), Some("3ksubnew"));
4481 assert_eq!(wr.into_rkey(), "3ksubnew");
4482 }
4483
4484 // -- reader-facing CRUD: deterministic sort orders ----------------------
4485 //
4486 // The `list_*_sorted` wrappers only add an ordering on top of the network
4487 // `list_*` read, so we exercise the *comparator* here on representative
4488 // data (parsed from a listRecords-shaped envelope) with no network.
4489
4490 // -- reader-facing CRUD: bulk applyWrites shape (OPML import) ------------
4491
4492 /// **Bulk subscribe, through the real client, asserted on the bytes it
4493 /// sent.** The test this replaces built the `WriteOp::Create` ops itself
4494 /// ("mirror what `add_subscriptions_bulk` builds") and asserted on its own
4495 /// construction; the function was never called, and writing every feed
4496 /// into the wrong collection with server-assigned rkeys left the suite
4497 /// green. Three atproto sort tests that re-implemented the comparator
4498 /// inline are deleted alongside — `lexicon::sort_tests` fails their
4499 /// mutation, and they added nothing but a misleading name.
4500 #[tokio::test]
4501 async fn bulk_subscribe_writes_client_assigned_ordered_rkeys_to_the_right_collection() {
4502 let (base, log) =
4503 crate::net::tests::serve_json_capturing(br#"{"ok":true,"data":{}}"#.to_vec()).await;
4504 let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
4505 let subs: Vec<crate::vetted::VettedSubscription> = (0..3)
4506 .map(|i| {
4507 crate::vetted::VettedSubscription::new(&lexicon::Subscription::new(
4508 format!("https://f{i}.example/feed.xml"),
4509 "2026-07-12T00:00:00.000Z",
4510 ))
4511 })
4512 .collect();
4513
4514 let rkeys = client
4515 .add_subscriptions_bulk("did:plc:ewvi7nxzyoun6zhxrhs64oiz", &subs)
4516 .await
4517 .expect("bulk write failed");
4518
4519 let sent = log.lock().unwrap().clone();
4520 assert_eq!(
4521 sent.len(),
4522 1,
4523 "expected one applyWrites request, got {sent:?}"
4524 );
4525 let body: Value = serde_json::from_str(sent[0].split("\r\n\r\n").nth(1).unwrap())
4526 .expect("request body is JSON");
4527 let writes = body["writes"].as_array().expect("writes array");
4528 assert_eq!(writes.len(), 3);
4529 for (i, w) in writes.iter().enumerate() {
4530 assert_eq!(
4531 w["collection"],
4532 lexicon::nsid::SUBSCRIPTION,
4533 "write {i} went to the wrong collection"
4534 );
4535 assert_eq!(
4536 w["rkey"].as_str(),
4537 Some(rkeys[i].as_str()),
4538 "write {i} does not carry the rkey the client returned"
4539 );
4540 }
4541 let mut sorted = rkeys.clone();
4542 sorted.sort();
4543 assert_eq!(rkeys, sorted, "client-assigned rkeys must ascend");
4544 assert_eq!(
4545 rkeys.iter().collect::<std::collections::HashSet<_>>().len(),
4546 3,
4547 "rkeys must be distinct"
4548 );
4549 }
4550
4551 /// **The walk stops on a repeated cursor.** `MAX_LIST_PAGES`, the
4552 /// same-cursor guard and the `got > 0` guard had no test; only
4553 /// `extend_bounded` was covered directly. A PDS that echoes the same
4554 /// cursor forever would otherwise be walked for 200 pages.
4555 #[tokio::test]
4556 async fn list_all_records_stops_on_a_repeated_cursor() {
4557 let body = serde_json::json!({
4558 "records": [{"uri": "at://did:plc:x/c/1", "value": {}}],
4559 "cursor": "same-every-time"
4560 })
4561 .to_string();
4562 let base = crate::net::tests::serve_body(body.into_bytes()).await;
4563 let port: u16 = base
4564 .trim_end_matches('/')
4565 .rsplit(':')
4566 .next()
4567 .unwrap()
4568 .parse()
4569 .unwrap();
4570 crate::net::test_host_override(
4571 "repeated-cursor.test",
4572 std::net::SocketAddr::from(([127, 0, 0, 1], port)),
4573 );
4574 let client = PdsClient::anonymous(
4575 ssrf_test_client(),
4576 format!("http://repeated-cursor.test:{port}"),
4577 "did:plc:x",
4578 );
4579 let records = client.list_all_records("c").await.expect("walk failed");
4580 // Page 1: cursor None → "same". Page 2: "same" again → stop, after
4581 // taking that page. Two pages, not two hundred.
4582 assert_eq!(records.len(), 2, "a repeated cursor was followed");
4583 }
4584
4585 // -- bounding the parse before it allocates -----------------------------
4586
4587 /// **Strings cannot invent structure.** The subtle half of the bound: an
4588 /// article containing a million commas is one node, and counting naively
4589 /// would refuse it.
4590 #[test]
4591 fn structure_inside_a_string_is_not_structure() {
4592 let prose = format!(
4593 r#"{{"records":[{{"uri":"at://d/c/r","value":{{"t":"{}"}}}}]}}"#,
4594 "a,b,[c],{d}:e,".repeat(50_000)
4595 );
4596 let bound = count_structural_chars(prose.as_bytes());
4597 assert!(
4598 bound < 100,
4599 "a page of prose full of punctuation was counted as {bound} nodes"
4600 );
4601 assert!(
4602 parse_list_records(prose.as_bytes()).is_ok(),
4603 "a legitimate page of prose was refused"
4604 );
4605 }
4606
4607 #[test]
4608 fn a_node_explosion_is_refused_before_it_is_parsed() {
4609 // ~8 MB of the cheapest node there is, which is the measured attack.
4610 let mut body = String::from(r#"{"records":[{"uri":"at://d/c/r","value":["#);
4611 for _ in 0..1_200_000 {
4612 body.push_str("{},");
4613 }
4614 body.push_str(r#"{}]}]}"#);
4615 assert!(
4616 body.len() > 3_000_000,
4617 "the probe body is {} bytes",
4618 body.len()
4619 );
4620
4621 let bound = count_structural_chars(body.as_bytes());
4622 assert!(
4623 bound > MAX_LIST_STRUCTURAL_CHARS,
4624 "the attack shape was counted as only {bound} nodes"
4625 );
4626 let err = parse_list_records(body.as_bytes())
4627 .expect_err("a node explosion was parsed rather than refused");
4628 assert!(
4629 format!("{err:#}").contains("structural characters"),
4630 "failed for the wrong reason: {err:#}"
4631 );
4632 }
4633
4634 /// **The densest page the lexicons permit must fit, with room.**
4635 ///
4636 /// This is the floor under [`MAX_LIST_STRUCTURAL_CHARS`], and it is the reason the cap
4637 /// is 640 000 rather than the ~150 000 that would otherwise hold the memory
4638 /// claim comfortably. A `readState` record carries up to
4639 /// [`crate::lexicon::ReadState::MAX_IDS`] read ids, and a page carries 100 of
4640 /// them — far denser in nodes than a page of articles, which is mostly
4641 /// prose. Tighten the cap below this and a reader with a lot of history
4642 /// stops being able to sync at all.
4643 #[test]
4644 fn a_full_read_state_page_fits_under_the_cap() {
4645 let ids: Vec<String> = (0..crate::lexicon::ReadState::MAX_IDS)
4646 .map(|i| format!("https://example.com/blog/post-{i}"))
4647 .collect();
4648 let records: Vec<serde_json::Value> = (0..100)
4649 .map(|i| {
4650 serde_json::json!({
4651 "uri": format!("at://did:plc:ohutz6x5acjmpuulp3x7wxxc/community.lexicon.rss.readState/3lab{i}"),
4652 "cid": "bafyreiabc123def456ghi789jkl012mno345pqr678stu901",
4653 "value": {
4654 "$type": "community.lexicon.rss.readState",
4655 "feedUrl": "https://example.com/feed.xml",
4656 "readThrough": "2026-07-11T09:30:00Z",
4657 // **BOTH arrays, because the lexicon permits both.**
4658 // Filling only `readIds` counted 203 503 nodes, so a cap
4659 // as low as 300 000 passed every test in the suite while
4660 // refusing the very page this test exists to protect.
4661 "readIds": ids,
4662 "unreadIds": ids,
4663 }
4664 })
4665 })
4666 .collect();
4667 let body = serde_json::json!({ "records": records }).to_string();
4668 let bound = count_structural_chars(body.as_bytes());
4669 assert!(
4670 bound < MAX_LIST_STRUCTURAL_CHARS,
4671 "the densest legitimate page counts {bound} nodes against a cap of {MAX_LIST_STRUCTURAL_CHARS}"
4672 );
4673 // And it is dense enough to be the floor the cap was chosen for: a page
4674 // counting only a fifth of the cap would pass the assertion above while
4675 // leaving the cap free to drop far below real traffic.
4676 assert!(
4677 bound > MAX_LIST_STRUCTURAL_CHARS / 2,
4678 "this page counts only {bound} nodes, so it is no longer the floor \
4679 `MAX_LIST_STRUCTURAL_CHARS` was measured against and a much tighter cap would \
4680 pass it"
4681 );
4682 assert!(
4683 parse_list_records(body.as_bytes()).is_ok(),
4684 "a full read-state page was refused"
4685 );
4686 }
4687
4688 /// The write path takes the same guard as the listing path.
4689 #[tokio::test]
4690 async fn the_write_path_refuses_a_node_explosion() {
4691 let mut data = String::from(r#"{"ok":true,"data":{"records":["#);
4692 for _ in 0..700_000 {
4693 data.push_str("{},");
4694 }
4695 data.push_str(r#"{}]}}"#);
4696 let base = crate::net::tests::serve_body(data.into_bytes()).await;
4697 let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
4698 let err = client
4699 .delete_record("did:plc:ewvi7nxzyoun6zhxrhs64oiz", "c", "r")
4700 .await
4701 .expect_err("a node explosion reached the parser on the write path");
4702 assert!(
4703 format!("{err:#}").contains("structural characters"),
4704 "failed for the wrong reason: {err:#}"
4705 );
4706 }
4707
4708 #[test]
4709 fn an_ordinary_page_is_nowhere_near_the_structure_bound() {
4710 // **A page, not a record.** This served `paged_bodies(1, 17_000, false)` —
4711 // ONE page holding ONE record — which counts about eleven characters, so
4712 // the old `< 1_000` assertion held by three orders of magnitude and would
4713 // have passed with the cap at 1 000. It also made the doc comment's "a
4714 // page of 100 documents measures 40 000" untested, and that figure was
4715 // wrong by 10x.
4716 let records: Vec<serde_json::Value> = (0..100)
4717 .map(|i| {
4718 serde_json::json!({
4719 "uri": format!("at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.document/3lab{i}"),
4720 "cid": "bafyreiabc123def456ghi789jkl012mno345pqr678stu901",
4721 "value": {
4722 "$type": "site.standard.document",
4723 "title": "A reasonably typical post title",
4724 "path": format!("/posts/{i}"),
4725 "site": "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
4726 "publishedAt": "2026-07-11T09:30:00Z",
4727 "description": "x".repeat(120),
4728 "textContent": "y".repeat(15_000),
4729 }
4730 })
4731 })
4732 .collect();
4733 let body = serde_json::json!({ "records": records }).to_string();
4734 // A real page of documents is mostly prose: 1.5 MB on the wire for 4 003
4735 // counted characters.
4736 assert!(
4737 body.len() > 1_000_000,
4738 "the probe page is only {} bytes, so it is not a full page",
4739 body.len(),
4740 );
4741 let counted = count_structural_chars(body.as_bytes());
4742 assert!(
4743 (3_500..4_500).contains(&counted),
4744 "a page of 100 documents counted {counted}, not the ~4 003 the cap's \
4745 doc comment claims — the ordinary-traffic end of the bracket moved",
4746 );
4747 assert!(
4748 counted * 100 < MAX_LIST_STRUCTURAL_CHARS,
4749 "ordinary traffic is within 100x of the cap ({counted} against \
4750 {MAX_LIST_STRUCTURAL_CHARS}), which is not the headroom the cap claims",
4751 );
4752 }
4753
4754 /// **An escaped quote does not end the string**, asserted without a magic
4755 /// number: the same document with the escape replaced by a plain letter has
4756 /// the same structure, so it must count the same. Get the escape wrong and the
4757 /// scanner leaves the string early and counts the rest as structure.
4758 #[test]
4759 fn an_escaped_quote_does_not_end_the_string() {
4760 let escaped = br#"{"records":[{"uri":"a\"b","value":{}}],"cursor":"x"}"#;
4761 let plain = br#"{"records":[{"uri":"axb","value":{}}],"cursor":"x"}"#;
4762 assert_eq!(
4763 count_structural_chars(escaped),
4764 count_structural_chars(plain),
4765 "an escaped quote changed the structure count"
4766 );
4767 let backslash = br#"{"records":[],"cursor":"x\\"}"#;
4768 let letter = br#"{"records":[],"cursor":"xy"}"#;
4769 assert_eq!(
4770 count_structural_chars(backslash),
4771 count_structural_chars(letter),
4772 "an escaped backslash changed the structure count"
4773 );
4774 }
4775
4776 /// A string is itself a node, so a page of strings costs more than a page of
4777 /// numbers. Without that, an array of a million short strings reads as cheap.
4778 #[test]
4779 fn a_string_counts_as_a_node() {
4780 assert!(
4781 count_structural_chars(br#"["a","b","c"]"#) > count_structural_chars(br#"[1,1,1]"#),
4782 "strings were not counted, so an array of them looks free"
4783 );
4784 }
4785
4786 #[tokio::test]
4787 async fn the_sidecar_refuses_a_node_explosion_too() {
4788 let mut data =
4789 String::from(r#"{"ok":true,"data":{"records":[{"uri":"at://d/c/r","value":["#);
4790 for _ in 0..1_200_000 {
4791 data.push_str("{},");
4792 }
4793 data.push_str(r#"{}]}]}}"#);
4794 let base = crate::net::tests::serve_body(data.into_bytes()).await;
4795 let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
4796 let err = client
4797 .list_records("did:plc:ewvi7nxzyoun6zhxrhs64oiz", "c", None, None)
4798 .await
4799 .expect_err("a node explosion reached the parser");
4800 assert!(
4801 format!("{err:#}").contains("structural characters"),
4802 "failed for the wrong reason: {err:#}"
4803 );
4804 }
4805
4806 // -- walk byte budget ---------------------------------------------------
4807
4808 /// What a parsed value really costs, counted independently of the code
4809 /// under test: every node occupies a `Value`, wherever it sits.
4810 fn node_count(v: &serde_json::Value) -> usize {
4811 1 + match v {
4812 serde_json::Value::Array(a) => a.iter().map(node_count).sum::<usize>(),
4813 serde_json::Value::Object(o) => o.values().map(node_count).sum::<usize>(),
4814 _ => 0,
4815 }
4816 }
4817
4818 fn record_of(value: serde_json::Value) -> RecordEntry {
4819 RecordEntry {
4820 uri: "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/c/3lab".to_string(),
4821 cid: Some("bafyreiabc123def456ghi789jkl012mno345pqr678stu901".to_string()),
4822 value,
4823 }
4824 }
4825
4826 /// **The estimate must never under-report, on any shape.**
4827 ///
4828 /// The version this replaces charged serialized length, which is accurate
4829 /// on prose-shaped records and 42x optimistic on the shapes an attacker
4830 /// picks. A bound that is only correct on benign input is not a bound.
4831 #[test]
4832 fn the_estimate_charges_every_node_at_least_what_a_parsed_value_costs() {
4833 let deep: serde_json::Value =
4834 serde_json::from_str(&format!("{}{}", "[".repeat(100), "]".repeat(100))).unwrap();
4835 let shapes: Vec<(&str, serde_json::Value)> = vec![
4836 ("100 nested empty arrays", deep),
4837 (
4838 "4096 empty arrays",
4839 serde_json::json!(vec![serde_json::json!([]); 4096]),
4840 ),
4841 ("4096 empty strings", serde_json::json!(vec![""; 4096])),
4842 (
4843 "4096 nulls",
4844 serde_json::json!(vec![serde_json::Value::Null; 4096]),
4845 ),
4846 ("4096 bools", serde_json::json!(vec![true; 4096])),
4847 ("4096 small numbers", serde_json::json!(vec![0; 4096])),
4848 (
4849 "object with short keys",
4850 serde_json::Value::Object(
4851 (0..4096)
4852 .map(|i| (format!("k{i}"), serde_json::json!([])))
4853 .collect(),
4854 ),
4855 ),
4856 (
4857 "a realistic document",
4858 serde_json::json!({
4859 "$type": "site.standard.document",
4860 "title": "A post with a reasonably typical title",
4861 "path": "/posts/one",
4862 "publishedAt": "2026-07-11T09:30:00Z",
4863 "textContent": "x".repeat(17_000),
4864 }),
4865 ),
4866 ];
4867 for (label, value) in shapes {
4868 let entry = record_of(value);
4869 let charged = approx_bytes(&entry);
4870 let floor = node_count(&entry.value) * std::mem::size_of::<serde_json::Value>();
4871 assert!(
4872 charged >= floor,
4873 "{label}: charged {charged} for {} nodes, which cannot cost less than {floor}",
4874 node_count(&entry.value)
4875 );
4876 let wire = serde_json::to_vec(&entry.value).unwrap().len();
4877 assert!(
4878 charged >= wire,
4879 "{label}: charged {charged}, under the {wire} bytes it takes on the wire alone"
4880 );
4881 }
4882 }
4883
4884 /// **Known answers, taken from a real allocator elsewhere.**
4885 ///
4886 /// The property above models `Value` nodes and nothing else, which is how an
4887 /// object-shaped under-charge of about half slipped past it: a
4888 /// `serde_json::Map` is a `BTreeMap` whose leaf is allocated whole, so the
4889 /// entries' own nodes are not the cost.
4890 ///
4891 /// **This test does not measure anything.** The two figures were obtained
4892 /// with a counting global allocator against the `serde_json` in this
4893 /// lockfile and are hardcoded here, because a global allocator is not
4894 /// something to install in the suite for one assertion. That makes this a
4895 /// tripwire for the *estimate* changing, not for the *real cost* changing: a
4896 /// dependency or toolchain bump that grows a map's true footprint leaves this
4897 /// green and the estimate quietly short again. Re-taking these numbers is the
4898 /// price of trusting them.
4899 #[test]
4900 fn the_estimate_covers_shapes_measured_against_a_real_allocator() {
4901 let many_small = serde_json::json!(vec![serde_json::json!({"a": 0}); 5000]);
4902 let mut deep = serde_json::json!({"a": 0});
4903 for _ in 0..99 {
4904 deep = serde_json::json!({ "a": deep });
4905 }
4906 for (label, value, measured) in [
4907 ("5000 one-key objects", many_small, 3_430_000usize),
4908 ("a 100-deep chain of one-key objects", deep, 63_350),
4909 ] {
4910 let charged = approx_bytes(&record_of(value));
4911 assert!(
4912 charged >= measured,
4913 "{label}: charged {charged} against {measured} bytes actually held"
4914 );
4915 }
4916 }
4917
4918 #[test]
4919 fn the_estimate_counts_the_uri_and_cid_too() {
4920 let bare = RecordEntry {
4921 uri: String::new(),
4922 cid: None,
4923 value: serde_json::json!(null),
4924 };
4925 let addressed = record_of(serde_json::json!(null));
4926 assert!(
4927 approx_bytes(&addressed) > approx_bytes(&bare),
4928 "a record's own identifiers are retained alongside its value"
4929 );
4930 }
4931
4932 #[test]
4933 fn the_budget_admits_a_page_that_exactly_fills_it() {
4934 let page = vec![record_of(serde_json::json!({"t": "x".repeat(1000)}))];
4935 let exact: usize = page.iter().map(approx_bytes).sum();
4936 assert!(
4937 ByteBudget::new(exact).admit(&page),
4938 "a page that exactly fits was refused; the fence-post is one byte out"
4939 );
4940 assert!(
4941 !ByteBudget::new(exact - 1).admit(&page),
4942 "a page one byte over the budget was admitted"
4943 );
4944 }
4945
4946 #[test]
4947 fn a_refused_page_leaves_the_running_total_alone() {
4948 let small = vec![record_of(serde_json::json!({"t": "x".repeat(100)}))];
4949 let huge = vec![record_of(serde_json::json!({"t": "x".repeat(100_000)}))];
4950 let cost: usize = small.iter().map(approx_bytes).sum();
4951 let mut budget = ByteBudget::new(cost * 3);
4952
4953 assert!(budget.admit(&small), "the first page fits");
4954 let after_one = budget.used();
4955 assert!(after_one > 0, "an admitted page must be charged");
4956
4957 assert!(!budget.admit(&huge), "the oversized page must be refused");
4958 assert_eq!(
4959 budget.used(),
4960 after_one,
4961 "a refused page moved the total — either charged, or reset"
4962 );
4963 assert!(
4964 budget.admit(&small),
4965 "the walk could not continue against the total it had before the refusal"
4966 );
4967 }
4968
4969 /// Build `pages` responses, each holding one record of about `bytes`, each
4970 /// pointing at the next. Returns the base URL and what one page costs.
4971 ///
4972 /// **Pages that differ is the whole point.** A walk served the same body
4973 /// twice stops on its repeated-cursor guard, so every test built on the
4974 /// fixed-body server refuses on page one and never exercises accumulation
4975 /// at all — which is how a per-page budget once passed a whole suite.
4976 pub(crate) fn paged_bodies(
4977 pages: usize,
4978 bytes: usize,
4979 envelope: bool,
4980 ) -> (Vec<Vec<u8>>, usize) {
4981 let record = |i: usize| {
4982 serde_json::json!({
4983 "uri": format!("at://did:plc:ohutz6x5acjmpuulp3x7wxxc/c/3lab{i}"),
4984 "cid": "bafyreiabc123def456ghi789jkl012mno345pqr678stu901",
4985 "value": { "t": "x".repeat(bytes) }
4986 })
4987 };
4988 let bodies = (0..pages)
4989 .map(|i| {
4990 let mut page = serde_json::json!({ "records": [record(i)] });
4991 if i + 1 < pages {
4992 page["cursor"] = serde_json::json!(format!("p{}", i + 1));
4993 }
4994 if envelope {
4995 page = serde_json::json!({ "ok": true, "data": page });
4996 }
4997 page.to_string().into_bytes()
4998 })
4999 .collect();
5000 let entry: RecordEntry = serde_json::from_value(record(0)).unwrap();
5001 (bodies, approx_bytes(&entry))
5002 }
5003
5004 async fn host_for(bodies: Vec<Vec<u8>>, host: &str) -> (String, u16) {
5005 let base = crate::net::tests::serve_bodies_in_sequence(bodies).await;
5006 let port: u16 = base
5007 .trim_end_matches('/')
5008 .rsplit(':')
5009 .next()
5010 .unwrap()
5011 .parse()
5012 .unwrap();
5013 crate::net::test_host_override(host, std::net::SocketAddr::from(([127, 0, 0, 1], port)));
5014 (format!("http://{host}:{port}"), port)
5015 }
5016
5017 // ---- #177: one malformed envelope ---------------------------------------
5018
5019 /// A record with no `uri`, which is what #177 probed on `main`.
5020 fn malformed_page(cursor: Option<&str>) -> Vec<u8> {
5021 let mut page = serde_json::json!({
5022 "records": [
5023 { "uri": "at://did:plc:x/c/3labGOOD", "value": {} },
5024 { "cid": "bafy", "value": {} },
5025 ]
5026 });
5027 if let Some(c) = cursor {
5028 page["cursor"] = serde_json::json!(c);
5029 }
5030 page.to_string().into_bytes()
5031 }
5032
5033 #[test]
5034 fn one_malformed_envelope_is_counted_not_fatal_to_the_page() {
5035 let page = parse_list_records(&malformed_page(None))
5036 .expect("one malformed envelope failed the whole page");
5037 assert_eq!(page.records.len(), 1, "the good record was not kept");
5038 assert_eq!(page.records[0].uri, "at://did:plc:x/c/3labGOOD");
5039 assert_eq!(page.malformed, 1, "the malformed record was not counted");
5040 }
5041
5042 /// The publications listing in `standard_site::fetch` reads a stranger's
5043 /// repo through this walk: it skips, counts, and keeps paging.
5044 #[tokio::test]
5045 async fn the_skipping_walk_skips_malformed_records_and_keeps_paging() {
5046 let only_bad = serde_json::json!({
5047 "records": [{ "cid": "bafy", "value": {} }], "cursor": "p1"
5048 })
5049 .to_string()
5050 .into_bytes();
5051 let bodies = vec![
5052 only_bad,
5053 malformed_page(Some("p2")),
5054 serde_json::json!({ "records": [] })
5055 .to_string()
5056 .into_bytes(),
5057 ];
5058 let (base, _) = host_for(bodies, "skipping-walk-malformed.test").await;
5059 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5060 let (records, skipped) = client
5061 .list_all_records_skipping_within("c", &mut ByteBudget::new(MAX_LIST_BYTES))
5062 .await
5063 .expect("a malformed record failed a stranger's walk");
5064 assert_eq!(
5065 records.len(),
5066 1,
5067 "the good record behind the bad page was lost"
5068 );
5069 assert_eq!(skipped, 2, "skipped records were not counted");
5070 }
5071
5072 /// Pages of nothing but junk records, each ~`junk` bytes, with fresh
5073 /// cursors, so only the budget can stop the walk.
5074 fn junk_pages(n: usize, junk: usize) -> Vec<Vec<u8>> {
5075 (0..n)
5076 .map(|i| {
5077 serde_json::json!({
5078 "records": [{ "cid": "bafy", "value": "x".repeat(junk) }],
5079 "cursor": format!("p{}", i + 1),
5080 })
5081 .to_string()
5082 .into_bytes()
5083 })
5084 .collect()
5085 }
5086
5087 /// Review of #224: skipped records were never charged, so a stranger's
5088 /// repo serving junk pages walked all MAX_LIST_PAGES of them — gigabytes
5089 /// per poll — under a budget meant to stop at 128 MiB.
5090 #[tokio::test]
5091 async fn skipped_records_are_charged_against_the_budget() {
5092 let (base, _) = host_for(junk_pages(40, 256 * 1024), "junk-recent.test").await;
5093 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5094 let mut budget = ByteBudget::new(1024 * 1024);
5095 let walk = client
5096 .list_recent_matching_within("c", 100, &mut budget, 25, |_| true)
5097 .await
5098 .unwrap();
5099 assert!(
5100 !walk.complete,
5101 "40 pages of junk were walked to the end under a 1 MiB budget"
5102 );
5103
5104 let (base, _) = host_for(junk_pages(40, 256 * 1024), "junk-skipping.test").await;
5105 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5106 client
5107 .list_all_records_skipping_within("c", &mut ByteBudget::new(1024 * 1024))
5108 .await
5109 .expect_err("40 pages of junk were walked to the end under a 1 MiB budget");
5110 }
5111
5112 /// Review of #224: a page that skipped a record was charged its whole
5113 /// wire size AND its good records again, so a large publication near the
5114 /// budget failed only because one tiny malformed record sat beside it.
5115 #[tokio::test]
5116 async fn a_skipped_record_does_not_double_charge_its_page() {
5117 let page = |i: usize, with_bad: bool| {
5118 let mut records = vec![serde_json::json!({
5119 "uri": format!("at://did:plc:x/c/3lab{i}"), "value": "v".repeat(100_000)
5120 })];
5121 if with_bad {
5122 records.push(serde_json::json!({ "cid": "b", "value": {} }));
5123 }
5124 let mut body = serde_json::json!({ "records": records });
5125 if i < 4 {
5126 body["cursor"] = serde_json::json!(format!("p{}", i + 1));
5127 }
5128 body.to_string().into_bytes()
5129 };
5130 // A budget that holds the five clean pages with room to spare, and
5131 // less than twice that.
5132 let clean: Vec<_> = (0..5).map(|i| page(i, false)).collect();
5133 let (base, _) = host_for(clean, "double-charge-clean.test").await;
5134 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5135 let mut budget = ByteBudget::new(MAX_LIST_BYTES);
5136 let (clean_records, _) = client
5137 .list_all_records_skipping_within("c", &mut budget)
5138 .await
5139 .unwrap();
5140 let fits = budget.used() + budget.used() / 2;
5141
5142 let mixed: Vec<_> = (0..5).map(|i| page(i, true)).collect();
5143 let (base, _) = host_for(mixed, "double-charge-mixed.test").await;
5144 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5145 let (records, skipped) = client
5146 .list_all_records_skipping_within("c", &mut ByteBudget::new(fits))
5147 .await
5148 .expect("one tiny malformed record per page failed a walk that fits");
5149 assert_eq!(records.len(), clean_records.len());
5150 assert_eq!(skipped, 5);
5151
5152 // The documents walk, the same way: a page that skipped a record pays
5153 // its wire size once, not that and its kept records again.
5154 let mixed: Vec<_> = (0..5).map(|i| page(i, true)).collect();
5155 let (base, _) = host_for(mixed, "double-charge-recent.test").await;
5156 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5157 let walk = client
5158 .list_recent_matching_within("c", 100, &mut ByteBudget::new(fits), 25, |_| true)
5159 .await
5160 .unwrap();
5161 assert!(
5162 walk.complete,
5163 "one tiny malformed record per page cut short a walk that fits"
5164 );
5165 assert_eq!(walk.records.len(), clean_records.len());
5166 }
5167
5168 /// Third review of #224: charging a page that skipped a record only its
5169 /// wire size let a walk retain what it never paid for — a parsed record
5170 /// can hold up to 42x its wire size. Pages of dense `[[],[],…]` values, each
5171 /// with one malformed record beside them, retained 18x the budget.
5172 #[tokio::test]
5173 async fn a_page_that_skipped_a_record_still_pays_for_what_it_keeps() {
5174 let dense = format!("[{}]", vec!["[]"; 2000].join(","));
5175 let pages = |with_bad: bool| -> Vec<Vec<u8>> {
5176 (0..3)
5177 .map(|i| {
5178 let mut records: Vec<String> = (0..99)
5179 .map(|r| {
5180 format!(r#"{{"uri":"at://did:plc:x/c/3l{i}x{r}","value":{dense}}}"#)
5181 })
5182 .collect();
5183 if with_bad {
5184 records.push(r#"{"cid":"b","value":{}}"#.to_string());
5185 }
5186 let cursor = if i < 2 {
5187 format!(r#","cursor":"p{}""#, i + 1)
5188 } else {
5189 String::new()
5190 };
5191 format!(r#"{{"records":[{}]{cursor}}}"#, records.join(",")).into_bytes()
5192 })
5193 .collect()
5194 };
5195 const BUDGET: usize = 4 * 1024 * 1024;
5196 // Control: without the malformed records the walk is refused.
5197 let (base, _) = host_for(pages(false), "retained-control.test").await;
5198 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5199 client
5200 .list_all_records_skipping_within("c", &mut ByteBudget::new(BUDGET))
5201 .await
5202 .expect_err("control: the clean pages fit a budget they exceed");
5203
5204 let (base, _) = host_for(pages(true), "retained-skipping.test").await;
5205 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5206 client
5207 .list_all_records_skipping_within("c", &mut ByteBudget::new(BUDGET))
5208 .await
5209 .expect_err("one malformed record per page bought pages the budget refuses");
5210
5211 // The documents walk, with pages that each FIT the remaining budget
5212 // but together exceed it: only charging what the kept records retain
5213 // can stop it. (Pages each bigger than the budget are stopped by the
5214 // transient check alone and prove nothing about the charge.)
5215 let small_dense = format!("[{}]", vec!["[]"; 120].join(","));
5216 let fitting_pages: Vec<Vec<u8>> = (0..6)
5217 .map(|i| {
5218 let mut records: Vec<String> = (0..99)
5219 .map(|r| {
5220 format!(r#"{{"uri":"at://did:plc:x/c/3m{i}x{r}","value":{small_dense}}}"#)
5221 })
5222 .collect();
5223 records.push(r#"{"cid":"b","value":{}}"#.to_string());
5224 let cursor = if i < 5 {
5225 format!(r#","cursor":"q{}""#, i + 1)
5226 } else {
5227 String::new()
5228 };
5229 format!(r#"{{"records":[{}]{cursor}}}"#, records.join(",")).into_bytes()
5230 })
5231 .collect();
5232 let (base, _) = host_for(fitting_pages, "retained-recent.test").await;
5233 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5234 let walk = client
5235 .list_recent_matching_within("c", 10_000, &mut ByteBudget::new(BUDGET), 100, |_| true)
5236 .await
5237 .unwrap();
5238 let retained: usize = walk.records.iter().map(approx_bytes).sum();
5239 assert!(
5240 retained <= BUDGET,
5241 "retained {retained} under a {BUDGET}-byte budget"
5242 );
5243 assert!(
5244 !walk.complete,
5245 "pages totalling more than the budget were all kept"
5246 );
5247 }
5248
5249 /// The reader's own repo: this walk feeds `replace_sub_refs`, so skipping
5250 /// would drop a subscription silently. It refuses, by type.
5251 #[tokio::test]
5252 async fn the_own_repo_walk_refuses_a_page_with_a_malformed_record() {
5253 let (base, _) = host_for(vec![malformed_page(None)], "own-repo-malformed.test").await;
5254 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5255 let err = client
5256 .list_all_records("c")
5257 .await
5258 .expect_err("a page with a malformed record was accepted");
5259 let refused = err
5260 .downcast_ref::<MalformedRecords>()
5261 .unwrap_or_else(|| panic!("refused for the wrong reason: {err:#}"));
5262 assert_eq!(refused.count, 1);
5263 assert_eq!(refused.collection, "c");
5264 }
5265
5266 #[tokio::test]
5267 async fn the_sidecar_walk_refuses_a_page_with_a_malformed_record() {
5268 let body = serde_json::json!({
5269 "ok": true,
5270 "data": { "records": [
5271 { "uri": "at://did:plc:x/c/3labGOOD", "value": {} },
5272 { "cid": "bafy", "value": {} },
5273 ]}
5274 })
5275 .to_string()
5276 .into_bytes();
5277 let base = crate::net::tests::serve_bodies_in_sequence(vec![body]).await;
5278 let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
5279 let err = client
5280 .list_all_records("did:plc:x", "c")
5281 .await
5282 .expect_err("a page with a malformed record was accepted");
5283 assert!(
5284 err.downcast_ref::<MalformedRecords>().is_some(),
5285 "refused for the wrong reason: {err:#}"
5286 );
5287 }
5288
5289 /// A stranger's publication: skipping is right here, and the walk must keep
5290 /// paging past a page whose ONLY records were malformed — that page is not
5291 /// the end of the collection.
5292 #[tokio::test]
5293 async fn a_publication_walk_skips_malformed_records_and_keeps_paging() {
5294 let only_bad = serde_json::json!({
5295 "records": [{ "cid": "bafy", "value": {} }], "cursor": "p1"
5296 })
5297 .to_string()
5298 .into_bytes();
5299 let bodies = vec![
5300 only_bad,
5301 malformed_page(Some("p2")),
5302 serde_json::json!({ "records": [] })
5303 .to_string()
5304 .into_bytes(),
5305 ];
5306 let (base, _) = host_for(bodies, "publication-malformed.test").await;
5307 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5308 let mut budget = ByteBudget::new(MAX_LIST_BYTES);
5309 let walk = client
5310 .list_recent_matching_within("c", 100, &mut budget, 25, |_| true)
5311 .await
5312 .expect("a malformed record failed a stranger's publication walk");
5313 assert_eq!(
5314 walk.records.len(),
5315 1,
5316 "the good record behind the bad page was lost"
5317 );
5318 assert_eq!(walk.malformed, 2, "skipped records were not counted");
5319 assert!(
5320 walk.complete,
5321 "the walk stopped at a page of only malformed records"
5322 );
5323 }
5324
5325 /// **The budget is spent across pages, not reset by each one.**
5326 ///
5327 /// The single test this project most needed and did not have. Without it,
5328 /// moving the budget's construction inside the page loop — making the cap
5329 /// 200x weaker and effectively inert — passed every test in the suite.
5330 #[tokio::test]
5331 async fn a_refusing_walk_spends_its_budget_across_pages() {
5332 let (bodies, per_page) = paged_bodies(3, 4096, false);
5333 let (base, _) = host_for(bodies, "budget-accumulate.test").await;
5334 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5335
5336 let err = client
5337 .list_all_records_within("c", &mut ByteBudget::new(per_page * 2))
5338 .await
5339 .expect_err("three pages cannot fit in a two-page budget");
5340 let msg = format!("{err:#}");
5341 assert!(msg.contains("byte cap"), "wrong bound reported: {msg}");
5342 assert!(
5343 msg.contains("2 held"),
5344 "the walk did not keep exactly the two pages that fit: {msg}"
5345 );
5346 }
5347
5348 #[tokio::test]
5349 async fn a_truncating_walk_keeps_the_pages_that_fit() {
5350 let (bodies, per_page) = paged_bodies(3, 4096, false);
5351 let (base, _) = host_for(bodies, "budget-accumulate-trunc.test").await;
5352 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5353
5354 let walk = client
5355 .list_recent_matching_within("c", 100, &mut ByteBudget::new(per_page * 2), 100, |_| {
5356 true
5357 })
5358 .await
5359 .expect("an additive walk truncates rather than failing");
5360 assert_eq!(
5361 walk.records.len(),
5362 2,
5363 "the pages that fit were not kept, or the refused one was"
5364 );
5365 assert!(
5366 !walk.complete,
5367 "a walk stopped by the budget called itself complete"
5368 );
5369 }
5370
5371 /// **The budget must not bind before the record cap does, with room spare.**
5372 ///
5373 /// The walks that carry `MAX_LIST_RECORDS` REFUSE when a bound is hit, and a
5374 /// refusal drops the reader into `resolve_subscriptions`' fail-closed branch
5375 /// — so an account near the record cap would serve a stale projection on
5376 /// every poll, forever. The figure quoted in `MAX_LIST_BYTES`'s own comment
5377 /// is this calculation, and a review caught that figure being wrong by a
5378 /// factor of two because nothing computed it. This does.
5379 ///
5380 /// Double, not merely under: the margin is what stops a slightly longer
5381 /// title or one more optional field from turning a working account into a
5382 /// permanently failing one.
5383 #[test]
5384 fn a_full_subscription_repo_fits_the_budget_twice_over() {
5385 let record = record_of(serde_json::json!({
5386 "$type": "community.lexicon.rss.subscription",
5387 "url": "https://example.com/blog/feed.xml",
5388 "title": "Some Blog With A Longish Name",
5389 "siteUrl": "https://example.com/blog",
5390 "createdAt": "2026-07-11T09:30:00Z",
5391 "folder": "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/community.lexicon.rss.folder/3lab999",
5392 "fetchHint": "hourly",
5393 }));
5394 let per_record = approx_bytes(&record);
5395 let full_repo = per_record * MAX_LIST_RECORDS;
5396 assert!(
5397 full_repo * 2 <= MAX_LIST_BYTES,
5398 "a full repo charges {per_record} B x {MAX_LIST_RECORDS} = {} MB against a {} MB \
5399 budget — too close for a walk whose verdict is a refusal",
5400 full_repo / (1024 * 1024),
5401 MAX_LIST_BYTES / (1024 * 1024)
5402 );
5403 }
5404
5405 /// **Running out of pages is a refusal, not a short answer.**
5406 ///
5407 /// The three refusing walks fell out of `for _ in 0..MAX_LIST_PAGES` into a
5408 /// bare `Ok(out)`, so a repo bigger than the page budget returned a truncated
5409 /// list that looks exactly like a complete one. `resolve_subscriptions` needs
5410 /// an `Err` to take its fail-closed branch; given `Ok` it hands the short list
5411 /// to `replace_sub_refs`, which DELETEs the reader's whole `sub_ref`
5412 /// projection and reinserts only what it was given. Everything past the cap
5413 /// is gone from their account, on an ordinary poll, with no attacker.
5414 ///
5415 /// `extend_bounded`'s refusal cannot catch this: `MAX_LIST_PAGES` x the 100
5416 /// records we ask for is exactly `MAX_LIST_RECORDS`, so against any server
5417 /// that honours `limit` the page budget runs out first, every time.
5418 #[tokio::test]
5419 async fn a_walk_that_runs_out_of_pages_refuses_rather_than_truncating() {
5420 // One more page than the budget, every page still offering a cursor.
5421 let bodies: Vec<Vec<u8>> = (0..MAX_LIST_PAGES + 1)
5422 .map(|i| {
5423 serde_json::json!({
5424 "records": [{ "uri": format!("at://did:plc:x/c/3lab{i}"), "value": {} }],
5425 "cursor": format!("p{}", i + 1),
5426 })
5427 .to_string()
5428 .into_bytes()
5429 })
5430 .collect();
5431 let (base, _) = host_for(bodies, "pages-exhausted.test").await;
5432 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5433
5434 let err = client
5435 .list_all_records("c")
5436 .await
5437 .expect_err("a truncated list was returned as a complete one");
5438 let msg = format!("{err:#}");
5439 assert!(
5440 msg.contains("did not finish"),
5441 "failed for the wrong reason: {msg}"
5442 );
5443 }
5444
5445 /// **A walk that finishes cleanly across several pages still returns `Ok`.**
5446 ///
5447 /// The refusal's dangerous direction. Removing the flag's reset makes *every*
5448 /// multi-page walk refuse, which puts a reader with more than one page of
5449 /// records permanently into the fail-closed branch — and a review found that
5450 /// mutation surviving on the sidecar walk, which is the default backend,
5451 /// because nothing walked it to a clean finish and asserted success.
5452 #[tokio::test]
5453 async fn the_sidecar_walk_that_finishes_cleanly_returns_the_records() {
5454 let mut bodies: Vec<Vec<u8>> = (0..3)
5455 .map(|i| {
5456 serde_json::json!({
5457 "ok": true,
5458 "data": {
5459 "records": [{ "uri": format!("at://did:plc:x/c/3lab{i}"), "value": {} }],
5460 "cursor": format!("p{}", i + 1),
5461 }
5462 })
5463 .to_string()
5464 .into_bytes()
5465 })
5466 .collect();
5467 // **The terminator CARRIES a record.** Ending on an empty page left a
5468 // second mutation alive: drop the last page's records and the assertion
5469 // below still counts three, because the last page had none to drop. A
5470 // real PDS ends on a partial page, and that page's records are the ones
5471 // an off-by-one loses.
5472 bodies.push(
5473 serde_json::json!({
5474 "ok": true,
5475 "data": { "records": [{ "uri": "at://did:plc:x/c/3labLAST", "value": {} }] }
5476 })
5477 .to_string()
5478 .into_bytes(),
5479 );
5480 let base = crate::net::tests::serve_bodies_in_sequence(bodies).await;
5481 let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
5482 let records = client
5483 .list_all_records("did:plc:ewvi7nxzyoun6zhxrhs64oiz", "c")
5484 .await
5485 .expect("a walk that ran out of records is not a short list");
5486 assert_eq!(
5487 records.len(),
5488 4,
5489 "the pages that were served were not all kept"
5490 );
5491 assert!(
5492 records.iter().any(|r| r.uri.ends_with("3labLAST")),
5493 "the LAST page's records were dropped — the walk kept the right \
5494 count only because every page held one: {:?}",
5495 records.iter().map(|r| r.uri.as_str()).collect::<Vec<_>>(),
5496 );
5497 }
5498
5499 /// **The page cap is pinned exactly, not to within one.**
5500 ///
5501 /// `the_sidecar_walk_that_runs_out_of_pages_refuses` serves
5502 /// `MAX_LIST_PAGES + 1` pages, so a budget one page SHORT refuses too and
5503 /// that mutation survives it. A walk whose last allowed request is the
5504 /// terminating one must come back `Ok` — which fails the moment the loop
5505 /// allows one page fewer, and is the direction that costs a reader their
5506 /// subscriptions.
5507 #[tokio::test]
5508 async fn a_sidecar_walk_that_terminates_on_its_last_allowed_page_succeeds() {
5509 let mut bodies: Vec<Vec<u8>> = (0..MAX_LIST_PAGES - 1)
5510 .map(|i| {
5511 serde_json::json!({
5512 "ok": true,
5513 "data": {
5514 "records": [{ "uri": format!("at://did:plc:x/c/3lab{i}"), "value": {} }],
5515 "cursor": format!("p{}", i + 1),
5516 }
5517 })
5518 .to_string()
5519 .into_bytes()
5520 })
5521 .collect();
5522 // Request number `MAX_LIST_PAGES` — the last the loop allows — is the one
5523 // that terminates, and it carries a record of its own.
5524 bodies.push(
5525 serde_json::json!({
5526 "ok": true,
5527 "data": { "records": [{ "uri": "at://did:plc:x/c/3labLAST", "value": {} }] }
5528 })
5529 .to_string()
5530 .into_bytes(),
5531 );
5532 assert_eq!(bodies.len(), MAX_LIST_PAGES);
5533 let base = crate::net::tests::serve_bodies_in_sequence(bodies).await;
5534 let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
5535
5536 let records = client
5537 .list_all_records("did:plc:ewvi7nxzyoun6zhxrhs64oiz", "c")
5538 .await
5539 .expect("a walk that terminated inside its budget is not a short list");
5540 assert_eq!(
5541 records.len(),
5542 MAX_LIST_PAGES,
5543 "a walk that used its whole page budget and finished lost records",
5544 );
5545 }
5546
5547 /// **The direct walk's page cap, pinned exactly.**
5548 ///
5549 /// Twin of `a_sidecar_walk_that_terminates_on_its_last_allowed_page_succeeds`
5550 /// for the anonymous client. Verified needed: with only the `+ 1` refusal test
5551 /// above, `for _ in 0..MAX_LIST_PAGES - 1` left all 914 tests passing.
5552 #[tokio::test]
5553 async fn a_direct_walk_that_terminates_on_its_last_allowed_page_succeeds() {
5554 let mut bodies: Vec<Vec<u8>> = (0..MAX_LIST_PAGES - 1)
5555 .map(|i| {
5556 serde_json::json!({
5557 "records": [{ "uri": format!("at://did:plc:x/c/3lab{i}"), "value": {} }],
5558 "cursor": format!("p{}", i + 1),
5559 })
5560 .to_string()
5561 .into_bytes()
5562 })
5563 .collect();
5564 bodies.push(
5565 serde_json::json!({
5566 "records": [{ "uri": "at://did:plc:x/c/3labLAST", "value": {} }]
5567 })
5568 .to_string()
5569 .into_bytes(),
5570 );
5571 assert_eq!(bodies.len(), MAX_LIST_PAGES);
5572 let (base, _) = host_for(bodies, "last-allowed-page.test").await;
5573 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5574
5575 let records = client
5576 .list_all_records("c")
5577 .await
5578 .expect("a walk that terminated inside its budget is not a short list");
5579 assert_eq!(
5580 records.len(),
5581 MAX_LIST_PAGES,
5582 "a walk that used its whole page budget and finished lost records",
5583 );
5584 assert!(
5585 records.iter().any(|r| r.uri.ends_with("3labLAST")),
5586 "the LAST page's records were dropped",
5587 );
5588 }
5589
5590 /// **The TRUNCATING walk's page cap, pinned exactly — it reports completeness
5591 /// rather than refusing, so an off-by-one here is a silent short read.**
5592 ///
5593 /// A publication whose archive needs exactly the page budget to exhaust is
5594 /// `complete`; one page fewer makes it `complete = false`, which
5595 /// `store_publication` treats as a partial read. Verified needed:
5596 /// `for _ in 0..MAX_LIST_PAGES - 1` on this walk left all 914 tests passing.
5597 #[tokio::test]
5598 async fn a_truncating_walk_that_exhausts_on_its_last_allowed_page_is_complete() {
5599 let mut bodies: Vec<Vec<u8>> = (0..MAX_LIST_PAGES - 1)
5600 .map(|i| {
5601 serde_json::json!({
5602 "records": [{ "uri": format!("at://did:plc:x/c/3lab{i}"), "value": {} }],
5603 "cursor": format!("p{}", i + 1),
5604 })
5605 .to_string()
5606 .into_bytes()
5607 })
5608 .collect();
5609 bodies.push(
5610 serde_json::json!({
5611 "records": [{ "uri": "at://did:plc:x/c/3labLAST", "value": {} }]
5612 })
5613 .to_string()
5614 .into_bytes(),
5615 );
5616 assert_eq!(bodies.len(), MAX_LIST_PAGES);
5617 let (base, _) = host_for(bodies, "last-allowed-page-truncating.test").await;
5618 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5619
5620 // `max_records` well above what is served, so the cap under test is the
5621 // PAGE budget and not the record one.
5622 let walk = client
5623 .list_recent_matching("c", MAX_LIST_PAGES * 10, 1, |_| true)
5624 .await
5625 .expect("walk failed");
5626 assert_eq!(
5627 walk.records.len(),
5628 MAX_LIST_PAGES,
5629 "a walk that used its whole page budget and exhausted the collection \
5630 lost records",
5631 );
5632 assert!(
5633 walk.complete,
5634 "a collection that ran out on the last allowed page was reported as a \
5635 partial read, which is a starvation warning for a complete archive",
5636 );
5637 }
5638
5639 /// The sidecar walk refuses a short list too — and it is the default backend.
5640 #[tokio::test]
5641 async fn the_sidecar_walk_that_runs_out_of_pages_refuses() {
5642 let bodies: Vec<Vec<u8>> = (0..MAX_LIST_PAGES + 1)
5643 .map(|i| {
5644 serde_json::json!({
5645 "ok": true,
5646 "data": {
5647 "records": [{ "uri": format!("at://did:plc:x/c/3lab{i}"), "value": {} }],
5648 "cursor": format!("p{}", i + 1),
5649 }
5650 })
5651 .to_string()
5652 .into_bytes()
5653 })
5654 .collect();
5655 let base = crate::net::tests::serve_bodies_in_sequence(bodies).await;
5656 let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
5657 let err = client
5658 .list_all_records("did:plc:ewvi7nxzyoun6zhxrhs64oiz", "c")
5659 .await
5660 .expect_err("a truncated list was returned as a complete one");
5661 assert!(
5662 format!("{err:#}").contains("did not finish"),
5663 "failed for the wrong reason: {err:#}"
5664 );
5665 }
5666
5667 /// **A budget passed to two walks is spent by both of them.**
5668 ///
5669 /// The reason it is passed rather than constructed: a publication read runs
5670 /// a second walk while still holding the first's records, so two independent
5671 /// ceilings let one read hold twice the bound. Here the first walk spends
5672 /// the budget and the second finds it spent.
5673 #[tokio::test]
5674 async fn two_walks_sharing_a_budget_do_not_each_get_the_whole_of_it() {
5675 let (bodies, per_page) = paged_bodies(4, 4096, false);
5676 let (base, _) = host_for(bodies, "budget-shared.test").await;
5677 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5678 let mut budget = ByteBudget::new(per_page * 3);
5679
5680 let err = client
5681 .list_all_records_within("c", &mut budget)
5682 .await
5683 .expect_err("four pages cannot fit a three-page budget");
5684 assert!(format!("{err:#}").contains("3 held"), "{err:#}");
5685
5686 // Same budget, nothing left in it.
5687 let err = client
5688 .list_all_records_within("c", &mut budget)
5689 .await
5690 .expect_err("the second walk was handed a fresh ceiling");
5691 assert!(
5692 format!("{err:#}").contains("0 held"),
5693 "the second walk kept something out of an exhausted budget: {err:#}"
5694 );
5695 }
5696
5697 /// **A transient page has to fit what is LEFT of the budget.**
5698 ///
5699 /// Measuring it against the ceiling lets a walk that has already retained
5700 /// most of its budget hold a further ceiling's worth of page on top. The
5701 /// filter keeps nothing here, so the running total cannot stop the walk and
5702 /// only the remaining-budget comparison can.
5703 #[tokio::test]
5704 async fn a_transient_page_must_fit_what_is_left_not_the_ceiling() {
5705 let (bodies, per_page) = paged_bodies(3, 4096, false);
5706 let (base, _) = host_for(bodies, "budget-remaining.test").await;
5707 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5708
5709 // Ceiling of two and a half pages, two of them already spent.
5710 let mut budget = ByteBudget::new(per_page * 5 / 2);
5711 let spent = vec![
5712 record_of(serde_json::json!({ "t": "x".repeat(4096) })),
5713 record_of(serde_json::json!({ "t": "x".repeat(4096) })),
5714 ];
5715 assert!(budget.admit(&spent), "the pre-spend has to fit");
5716 assert!(
5717 budget.remaining() < per_page,
5718 "and has to leave less than a page"
5719 );
5720
5721 let walk = client
5722 .list_recent_matching_within("c", 100, &mut budget, 100, |_| false)
5723 .await
5724 .expect("an additive walk truncates rather than failing");
5725 assert!(
5726 !walk.complete,
5727 "a page larger than the remaining budget was walked past"
5728 );
5729 }
5730
5731 /// **A filter that keeps nothing must not let the walk run unbounded.**
5732 ///
5733 /// The running total charges what is kept, so a filter matching nothing
5734 /// charges zero and the total can never stop the walk. What it holds is
5735 /// another matter: each page is fully parsed before the filter sees it, and
5736 /// `read_capped`'s 8 MB bounds the wire, not the tree. Only the per-page
5737 /// bound stands between that and the box.
5738 #[tokio::test]
5739 async fn a_filter_that_keeps_nothing_still_cannot_outrun_the_budget() {
5740 let (bodies, per_page) = paged_bodies(3, 4096, false);
5741 let (base, _) = host_for(bodies, "budget-filtered.test").await;
5742 let client = PdsClient::anonymous(ssrf_test_client(), base, "did:plc:x");
5743
5744 let walk = client
5745 .list_recent_matching_within("c", 100, &mut ByteBudget::new(per_page / 2), 100, |_| {
5746 false
5747 })
5748 .await
5749 .expect("an additive walk truncates rather than failing");
5750 assert!(
5751 !walk.complete,
5752 "a page too large to hold was walked past because the filter dropped it"
5753 );
5754 assert!(walk.records.is_empty(), "the filter kept nothing");
5755 }
5756
5757 #[tokio::test]
5758 async fn the_sidecar_walk_spends_its_budget_across_pages() {
5759 let (bodies, per_page) = paged_bodies(3, 4096, true);
5760 let base = crate::net::tests::serve_bodies_in_sequence(bodies).await;
5761 let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
5762
5763 let err = client
5764 .list_all_records_within(
5765 "did:plc:ewvi7nxzyoun6zhxrhs64oiz",
5766 "app.feather.subscription",
5767 &mut ByteBudget::new(per_page * 2),
5768 )
5769 .await
5770 .expect_err("the sidecar walk was the one with no budget at all");
5771 let msg = format!("{err:#}");
5772 assert!(msg.contains("byte cap"), "wrong bound reported: {msg}");
5773 assert!(
5774 msg.contains("2 held"),
5775 "did not accumulate across pages: {msg}"
5776 );
5777 }
5778
5779 /// **Shapes that used to read as a healthy empty page.**
5780 ///
5781 /// Each of these was accepted by the `Value` route as `records: []`, and an
5782 /// empty page is not inert: `resolve_subscriptions` passes it to
5783 /// `replace_sub_refs`, which DELETEs the reader's projection and rewrites
5784 /// what it was handed. A page that is wrong in this direction costs them
5785 /// every feed.
5786 #[test]
5787 fn a_page_that_is_not_a_listing_is_never_read_as_an_empty_one() {
5788 for (label, body) in [
5789 (
5790 "a non-string error alongside records",
5791 &br#"{"error":404,"records":[]}"#[..],
5792 ),
5793 (
5794 "an object error alongside records",
5795 &br#"{"error":{"code":"x"},"records":[]}"#[..],
5796 ),
5797 (
5798 "a duplicated records key, the second one empty",
5799 &br#"{"records":[{"uri":"at://d/c/r","value":{}}],"records":[]}"#[..],
5800 ),
5801 ("an explicit null records", &br#"{"records":null}"#[..]),
5802 ] {
5803 assert!(
5804 parse_list_records(body).is_err(),
5805 "{label} was read as a page"
5806 );
5807 }
5808 }
5809
5810 /// **A non-string `error` is an envelope, and is reported as one.**
5811 ///
5812 /// Typing the field as a `String` made these fail as "invalid type" — the
5813 /// wrong reason for the exact shape the guard exists for, which is the same
5814 /// looseness that once let the guard be deleted unnoticed. So the reason is
5815 /// asserted, not just the refusal.
5816 #[test]
5817 fn a_non_string_error_is_reported_as_an_envelope() {
5818 for body in [
5819 &br#"{"error":404,"records":[]}"#[..],
5820 &br#"{"error":{"code":"x"},"records":[]}"#[..],
5821 &br#"{"error":[],"records":[]}"#[..],
5822 &br#"{"error":true,"records":[]}"#[..],
5823 ] {
5824 let err = parse_list_records(body)
5825 .expect_err("a non-string error envelope was read as an empty page");
5826 assert!(
5827 format!("{err:#}").contains("error envelope"),
5828 "{} failed for the wrong reason: {err:#}",
5829 String::from_utf8_lossy(body)
5830 );
5831 }
5832 // An empty name IS an envelope, as it was before this work: the route
5833 // this replaced keyed on `as_str`, so `Some("")` bailed. Exempting it
5834 // was a loosening made on speculation about proxy conventions, and a
5835 // loosening in this direction is a page accepted that used to be
5836 // refused.
5837 for body in [
5838 &br#"{"error":"","records":[]}"#[..],
5839 // Not zero on the wire, but zero once read: an exemption keyed on
5840 // `as_f64` swallowed anything that underflows.
5841 &br#"{"error":1e-400,"records":[]}"#[..],
5842 ] {
5843 let err = parse_list_records(body).expect_err("this is an envelope");
5844 assert!(
5845 format!("{err:#}").contains("error envelope"),
5846 "{} failed for the wrong reason: {err:#}",
5847 String::from_utf8_lossy(body)
5848 );
5849 }
5850 // The four spellings of "no error". `null` is what an ordinary listing
5851 // carries; `false` and integer `0` are a proxy convention, and refusing
5852 // those would fail a good page outright.
5853 for body in [
5854 &br#"{"error":null,"records":[]}"#[..],
5855 &br#"{"error":false,"records":[]}"#[..],
5856 &br#"{"error":0,"records":[]}"#[..],
5857 &br#"{"records":[]}"#[..],
5858 ] {
5859 assert!(
5860 parse_list_records(body).is_ok(),
5861 "{} is not an error envelope",
5862 String::from_utf8_lossy(body)
5863 );
5864 }
5865 }
5866
5867 /// **A non-string `error` is named by its type, never by its contents.**
5868 ///
5869 /// Rendering the value would serialise the whole attacker-chosen subtree
5870 /// before truncating it, allocating a full extra copy of up to the body cap
5871 /// — in a change whose purpose is cutting peak allocation. The earlier
5872 /// version of this did exactly that and the comment claimed otherwise.
5873 #[test]
5874 fn a_structured_error_is_named_by_its_type_not_serialised() {
5875 let payload = "s".repeat(20_000);
5876 let body = format!(r#"{{"error":{{"deep":"{payload}"}},"records":[]}}"#);
5877 let err = parse_list_records(body.as_bytes()).expect_err("an envelope is a refusal");
5878 let msg = format!("{err:#}");
5879 assert!(
5880 !msg.contains("ssss"),
5881 "the error's contents reached the message: {} chars",
5882 msg.len()
5883 );
5884 assert!(
5885 msg.contains("non-string error: object"),
5886 "it should name the shape instead: {msg}"
5887 );
5888 }
5889
5890 /// `data` absent is not `data` empty, on the sidecar envelope too.
5891 ///
5892 /// `{"ok":true}` is what a proxy makes of an unexpected upstream body, and
5893 /// reading it as a page of zero records is the wipe this whole family of
5894 /// guards exists to prevent.
5895 #[tokio::test]
5896 async fn the_sidecar_refuses_an_envelope_with_no_data() {
5897 for body in [&br#"{"ok":true}"#[..], &br#"{"ok":true,"data":{}}"#[..]] {
5898 let base = crate::net::tests::serve_body(body.to_vec()).await;
5899 let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
5900 let err = client
5901 .list_records(
5902 "did:plc:ewvi7nxzyoun6zhxrhs64oiz",
5903 "app.feather.subscription",
5904 None,
5905 None,
5906 )
5907 .await
5908 .expect_err("an envelope without a listing was read as an empty page");
5909 assert!(
5910 format!("{err:#}").contains("no records"),
5911 "{} failed for the wrong reason: {err:#}",
5912 String::from_utf8_lossy(body)
5913 );
5914 }
5915 }
5916
5917 /// **The sidecar gets the duplicated-key refusal too.**
5918 ///
5919 /// It was the one client still reading a listing through a `Value`, where a
5920 /// repeated key resolves last-wins — so a body carrying a second, empty
5921 /// `records` array read as a successful empty page, and an empty page on this
5922 /// path is `replace_sub_refs` deleting every `sub_ref` the reader has. It is
5923 /// also the default backend, so it was the one that mattered most.
5924 #[tokio::test]
5925 async fn the_sidecar_refuses_a_duplicated_records_key() {
5926 let base = crate::net::tests::serve_body(
5927 br#"{"ok":true,"data":{"records":[{"uri":"at://d/c/r","value":{}}],"records":[]}}"#
5928 .to_vec(),
5929 )
5930 .await;
5931 let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
5932 let err = client
5933 .list_records(
5934 "did:plc:ewvi7nxzyoun6zhxrhs64oiz",
5935 "app.feather.subscription",
5936 None,
5937 None,
5938 )
5939 .await
5940 .expect_err("a duplicated records key was read as an empty page");
5941 assert!(
5942 format!("{err:#}").contains("duplicate"),
5943 "failed for the wrong reason: {err:#}"
5944 );
5945 }
5946
5947 /// A non-string `message` must not fail an otherwise good page.
5948 /// **A listing has to be an object.**
5949 ///
5950 /// serde's derived `Deserialize` takes a struct positionally too, so with
5951 /// every field defaulted `[null,null,[]]` bound `records` to an empty vector
5952 /// and read as a healthy page — and a body with no keys defeats the envelope
5953 /// guard and the duplicated-key refusal at the same time, because neither has
5954 /// anything to look at. Fourteen bytes, and `replace_sub_refs` deletes every
5955 /// feed the reader has.
5956 #[test]
5957 fn a_listing_that_is_not_an_object_is_not_a_page() {
5958 for body in [
5959 &b"[null,null,[]]"[..],
5960 &b"[null,null,[],null]"[..],
5961 &br#"[null,null,[{"uri":"at://d/c/r","value":{}}],"c"]"#[..],
5962 &b"[]"[..],
5963 &br#""a string""#[..],
5964 &b"0"[..],
5965 &b"true"[..],
5966 ] {
5967 assert!(
5968 parse_list_records(body).is_err(),
5969 "{} was read as a page",
5970 String::from_utf8_lossy(body)
5971 );
5972 }
5973 }
5974
5975 /// **The rendering is bounded in bytes, whatever the input is made of.**
5976 ///
5977 /// Counting characters bounds nothing a log cares about: 120 astral-plane
5978 /// code points are 480 bytes. The invariant is on the output's byte length.
5979 #[test]
5980 fn a_truncated_message_is_bounded_in_bytes() {
5981 for (label, input) in [
5982 ("ascii", "e".repeat(50_000)),
5983 ("astral", "\u{1f600}".repeat(20_000)),
5984 (
5985 "mixed",
5986 format!("{}{}", "e".repeat(200), "\u{1f600}".repeat(200)),
5987 ),
5988 (
5989 "just over in bytes, just under in chars",
5990 "\u{1f600}".repeat(40),
5991 ),
5992 ] {
5993 let out = truncate_for_message(&input);
5994 assert!(
5995 out.len() <= 200,
5996 "{label}: rendered {} bytes from {} bytes of input",
5997 out.len(),
5998 input.len()
5999 );
6000 }
6001 // Short inputs pass through untouched.
6002 assert_eq!(truncate_for_message("Boom"), "Boom");
6003 }
6004
6005 /// **The sidecar has two envelope layers, and both are guards.**
6006 ///
6007 /// `page_from_body` covers the PDS's, which arrives inside `data`. The
6008 /// sidecar's own can say `ok:false` or carry its own `error` on a 200 while
6009 /// `data` still holds something that reads as a perfectly good empty page —
6010 /// and an empty page here is `replace_sub_refs` deleting every feed.
6011 #[tokio::test]
6012 async fn the_sidecar_refuses_its_own_error_envelope_on_a_2xx() {
6013 for body in [
6014 &br#"{"ok":false,"error":"ExpiredToken","data":{"records":[]}}"#[..],
6015 &br#"{"ok":true,"error":"ExpiredToken","data":{"records":[]}}"#[..],
6016 &br#"{"ok":false,"data":{"records":[]}}"#[..],
6017 ] {
6018 let base = crate::net::tests::serve_body(body.to_vec()).await;
6019 let client = SidecarClient::new(Client::new(), base.clone(), base, "secret");
6020 let err = client
6021 .list_records(
6022 "did:plc:ewvi7nxzyoun6zhxrhs64oiz",
6023 "app.feather.subscription",
6024 None,
6025 None,
6026 )
6027 .await
6028 .expect_err("the sidecar's own envelope was read as a page");
6029 let msg = format!("{err:#}");
6030 assert!(
6031 msg.contains("sidecar answered 2xx"),
6032 "{} failed for the wrong reason: {msg}",
6033 String::from_utf8_lossy(body)
6034 );
6035 }
6036 }
6037
6038 /// **An unknown field's contents are still validated.**
6039 ///
6040 /// `IgnoredAny` skips without validating, so a body that is not valid JSON at
6041 /// all read as a healthy empty page where the route this replaced refused it.
6042 #[test]
6043 fn an_unknown_field_holding_invalid_json_is_not_a_page() {
6044 for body in [
6045 &b"{\"records\":[],\"x\":\"\xff\xfe\"}"[..],
6046 &br#"{"records":[],"x":"\ud800"}"#[..],
6047 ] {
6048 assert!(
6049 parse_list_records(body).is_err(),
6050 "{} was read as a page",
6051 String::from_utf8_lossy(body)
6052 );
6053 }
6054 }
6055
6056 #[test]
6057 fn a_non_string_message_does_not_cost_the_page() {
6058 let page = parse_list_records(br#"{"records":[],"message":5,"cursor":"c"}"#)
6059 .expect("message carries no guard; typing it strictly failed whole listings");
6060 assert_eq!(page.cursor.as_deref(), Some("c"));
6061 }
6062
6063 /// The name that reaches the log is bounded, because the PDS chooses it.
6064 #[test]
6065 fn an_enormous_error_name_is_truncated_before_it_reaches_a_log() {
6066 let huge = "e".repeat(50_000);
6067 let body = format!(r#"{{"error":"{huge}","records":[]}}"#);
6068 let err = parse_list_records(body.as_bytes()).expect_err("an envelope is a refusal");
6069 let msg = format!("{err:#}");
6070 assert!(
6071 msg.len() < 400,
6072 "the error message carried {} bytes of attacker-chosen text",
6073 msg.len()
6074 );
6075 assert!(
6076 msg.contains("50000 bytes"),
6077 "it should say what it dropped: {msg}"
6078 );
6079
6080 // Astral-plane code points: the bound must hold in BYTES, because a log
6081 // line is bytes. Counting characters made this four times the stated cap.
6082 let wide = "\u{1f600}".repeat(20_000);
6083 let body = format!(r#"{{"error":"{wide}","message":"{wide}","records":[]}}"#);
6084 let err = parse_list_records(body.as_bytes()).expect_err("an envelope is a refusal");
6085 let msg = format!("{err:#}");
6086 assert!(
6087 msg.len() < 400,
6088 "a wide-character error rendered {} bytes",
6089 msg.len()
6090 );
6091 }
6092
6093 // -- TID rkeys ----------------------------------------------------------
6094
6095 #[test]
6096 fn tid_rkeys_are_13_char_s32_and_monotonic() {
6097 let mut gen = TidGenerator::new();
6098 let mut prev: Option<String> = None;
6099 for _ in 0..1000 {
6100 let tid = gen.next();
6101 assert_eq!(tid.len(), 13, "a TID is 13 s32 chars");
6102 assert!(
6103 tid.bytes().all(|b| S32_ALPHABET.contains(&b)),
6104 "TID {tid} uses only the s32 alphabet"
6105 );
6106 if let Some(p) = &prev {
6107 assert!(*p < tid, "TIDs must be strictly increasing ({p} < {tid})");
6108 }
6109 prev = Some(tid);
6110 }
6111 }
6112
6113 #[test]
6114 fn tid_rkeys_are_valid_atproto_record_keys() {
6115 // atproto rkey charset: [A-Za-z0-9._~:-], length 1..=512, not "."/"..".
6116 let mut gen = TidGenerator::new();
6117 let tid = gen.next();
6118 assert!(is_valid_rkey(&tid), "{tid:?}");
6119 assert!(tid
6120 .bytes()
6121 .all(|b| b.is_ascii_alphanumeric() || matches!(b, b'.' | b'_' | b'~' | b':' | b'-')));
6122 }
6123
6124 #[test]
6125 fn tid_values_round_trip_through_the_decoder() {
6126 // The decoder is the inverse of the encoder across the whole range a
6127 // TID can hold, boundaries included.
6128 let max_tid = (0x001f_ffff_ffff_ffffu64 << 10) | 0x3ff;
6129 for v in [0u64, 1, 31, 32, 1023, 1024, 1_000_000, max_tid] {
6130 let encoded = encode_s32_tid(v);
6131 assert_eq!(
6132 decode_s32_tid(&encoded),
6133 Some(v),
6134 "{v} encoded to {encoded}, which did not decode back"
6135 );
6136 }
6137
6138 // **A round trip alone proves too little.** Encoder and decoder share
6139 // the alphabet, so swapping two of its symbols round-trips perfectly
6140 // and still reads every real record key wrong. These two are the
6141 // known answer: a record key from a real atproto repo, and the value
6142 // it holds, computed independently of this code.
6143 assert_eq!(
6144 decode_s32_tid("3jzfcijpj2z2a"),
6145 Some(1_728_652_679_052_295_174)
6146 );
6147 assert_eq!(encode_s32_tid(1_728_652_679_052_295_174), "3jzfcijpj2z2a");
6148 assert_eq!(
6149 decode_s32_tid("3jzfcijpj2z2a").map(|raw| raw >> 10),
6150 Some(1_688_137_381_887_007),
6151 "that key was written at 2023-06-30T15:03:01.887007Z"
6152 );
6153 }
6154
6155 #[test]
6156 fn the_first_tid_of_a_generator_decodes_to_the_microsecond_it_was_minted() {
6157 let micros = || {
6158 std::time::SystemTime::now()
6159 .duration_since(std::time::UNIX_EPOCH)
6160 .map(|d| d.as_micros() as u64)
6161 .unwrap_or(0)
6162 };
6163 // The FIRST `next()` only. `TidGenerator` bumps a TID to `last + 1`
6164 // to stay strictly increasing, and on a generator whose clock id is
6165 // already at its maximum that carry lands in the timestamp bits — so a
6166 // later TID can decode a microsecond or two past when it was really
6167 // minted. A fresh generator has `last: 0`, where the bump cannot fire.
6168 let before = micros();
6169 let tid = TidGenerator::new().next();
6170 let after = micros();
6171 let raw = decode_s32_tid(&tid).expect("a generated TID must decode");
6172 let minted = raw >> 10;
6173 assert!(
6174 (before..=after).contains(&minted),
6175 "TID {tid} decoded to {minted}, outside the {before}..={after} window it was minted in"
6176 );
6177 }
6178
6179 #[test]
6180 fn the_decoder_rejects_strings_that_are_not_13_char_s32_values() {
6181 for rkey in [
6182 "", // empty
6183 "self", // the common non-TID rkey
6184 "3jzfcijpj2z2", // 12 chars: one short
6185 "3jzfcijpj2z2aa", // 14 chars: one long
6186 "3jzfcijpj2z2A", // uppercase is outside the s32 alphabet
6187 "3jzfcijpj2z-a", // a legal rkey character, but not an s32 one
6188 "3jzfcijpj2z2!", // not a legal rkey character at all
6189 "c222222222222", // decodes with bit 63 set: the reserved top bit
6190 "k222222222222", // decodes past 64 bits entirely
6191 "zzzzzzzzzzzzz", // the largest 13-char s32 string
6192 ] {
6193 assert_eq!(
6194 decode_s32_tid(rkey),
6195 None,
6196 "{rkey:?} is not a 13-character s32 value"
6197 );
6198 }
6199 }
6200
6201 /// **The window is not a slug detector, and this is what that costs.**
6202 ///
6203 /// A 13-character slug beginning `3` decodes into the last few years just
6204 /// as a record key does, and nothing in the string tells them apart. These
6205 /// are read as dates, and pinning that here is the honest alternative to a
6206 /// doc comment claiming otherwise. The damage is bounded: a wrong date is
6207 /// an ordinary past instant that ages, sweeps and is outranked normally.
6208 #[test]
6209 fn a_slug_that_decodes_inside_the_window_is_read_as_a_date() {
6210 for (slug, reads_as) in [
6211 ("3hoursinparis", "2020-11-24T08:17:26Z"),
6212 ("3ideasforjune", "2021-08-12T00:19:38Z"),
6213 ("3jokesaweekly", "2023-02-12T15:50:26Z"),
6214 ] {
6215 assert_eq!(
6216 tid_timestamp(slug).map(crate::feed::fmt_time),
6217 Some(reads_as.to_string()),
6218 "{slug} is indistinguishable from a record key written then"
6219 );
6220 }
6221 }
6222
6223 #[test]
6224 fn a_tid_minted_slightly_ahead_of_our_clock_is_still_believed() {
6225 let now = chrono::Utc::now();
6226 let of = |at: chrono::DateTime<chrono::Utc>| {
6227 encode_s32_tid((at.timestamp_micros() as u64) << 10)
6228 };
6229 assert!(
6230 tid_timestamp(&of(now + chrono::Duration::seconds(2))).is_some(),
6231 "a PDS two seconds fast must not leave a fresh document undated"
6232 );
6233 assert_eq!(
6234 tid_timestamp(&of(now + chrono::Duration::hours(1))),
6235 None,
6236 "an hour ahead is a broken clock or a slug, not skew"
6237 );
6238 }
6239
6240 #[test]
6241 fn a_tid_timestamp_is_bounded_at_both_ends() {
6242 let now = chrono::Utc::now();
6243 let of = |micros: i64| encode_s32_tid((micros as u64) << 10);
6244
6245 // A TID minted now dates to now.
6246 let fresh = TidGenerator::new().next();
6247 let dated = tid_timestamp(&fresh).expect("a freshly minted TID has a timestamp");
6248 assert!(
6249 (now - chrono::Duration::minutes(1)..=now + chrono::Duration::minutes(1))
6250 .contains(&dated),
6251 "{fresh} dated to {dated}, not to now ({now})"
6252 );
6253
6254 // Before atproto existed: not a date.
6255 assert_eq!(
6256 tid_timestamp(&of(TID_FLOOR_MICROS - 1)),
6257 None,
6258 "a TID predating atproto must not date an entry"
6259 );
6260 assert!(
6261 tid_timestamp(&of(TID_FLOOR_MICROS)).is_some(),
6262 "the floor itself is a real instant"
6263 );
6264
6265 // In the future: not a date. A slug of 13 s32 characters lands here,
6266 // which is the case this bound exists for.
6267 let far_future = (now + chrono::Duration::days(365)).timestamp_micros();
6268 assert_eq!(
6269 tid_timestamp(&of(far_future)),
6270 None,
6271 "a TID from the future must not date an entry"
6272 );
6273 assert_eq!(
6274 tid_timestamp("abcdefghijklm"),
6275 None,
6276 "a 13-character slug decodes to the year 2192; it is not a date"
6277 );
6278 }
6279
6280 #[test]
6281 fn s32_encoding_is_ascending_for_ascending_values() {
6282 // The whole point of s32: numeric order == lexicographic string order.
6283 assert!(encode_s32_tid(1) < encode_s32_tid(2));
6284 assert!(encode_s32_tid(31) < encode_s32_tid(32));
6285 assert!(encode_s32_tid(1_000_000) < encode_s32_tid(1_000_001));
6286 // Ordering holds all the way to the largest real TID value (a 53-bit
6287 // microsecond timestamp shifted into bits 63..10, plus the clock id).
6288 let max_tid = (0x001f_ffff_ffff_ffffu64 << 10) | 0x3ff;
6289 assert!(encode_s32_tid(max_tid - 1) < encode_s32_tid(max_tid));
6290 }
6291 /// **Exceeding the record cap is an ERROR, not a silent truncation.**
6292 ///
6293 /// The page cap bounds how many requests a walk makes; it bounds the
6294 /// accumulated memory only if the server honours `limit=100`, and a host we
6295 /// did not choose has no obligation to. A review measured an 8 MB page
6296 /// holding ~95 000 minimal records and retaining 23 MB as
6297 /// `Vec<RecordEntry>` — 200 such pages is gigabytes on a 512 MB box.
6298 ///
6299 /// Truncating instead would be worse than the OOM it prevents. The caller
6300 /// of the live walk is `resolve_subscriptions`, whose result feeds
6301 /// `replace_sub_refs` — a `DELETE` plus reinsert of exactly what it was
6302 /// handed. A short list there is not a short list, it is **revoked access**
6303 /// to the feeds that fell off the end. That is the failure PR #167 was
6304 /// closed for reintroducing, so this returns `Err` and lets the existing
6305 /// fail-closed branch serve the last-known projection.
6306 #[test]
6307 fn exceeding_the_record_cap_is_an_error_not_a_truncation() {
6308 let page = |n: usize| -> Vec<RecordEntry> {
6309 (0..n)
6310 .map(|i| RecordEntry {
6311 uri: format!("at://did:plc:x/c/{i}"),
6312 cid: None,
6313 value: serde_json::Value::Null,
6314 })
6315 .collect()
6316 };
6317
6318 let mut out = page(90);
6319 let err = extend_bounded(&mut out, page(20), 100, "c")
6320 .expect_err("a page past the cap was accepted");
6321 let msg = format!("{err:#}");
6322 assert!(msg.contains("100"), "the cap is not named: {msg}");
6323 assert_eq!(
6324 out.len(),
6325 90,
6326 "the partial page was kept — a truncated list must not survive the error"
6327 );
6328 }
6329
6330 #[test]
6331 fn accumulating_within_the_cap_succeeds() {
6332 let page = |n: usize| -> Vec<RecordEntry> {
6333 (0..n)
6334 .map(|i| RecordEntry {
6335 uri: format!("at://did:plc:x/c/{i}"),
6336 cid: None,
6337 value: serde_json::Value::Null,
6338 })
6339 .collect()
6340 };
6341 let mut out = Vec::new();
6342 extend_bounded(&mut out, page(60), 100, "c").unwrap();
6343 extend_bounded(&mut out, page(40), 100, "c").unwrap();
6344 assert_eq!(out.len(), 100, "exactly the cap must be allowed");
6345 }
6346
6347 // -- applyWrites chunking (#240) ----------------------------------------
6348
6349 /// Every `applyWrites` call a fake saw, as the `writes` array it carried.
6350 pub(crate) type ApplyWritesLog = Arc<std::sync::Mutex<Vec<Vec<Value>>>>;
6351
6352 /// A fake that answers `applyWrites` the way a strict PDS does, on BOTH
6353 /// shapes this crate sends it in: the PDS's own
6354 /// `/xrpc/com.atproto.repo.applyWrites` (the direct and OAuth clients) and
6355 /// the sidecar's `/internal/repo` (`action: "applyWrites"`).
6356 ///
6357 /// It refuses what the reference PDS refuses — more than 200 writes
6358 /// (`InvalidRequest: Too many writes. Max: 200`, from
6359 /// `packages/pds/src/api/com/atproto/repo/applyWrites.ts`) and a body over
6360 /// the 150 KiB `jsonLimit` every reference PDS before atproto#4989 applied
6361 /// to it — so a test that sends an unchunked batch fails the way production
6362 /// would, rather than passing against a fake that accepts anything.
6363 ///
6364 /// Every call is logged, refused or not, so a test can assert that a later
6365 /// chunk was never SENT. `fail_call` (1-based) answers that call with a 500.
6366 pub(crate) async fn serve_apply_writes(fail_call: Option<usize>) -> (String, ApplyWritesLog) {
6367 use axum::body::Bytes;
6368 use axum::http::{StatusCode as Status, Uri};
6369 use axum::response::IntoResponse;
6370
6371 const PRE_4989_JSON_LIMIT: usize = 150 * 1024;
6372 let log: ApplyWritesLog = Arc::default();
6373 let sink = Arc::clone(&log);
6374 let app = axum::Router::new()
6375 .fallback(move |uri: Uri, body: Bytes| {
6376 let sink = Arc::clone(&sink);
6377 async move {
6378 let sidecar = uri.path() == "/internal/repo";
6379 let reply = |status: Status, error: &str, message: &str| {
6380 let body = if sidecar {
6381 json!({ "ok": false, "error": error, "message": message, "status": status.as_u16() })
6382 } else {
6383 json!({ "error": error, "message": message })
6384 };
6385 (status, axum::Json(body)).into_response()
6386 };
6387 let parsed: Value = serde_json::from_slice(&body).unwrap_or(Value::Null);
6388 // Anything else a handler sends on the way (a folder
6389 // listing, say) is answered empty and not logged, so the
6390 // log and `fail_call` count `applyWrites` calls only.
6391 let Some(writes) = parsed["writes"].as_array().cloned() else {
6392 let empty = json!({ "records": [] });
6393 return if sidecar {
6394 axum::Json(json!({ "ok": true, "data": empty })).into_response()
6395 } else {
6396 axum::Json(empty).into_response()
6397 };
6398 };
6399 let call = {
6400 let mut calls = sink.lock().unwrap();
6401 calls.push(writes.clone());
6402 calls.len()
6403 };
6404 if body.len() > PRE_4989_JSON_LIMIT {
6405 return reply(
6406 Status::PAYLOAD_TOO_LARGE,
6407 "PayloadTooLarge",
6408 "request entity too large",
6409 );
6410 }
6411 if writes.len() > 200 {
6412 return reply(
6413 Status::BAD_REQUEST,
6414 "InvalidRequest",
6415 "Too many writes. Max: 200",
6416 );
6417 }
6418 if fail_call == Some(call) {
6419 return reply(
6420 Status::INTERNAL_SERVER_ERROR,
6421 "InternalServerError",
6422 "boom",
6423 );
6424 }
6425 let data =
6426 json!({ "commit": { "cid": "bafycommit", "rev": "3l" }, "results": [] });
6427 if sidecar {
6428 axum::Json(json!({ "ok": true, "data": data })).into_response()
6429 } else {
6430 axum::Json(data).into_response()
6431 }
6432 }
6433 })
6434 // The fake must see an oversized body to refuse it, not have axum
6435 // refuse it first at its own 2 MB default.
6436 .layer(axum::extract::DefaultBodyLimit::disable());
6437 let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
6438 let addr = listener.local_addr().unwrap();
6439 tokio::spawn(async move { axum::serve(listener, app).await.unwrap() });
6440 (format!("http://{addr}"), log)
6441 }
6442
6443 /// `n` distinct vetted subscriptions, in a known order.
6444 fn vetted_subs(n: usize) -> Vec<crate::vetted::VettedSubscription> {
6445 (0..n)
6446 .map(|i| {
6447 crate::vetted::VettedSubscription::new(&lexicon::Subscription::new(
6448 format!("https://f{i}.example/feed.xml"),
6449 "2026-07-12T00:00:00.000Z",
6450 ))
6451 })
6452 .collect()
6453 }
6454
6455 /// The `rkey` of every write a run of calls carried, flattened in send order.
6456 pub(crate) fn sent_rkeys(log: &ApplyWritesLog) -> Vec<String> {
6457 log.lock()
6458 .unwrap()
6459 .iter()
6460 .flatten()
6461 .map(|w| w["rkey"].as_str().unwrap_or_default().to_string())
6462 .collect()
6463 }
6464
6465 /// How many writes each call carried, in send order.
6466 pub(crate) fn call_sizes(log: &ApplyWritesLog) -> Vec<usize> {
6467 log.lock().unwrap().iter().map(Vec::len).collect()
6468 }
6469
6470 const CHUNK_DID: &str = "did:plc:ewvi7nxzyoun6zhxrhs64oiz";
6471
6472 fn chunk_sidecar(base: &str) -> SidecarClient {
6473 SidecarClient::new(Client::new(), base, base, "secret")
6474 }
6475
6476 /// **201 writes are two calls, 200 then 1, in order.** The OPML import used
6477 /// to send every feed in one `applyWrites`, which the reference PDS refuses
6478 /// past 200 — so any import over 200 feeds failed outright.
6479 #[tokio::test]
6480 async fn sidecar_bulk_add_of_201_is_two_calls_in_order() {
6481 let (base, log) = serve_apply_writes(None).await;
6482 let rkeys = chunk_sidecar(&base)
6483 .add_subscriptions_bulk(CHUNK_DID, &vetted_subs(201))
6484 .await
6485 .expect("a 201-feed import must succeed against a PDS that caps at 200");
6486
6487 assert_eq!(call_sizes(&log), vec![200, 1]);
6488 let urls: Vec<String> = log
6489 .lock()
6490 .unwrap()
6491 .iter()
6492 .flatten()
6493 .map(|w| w["value"]["url"].as_str().unwrap().to_string())
6494 .collect();
6495 let expected: Vec<String> = (0..201)
6496 .map(|i| format!("https://f{i}.example/feed.xml"))
6497 .collect();
6498 assert_eq!(urls, expected, "ops must keep input order across chunks");
6499 assert_eq!(
6500 sent_rkeys(&log),
6501 rkeys,
6502 "the returned rkeys are the ones written, in order"
6503 );
6504 }
6505
6506 /// 500 — the default per-DID cap, so the largest import a stock instance
6507 /// sends — is three calls.
6508 #[tokio::test]
6509 async fn sidecar_bulk_add_of_500_is_three_calls() {
6510 let (base, log) = serve_apply_writes(None).await;
6511 chunk_sidecar(&base)
6512 .add_subscriptions_bulk(CHUNK_DID, &vetted_subs(500))
6513 .await
6514 .expect("bulk write");
6515 assert_eq!(call_sizes(&log), vec![200, 200, 100]);
6516 }
6517
6518 /// Exactly the limit is ONE call: chunking must not split a batch that fits.
6519 #[tokio::test]
6520 async fn sidecar_bulk_add_of_exactly_200_is_one_call() {
6521 let (base, log) = serve_apply_writes(None).await;
6522 chunk_sidecar(&base)
6523 .add_subscriptions_bulk(CHUNK_DID, &vetted_subs(200))
6524 .await
6525 .expect("bulk write");
6526 assert_eq!(call_sizes(&log), vec![200]);
6527 }
6528
6529 /// Nothing to write is no call at all — the behaviour both live clients
6530 /// already had (the sidecar refuses an empty `writes[]` with a 400).
6531 #[tokio::test]
6532 async fn sidecar_bulk_add_of_nothing_sends_nothing() {
6533 let (base, log) = serve_apply_writes(None).await;
6534 let rkeys = chunk_sidecar(&base)
6535 .add_subscriptions_bulk(CHUNK_DID, &[])
6536 .await
6537 .expect("an empty import is not an error");
6538 assert!(rkeys.is_empty());
6539 assert!(
6540 call_sizes(&log).is_empty(),
6541 "an empty batch must not be sent"
6542 );
6543 }
6544
6545 /// **A failed chunk stops the run.** Chunk 2 of 3 fails: the call returns
6546 /// an error, and chunk 3 is never sent — sending it would commit writes
6547 /// after a gap, which no caller could describe as a prefix.
6548 #[tokio::test]
6549 async fn sidecar_bulk_add_stops_at_the_first_failed_chunk() {
6550 let (base, log) = serve_apply_writes(Some(2)).await;
6551 let err = chunk_sidecar(&base)
6552 .add_subscriptions_bulk(CHUNK_DID, &vetted_subs(500))
6553 .await
6554 .expect_err("a failed chunk must fail the call");
6555 assert_eq!(call_sizes(&log), vec![200, 200], "chunk 3 must NOT be sent");
6556 assert!(
6557 format!("{err:#}").contains("boom"),
6558 "the PDS's reason was lost: {err:#}"
6559 );
6560 let progress = ApplyWritesIncomplete::of(&err).expect("the error says how far it got");
6561 assert_eq!(
6562 (progress.landed, progress.in_doubt, progress.total),
6563 (200, 200, 500),
6564 "chunk 1 landed, chunk 2 is in doubt, chunk 3 was never sent"
6565 );
6566 // A plain `%err` log line still names the PDS's reason.
6567 assert!(err.to_string().contains("boom"), "{err}");
6568 }
6569
6570 /// A batch that fit in one call fails exactly as it did before chunking:
6571 /// the same message, and nothing landed.
6572 #[tokio::test]
6573 async fn a_single_chunk_failure_reads_as_it_always_did() {
6574 let (base, _log) = serve_apply_writes(Some(1)).await;
6575 let err = chunk_sidecar(&base)
6576 .add_subscriptions_bulk(CHUNK_DID, &vetted_subs(3))
6577 .await
6578 .expect_err("refused");
6579 let progress = ApplyWritesIncomplete::of(&err).expect("progress");
6580 assert_eq!((progress.landed, progress.in_doubt), (0, 3));
6581 assert!(
6582 !err.to_string().contains("applyWrites call"),
6583 "a one-call batch has no progress to report: {err}"
6584 );
6585 assert_eq!(
6586 format!("{err:#}").matches("boom").count(),
6587 1,
6588 "the cause must not print twice in the chain: {err:#}"
6589 );
6590 }
6591
6592 fn create_ops(n: usize, value_bytes: usize) -> Vec<WriteOp> {
6593 (0..n)
6594 .map(|i| WriteOp::Create {
6595 collection: lexicon::nsid::SUBSCRIPTION.to_string(),
6596 rkey: Some(format!("rk{i:05}")),
6597 value: json!({ "pad": "x".repeat(value_bytes) }),
6598 })
6599 .collect()
6600 }
6601
6602 /// The op-count boundary, on small ops the byte bound never touches.
6603 #[test]
6604 // A one-range Vec IS the expected value here: one chunk spanning the batch.
6605 #[allow(clippy::single_range_in_vec_init)]
6606 fn chunks_split_at_200_ops_and_not_before() {
6607 for (n, want) in [
6608 (0, vec![]),
6609 (1, vec![0..1]),
6610 (200, vec![0..200]),
6611 (201, vec![0..200, 200..201]),
6612 (500, vec![0..200, 200..400, 400..500]),
6613 ] {
6614 assert_eq!(chunk_writes(&create_ops(n, 8)), want, "{n} ops");
6615 }
6616 }
6617
6618 /// **The byte bound splits under 200 ops, and every chunk fits it.**
6619 #[test]
6620 fn chunks_split_on_bytes_and_each_fits() {
6621 let ops = create_ops(40, 10_000);
6622 let chunks = chunk_writes(&ops);
6623 assert!(chunks.len() > 1, "400 KB went out as {chunks:?}");
6624 let mut next = 0;
6625 for range in &chunks {
6626 assert_eq!(range.start, next, "chunks must be consecutive: {chunks:?}");
6627 next = range.end;
6628 let body = json!({
6629 "repo": CHUNK_DID,
6630 "writes": ops[range.clone()].iter().map(WriteOp::to_json).collect::<Vec<_>>(),
6631 })
6632 .to_string();
6633 assert!(
6634 body.len() <= APPLY_WRITES_MAX_BYTES + 200,
6635 "a {}-byte body for {range:?}",
6636 body.len()
6637 );
6638 }
6639 assert_eq!(next, ops.len(), "every op, once");
6640 }
6641
6642 /// **The byte bound does not split an ordinary import.** 200 subscriptions
6643 /// with a title and a site URL each fit one call, so the common OPML
6644 /// import pays one round trip per 200 feeds and no more — the figure the
6645 /// bound's doc comment rests on.
6646 #[test]
6647 fn a_realistic_200_feed_import_is_one_call() {
6648 let ops: Vec<WriteOp> = (0..200)
6649 .map(|i| {
6650 let mut sub = lexicon::Subscription::new(
6651 format!("https://www.example-blog-{i:03}.com/feeds/posts/default.xml"),
6652 "2026-07-12T00:00:00.000Z",
6653 );
6654 sub.title = Some(format!("An Example Blog With A Fairly Long Title {i}"));
6655 sub.site_url = Some(format!("https://www.example-blog-{i:03}.com/"));
6656 WriteOp::Create {
6657 collection: lexicon::nsid::SUBSCRIPTION.to_string(),
6658 rkey: Some(format!("3lab2c4d5e{i:03}")),
6659 value: serde_json::to_value(&sub).unwrap(),
6660 }
6661 })
6662 .collect();
6663 // ~76 KB measured.
6664 let bytes: usize = ops.iter().map(|op| op.to_json().to_string().len()).sum();
6665 assert_eq!(chunk_writes(&ops), vec![0..200], "{bytes} bytes");
6666 }
6667
6668 /// An op bigger than the bound cannot be split: it goes alone, and the
6669 /// ops around it are not dragged into its call.
6670 #[test]
6671 fn an_oversized_op_goes_alone() {
6672 let mut ops = create_ops(3, 8);
6673 ops.insert(1, create_ops(1, APPLY_WRITES_MAX_BYTES + 1).remove(0));
6674 assert_eq!(chunk_writes(&ops), vec![0..1, 1..2, 2..4]);
6675 }
6676
6677 /// Read-state cursors as large as the lexicon allows: 1,000 ids each.
6678 fn big_cursors(n: usize) -> Vec<(String, ReadState, bool)> {
6679 (0..n)
6680 .map(|i| {
6681 let mut state = ReadState::new(
6682 format!("https://f{i}.example/feed.xml"),
6683 None,
6684 "2026-07-12T00:00:00.000Z",
6685 );
6686 state.read_ids = (0..ReadState::MAX_IDS)
6687 .map(|j| format!("https://f{i}.example/posts/{j:04}/an-entry-permalink"))
6688 .collect();
6689 (format!("rk{i:04}"), state, i % 2 == 0)
6690 })
6691 .collect()
6692 }
6693
6694 /// **The byte bound splits a batch well under 200 ops.** Ten full cursors
6695 /// are ~500 KB: under the op cap, over every older reference PDS's 150 KiB
6696 /// body limit. Each call must fit, and together they must carry every
6697 /// cursor, once, in order.
6698 #[tokio::test]
6699 async fn sidecar_read_state_flush_splits_on_bytes_under_200_ops() {
6700 let (base, log) = serve_apply_writes(None).await;
6701 let cursors = big_cursors(10);
6702 chunk_sidecar(&base)
6703 .flush_read_states(CHUNK_DID, &cursors)
6704 .await
6705 .expect("a byte-heavy flush must succeed in chunks");
6706
6707 let sizes = call_sizes(&log);
6708 assert!(
6709 sizes.len() > 1,
6710 "a ~500 KB flush went out as one call: {sizes:?}"
6711 );
6712 let want: Vec<String> = cursors.iter().map(|(rkey, _, _)| rkey.clone()).collect();
6713 assert_eq!(sent_rkeys(&log), want, "every cursor, once, in order");
6714 }
6715
6716 /// The direct (app-password) client gets the same chunking: its
6717 /// `flush_read_states` is the same op list on a different wire.
6718 #[tokio::test]
6719 async fn direct_client_read_state_flush_of_201_is_two_calls() {
6720 let (base, log) = serve_apply_writes(None).await;
6721 let port: u16 = base.rsplit(':').next().unwrap().parse().unwrap();
6722 let host = format!("chunk-direct-{port}.test");
6723 crate::net::test_host_override(&host, std::net::SocketAddr::from(([127, 0, 0, 1], port)));
6724 let client = PdsClient::new(
6725 ssrf_test_client(),
6726 format!("http://{host}:{port}"),
6727 CHUNK_DID,
6728 Auth::Session(SessionAuth {
6729 did: CHUNK_DID.to_string(),
6730 handle: None,
6731 access_jwt: "jwt".to_string(),
6732 refresh_jwt: None,
6733 }),
6734 );
6735 let cursors: Vec<(String, ReadState, bool)> = (0..201)
6736 .map(|i| {
6737 let feed = format!("https://f{i}.example/feed.xml");
6738 let state = ReadState::new(feed, None, "2026-07-12T00:00:00.000Z");
6739 (format!("rk{i:04}"), state, true)
6740 })
6741 .collect();
6742 client.flush_read_states(&cursors).await.expect("flush");
6743 assert_eq!(call_sizes(&log), vec![200, 1]);
6744 let want: Vec<String> = cursors.iter().map(|(rkey, _, _)| rkey.clone()).collect();
6745 assert_eq!(sent_rkeys(&log), want);
6746 }
6747
6748 // -- #149: compare-and-swap putRecord ------------------------------------
6749
6750 pub(crate) const SWAP_DID: &str = "did:plc:ewvi7nxzyoun6zhxrhs64oiz";
6751 pub(crate) const OLD_CID: &str = "bafyreigh2akiscaildcqabsyg3dfr6chu3fgpregiymsck7e7aqa4s52zy";
6752
6753 /// A server answering every request with `status` and `body`, logging each
6754 /// request's JSON body (`Null` for a GET). Returns its loopback base URL,
6755 /// a hostname routed to it (the guarded clients refuse loopback), and the
6756 /// log.
6757 pub(crate) async fn serve_status_json(
6758 status: u16,
6759 body: Value,
6760 ) -> (String, String, Arc<std::sync::Mutex<Vec<Value>>>) {
6761 use axum::response::IntoResponse as _;
6762 let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
6763 let addr = listener.local_addr().unwrap();
6764 let host = format!("swap-{}.atproto.test", addr.port());
6765 crate::net::test_host_override(&host, addr);
6766 let log: Arc<std::sync::Mutex<Vec<Value>>> = Arc::default();
6767 let sink = Arc::clone(&log);
6768 let app = axum::Router::new().fallback(move |raw: axum::body::Bytes| {
6769 let sink = Arc::clone(&sink);
6770 let body = body.clone();
6771 async move {
6772 sink.lock()
6773 .unwrap()
6774 .push(serde_json::from_slice(&raw).unwrap_or(Value::Null));
6775 (
6776 axum::http::StatusCode::from_u16(status).unwrap(),
6777 axum::Json(body),
6778 )
6779 .into_response()
6780 }
6781 });
6782 tokio::spawn(async move { axum::serve(listener, app).await.unwrap() });
6783 (
6784 format!("http://{addr}"),
6785 format!("http://{host}:{}", addr.port()),
6786 log,
6787 )
6788 }
6789
6790 fn swap_direct_client(pds: &str) -> PdsClient {
6791 PdsClient::new(
6792 ssrf_test_client(),
6793 pds,
6794 SWAP_DID,
6795 Auth::Session(SessionAuth {
6796 did: SWAP_DID.to_string(),
6797 handle: None,
6798 access_jwt: "jwt".to_string(),
6799 refresh_jwt: None,
6800 }),
6801 )
6802 }
6803
6804 pub(crate) fn swap_sub() -> crate::vetted::VettedSubscription {
6805 crate::vetted::VettedSubscription::new(&Subscription::new(
6806 "https://example.com/feed.xml",
6807 "2024-03-01T00:00:00.000Z",
6808 ))
6809 }
6810
6811 pub(crate) fn write_ok() -> Value {
6812 json!({
6813 "uri": format!("at://{SWAP_DID}/{}/rk", lexicon::nsid::SUBSCRIPTION),
6814 "cid": "bafyreiafter",
6815 })
6816 }
6817
6818 /// **The direct client puts `swapRecord` on the wire when given one, and
6819 /// leaves the key out entirely when not.** Asserted on the request the PDS
6820 /// received: a value accepted by the method and dropped on the way out is
6821 /// the failure mode, and only the bytes can show it.
6822 #[tokio::test]
6823 async fn direct_put_record_sends_swap_record_only_when_given() {
6824 let (_, pds, log) = serve_status_json(200, write_ok()).await;
6825 let client = swap_direct_client(&pds);
6826
6827 client
6828 .put_record(
6829 lexicon::nsid::SUBSCRIPTION,
6830 "rk",
6831 &swap_sub(),
6832 Some(OLD_CID),
6833 )
6834 .await
6835 .expect("put with a swap");
6836 client
6837 .put_record(lexicon::nsid::SUBSCRIPTION, "rk", &swap_sub(), None)
6838 .await
6839 .expect("put without a swap");
6840
6841 let sent = log.lock().unwrap().clone();
6842 assert_eq!(sent.len(), 2, "{sent:?}");
6843 assert_eq!(sent[0]["rkey"], "rk", "captured no usable body: {sent:?}");
6844 assert_eq!(
6845 sent[0]["swapRecord"], OLD_CID,
6846 "the CID the caller read never reached the PDS: {}",
6847 sent[0]
6848 );
6849 assert_eq!(sent[1]["rkey"], "rk");
6850 assert!(
6851 sent[1].get("swapRecord").is_none(),
6852 "no swap was asked for, so none may be sent: {}",
6853 sent[1]
6854 );
6855 }
6856
6857 /// The same, for the sidecar client — whose body is the sidecar's own
6858 /// `/internal/repo` shape, not XRPC's.
6859 #[tokio::test]
6860 async fn sidecar_put_sends_swap_record_only_when_given() {
6861 let (base, _, log) =
6862 serve_status_json(200, json!({ "ok": true, "data": write_ok() })).await;
6863 let client = SidecarClient::new(Client::new(), &base, &base, "secret");
6864
6865 client
6866 .update_subscription(SWAP_DID, "rk", &swap_sub(), Some(OLD_CID))
6867 .await
6868 .expect("put with a swap");
6869 client
6870 .update_subscription(SWAP_DID, "rk", &swap_sub(), None)
6871 .await
6872 .expect("put without a swap");
6873
6874 let sent = log.lock().unwrap().clone();
6875 assert_eq!(sent.len(), 2, "{sent:?}");
6876 assert_eq!(
6877 sent[0]["action"], "put",
6878 "captured no usable body: {sent:?}"
6879 );
6880 assert_eq!(sent[0]["swapRecord"], OLD_CID, "{}", sent[0]);
6881 assert_eq!(sent[1]["action"], "put");
6882 assert!(sent[1].get("swapRecord").is_none(), "{}", sent[1]);
6883 }
6884
6885 /// What the reference PDS answers a stale `swapRecord` with.
6886 pub(crate) fn invalid_swap_xrpc() -> Value {
6887 json!({ "error": "InvalidSwap", "message": format!("Record was at {OLD_CID}") })
6888 }
6889
6890 /// The same refusal, after the sidecar has wrapped it.
6891 pub(crate) fn invalid_swap_sidecar() -> Value {
6892 json!({
6893 "ok": false,
6894 "error": "InvalidSwap",
6895 "message": format!("Record was at {OLD_CID}"),
6896 "status": 400,
6897 })
6898 }
6899
6900 /// **A refused swap is recognised from each client's real error.** Driven
6901 /// through the clients against a server answering what the PDS answers,
6902 /// not built by hand, so a client that changes how it wraps a rejection
6903 /// breaks this rather than the rename that depends on it.
6904 #[tokio::test]
6905 async fn an_invalid_swap_is_recognised_from_both_clients_errors() {
6906 let (_, pds, _) = serve_status_json(400, invalid_swap_xrpc()).await;
6907 let err = swap_direct_client(&pds)
6908 .put_record(
6909 lexicon::nsid::SUBSCRIPTION,
6910 "rk",
6911 &swap_sub(),
6912 Some(OLD_CID),
6913 )
6914 .await
6915 .expect_err("the PDS refused the swap");
6916 assert!(is_invalid_swap(&err), "direct client: {err:#}");
6917
6918 let (base, _, _) = serve_status_json(400, invalid_swap_sidecar()).await;
6919 let err = SidecarClient::new(Client::new(), &base, &base, "secret")
6920 .update_subscription(SWAP_DID, "rk", &swap_sub(), Some(OLD_CID))
6921 .await
6922 .expect_err("the PDS refused the swap");
6923 assert!(is_invalid_swap(&err), "sidecar client: {err:#}");
6924 }
6925
6926 /// **Every other failure is NOT a lost race.** Reading one of these as
6927 /// `InvalidSwap` would re-read and retry a write the PDS refused for a
6928 /// reason a retry cannot fix — or tell the reader someone else edited a
6929 /// record nobody touched.
6930 #[tokio::test]
6931 async fn other_failures_are_not_an_invalid_swap() {
6932 for (status, body) in [
6933 (
6934 400,
6935 json!({ "error": "InvalidRequest", "message": "bad record" }),
6936 ),
6937 (400, json!({ "error": "RecordNotFound" })),
6938 (500, json!({ "error": "InternalServerError" })),
6939 (401, json!({ "error": "AuthRequired" })),
6940 // The name in the MESSAGE, not the error field, is not the signal.
6941 (
6942 400,
6943 json!({ "error": "InvalidRequest", "message": "InvalidSwap" }),
6944 ),
6945 ] {
6946 let (_, pds, _) = serve_status_json(status, body.clone()).await;
6947 let err = swap_direct_client(&pds)
6948 .put_record(
6949 lexicon::nsid::SUBSCRIPTION,
6950 "rk",
6951 &swap_sub(),
6952 Some(OLD_CID),
6953 )
6954 .await
6955 .expect_err("refused");
6956 assert!(!is_invalid_swap(&err), "{status} {body}: {err:#}");
6957 }
6958
6959 // A transport failure: nothing listening.
6960 let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
6961 let dead = format!("http://{}", listener.local_addr().unwrap());
6962 drop(listener);
6963 let err = SidecarClient::new(Client::new(), &dead, &dead, "secret")
6964 .update_subscription(SWAP_DID, "rk", &swap_sub(), Some(OLD_CID))
6965 .await
6966 .expect_err("nothing is listening");
6967 assert!(!is_invalid_swap(&err), "transport: {err:#}");
6968
6969 // A string that merely SAYS it is not a typed refusal.
6970 assert!(!is_invalid_swap(&anyhow::anyhow!("InvalidSwap")));
6971 }
6972
6973 /// The typed refusal is found wherever it sits in the chain: under a
6974 /// context, and under [`ApplyWritesIncomplete`], whose `source()` skips its
6975 /// cause's top error — the trap `readstate::may_be_existence_mismatch` fell
6976 /// into on the merge with #240.
6977 #[tokio::test]
6978 async fn an_invalid_swap_is_found_under_context_and_a_split_batch() {
6979 let refusal = || AtProtoError::Xrpc {
6980 status: StatusCode::BAD_REQUEST,
6981 error: "InvalidSwap".to_string(),
6982 message: None,
6983 };
6984 assert!(is_invalid_swap(&anyhow::Error::new(refusal())));
6985 assert!(is_invalid_swap(
6986 &anyhow::Error::new(refusal()).context("com.atproto.repo.putRecord failed")
6987 ));
6988
6989 let writes: Vec<WriteOp> = (0..3)
6990 .map(|i| WriteOp::Delete {
6991 collection: lexicon::nsid::SUBSCRIPTION.to_string(),
6992 rkey: format!("rk{i}"),
6993 })
6994 .collect();
6995 let err = apply_writes_chunked(&writes, |_| async { Err(refusal().into()) })
6996 .await
6997 .expect_err("the only call failed");
6998 assert!(ApplyWritesIncomplete::of(&err).is_some(), "{err:#}");
6999 assert!(is_invalid_swap(&err), "{err:#}");
7000 }
7001
7002 /// Two subscription records with distinct CIDs, as `listRecords` returns.
7003 pub(crate) fn two_subs_page() -> Value {
7004 let rec = |rkey: &str, cid: &str, url: &str| {
7005 json!({
7006 "uri": format!("at://{SWAP_DID}/{}/{rkey}", lexicon::nsid::SUBSCRIPTION),
7007 "cid": cid,
7008 "value": {
7009 "$type": lexicon::nsid::SUBSCRIPTION,
7010 "url": url,
7011 "createdAt": "2024-03-01T00:00:00.000Z",
7012 },
7013 })
7014 };
7015 json!({ "records": [
7016 rec("rk-a", "bafyreiaaaaaaaaaa", "https://a.example/feed.xml"),
7017 rec("rk-b", "bafyreibbbbbbbbbb", "https://b.example/feed.xml"),
7018 ] })
7019 }
7020
7021 /// Asserts a CID listing paired each record with ITS CID.
7022 pub(crate) fn assert_listed_with_cids(listed: &[(String, Option<String>, Subscription)]) {
7023 let got: Vec<(&str, Option<&str>, &str)> = listed
7024 .iter()
7025 .map(|(rkey, cid, sub)| (rkey.as_str(), cid.as_deref(), sub.url.as_str()))
7026 .collect();
7027 assert_eq!(
7028 got,
7029 vec![
7030 (
7031 "rk-a",
7032 Some("bafyreiaaaaaaaaaa"),
7033 "https://a.example/feed.xml"
7034 ),
7035 (
7036 "rk-b",
7037 Some("bafyreibbbbbbbbbb"),
7038 "https://b.example/feed.xml"
7039 ),
7040 ],
7041 "each record must come back with the CID it was listed at"
7042 );
7043 }
7044
7045 /// Two folder records at distinct CIDs, each carrying a field this build
7046 /// does not know (#268).
7047 pub(crate) fn two_folders_page() -> Value {
7048 let rec = |rkey: &str, cid: &str, name: &str| {
7049 json!({
7050 "uri": format!("at://{SWAP_DID}/{}/{rkey}", lexicon::nsid::FOLDER),
7051 "cid": cid,
7052 "value": {
7053 "$type": lexicon::nsid::FOLDER,
7054 "name": name,
7055 "position": 3,
7056 "createdAt": "2024-01-01T00:00:00.000Z",
7057 "color": "#abc",
7058 },
7059 })
7060 };
7061 json!({ "records": [
7062 rec("fk-a", "bafyreifolderaaaa", "Tech"),
7063 rec("fk-b", "bafyreifolderbbbb", "News"),
7064 ] })
7065 }
7066
7067 /// Asserts a folder CID listing paired each record with ITS CID, and kept
7068 /// the record whole.
7069 pub(crate) fn assert_folders_listed_with_cids(listed: &[(String, Option<String>, Folder)]) {
7070 let got: Vec<(&str, Option<&str>, &str)> = listed
7071 .iter()
7072 .map(|(rkey, cid, f)| (rkey.as_str(), cid.as_deref(), f.name.as_str()))
7073 .collect();
7074 assert_eq!(
7075 got,
7076 vec![
7077 ("fk-a", Some("bafyreifolderaaaa"), "Tech"),
7078 ("fk-b", Some("bafyreifolderbbbb"), "News"),
7079 ],
7080 "each folder must come back with the CID it was listed at"
7081 );
7082 for (_, _, folder) in listed {
7083 assert_eq!(folder.position, Some(3));
7084 assert_eq!(folder.created_at, "2024-01-01T00:00:00.000Z");
7085 assert_eq!(folder.extra.get("color"), Some(&json!("#abc")));
7086 }
7087 }
7088
7089 #[tokio::test]
7090 async fn both_clients_list_folders_with_the_cid_each_was_read_at() {
7091 let (_, pds, _) = serve_status_json(200, two_folders_page()).await;
7092 let listed = swap_direct_client(&pds)
7093 .list_folders_with_cids()
7094 .await
7095 .expect("direct listing");
7096 assert_folders_listed_with_cids(&listed);
7097
7098 let (base, _, _) =
7099 serve_status_json(200, json!({ "ok": true, "data": two_folders_page() })).await;
7100 let listed = SidecarClient::new(Client::new(), &base, &base, "secret")
7101 .list_folders_with_cids(SWAP_DID)
7102 .await
7103 .expect("sidecar listing");
7104 assert_folders_listed_with_cids(&listed);
7105 }
7106
7107 #[tokio::test]
7108 async fn both_clients_list_subscriptions_with_the_cid_each_was_read_at() {
7109 let (_, pds, _) = serve_status_json(200, two_subs_page()).await;
7110 let listed = swap_direct_client(&pds)
7111 .list_subscriptions_with_cids()
7112 .await
7113 .expect("direct listing");
7114 assert_listed_with_cids(&listed);
7115
7116 let (base, _, _) =
7117 serve_status_json(200, json!({ "ok": true, "data": two_subs_page() })).await;
7118 let listed = SidecarClient::new(Client::new(), &base, &base, "secret")
7119 .list_subscriptions_with_cids(SWAP_DID)
7120 .await
7121 .expect("sidecar listing");
7122 assert_listed_with_cids(&listed);
7123 }
7124}