feather_reader/store.rs
1//! SQLite persistence layer (via `sqlx`, runtime queries).
2//!
3//! FeatherReader keeps the source of truth for *what a user follows* and *their
4//! read-position* in the user's own atproto PDS (as `community.lexicon.rss.*`
5//! records). This module is the **local per-DID cache + debounce
6//! buffer**: a single SQLite file that holds
7//!
8//! * `feeds` + `entries` — a shared cache of feed metadata and articles, keyed by
9//! feed URL / feed-native GUID and **shared across every DID** that follows the
10//! same feed (many users on one instance don't multiply fetch load), and
11//! * `entry_state` + `read_cursor` — per-DID read/star state and the per-feed
12//! read cursor that the (v1.1) batched flusher syncs up to the PDS.
13//!
14//! All queries here are **runtime** queries (`sqlx::query` / `sqlx::query_as`),
15//! not the compile-time `query!` macros — so the crate builds with no
16//! `DATABASE_URL` and no offline metadata. Schema creation is idempotent
17//! (`CREATE TABLE IF NOT EXISTS`) and runs inside [`init`].
18//!
19//! Errors propagate as [`anyhow::Result`]; nothing in the non-test paths panics.
20
21use anyhow::{Context, Result};
22use sqlx::sqlite::{SqliteConnectOptions, SqlitePool, SqlitePoolOptions};
23use sqlx::{ConnectOptions, FromRow, Row};
24use std::str::FromStr;
25
26use crate::config::Config;
27
28/// Typed failure modes for [`redeem_code`]. Distinct variants so the web layer
29/// can map each to the right user-facing message / HTTP status without string
30/// matching. Everything else (a real SQLite error) still propagates as
31/// [`anyhow::Error`] out of the `Result`.
32#[derive(Debug, thiserror::Error, PartialEq, Eq)]
33pub enum RedeemError {
34 /// No invite code with that value exists.
35 #[error("invite code not found")]
36 NotFound,
37 /// The code exists but is past its `expires_at` (or already flipped to
38 /// `expired`).
39 #[error("invite code expired")]
40 Expired,
41 /// The code has already been redeemed (or is otherwise not `active`).
42 #[error("invite code already redeemed")]
43 AlreadyRedeemed,
44 /// The closed-beta seat cap ([`Config`]'s `FEATHERREADER_BETA_CAP`) is full.
45 #[error("beta is at capacity")]
46 CapacityFull,
47}
48
49/// The SQLite connection pool type the rest of the crate refers to as
50/// [`Pool`]. A thin alias over `SqlitePool` so [`crate::AppState`] and the web
51/// layer name one stable type; if the backend ever changes, this is the single
52/// place to swap it.
53pub type Pool = SqlitePool;
54
55/// A cached syndication feed, shared across all DIDs that subscribe to its URL.
56///
57/// This mirrors the PDS-side `community.lexicon.rss.subscription.url`; the row is
58/// created/updated by the poller, never owned by a single user.
59#[derive(Debug, Clone, FromRow, PartialEq, Eq)]
60pub struct Feed {
61 pub id: i64,
62 pub url: String,
63 pub title: Option<String>,
64 pub site_url: Option<String>,
65 /// HTTP `ETag` from the last successful fetch, for conditional GET.
66 pub etag: Option<String>,
67 /// HTTP `Last-Modified` from the last successful fetch, for conditional GET.
68 pub last_modified: Option<String>,
69 /// When we last polled this feed (RFC3339), or `None` if never.
70 pub last_polled: Option<String>,
71 /// When this feed is next due to be polled (RFC3339), or `None`.
72 pub next_poll: Option<String>,
73 /// Count of consecutive poll FAILURES since the last success/304. Drives the
74 /// exponential poll backoff (reset to 0 on any success or 304).
75 #[sqlx(default)]
76 pub consecutive_errors: i64,
77}
78
79/// A cached article/item belonging to a [`Feed`]. Shared cache (not per-DID).
80#[derive(Debug, Clone, FromRow, PartialEq, Eq)]
81pub struct Entry {
82 pub id: i64,
83 pub feed_id: i64,
84 /// Feed-native GUID/id, unique within a feed (used for dedup on re-fetch).
85 pub guid: String,
86 pub url: Option<String>,
87 pub title: Option<String>,
88 pub author: Option<String>,
89 /// Publication time as reported by the feed (RFC3339), or `None`.
90 pub published: Option<String>,
91 /// Article body HTML, **already sanitized** (ammonia) before it reaches here.
92 pub content_html: Option<String>,
93 /// When FeatherReader first fetched/stored this entry (RFC3339).
94 pub fetched_at: String,
95}
96
97/// One row of a LIST view — deliberately **without** `content_html`.
98///
99/// The list queries used to be `SELECT e.*` into [`Entry`], which carries the
100/// sanitized article body. The body is essentially the whole of a cached entry
101/// (measured: 11.9 KB/entry), and no list surface has ever rendered it — the
102/// reader's `EntryRow` reads id, title, feed title, date, read, starred and
103/// link, and nothing else. So every article on every page load was read off
104/// disk, allocated, and dropped unexamined. On a 512 MB box with 250 concurrent
105/// requests permitted, one reader with a large backlog could ask for hundreds of
106/// megabytes in a single handler, and the resulting OOM/restart looked like a
107/// healthy machine that simply fell over.
108///
109/// `read` / `starred` come from the same `LEFT JOIN` that filters the view, so a
110/// caller does not have to fetch the whole unread or starred set a second time
111/// just to decorate the rows it is showing.
112///
113/// [`Entry`] is still the right type for the single-entry reader, which is the
114/// one surface that genuinely needs the body.
115#[derive(Debug, Clone, FromRow, PartialEq, Eq)]
116pub struct EntryListRow {
117 pub id: i64,
118 pub feed_id: i64,
119 /// Feed-native GUID — used to match a cached entry against a PDS saved record.
120 pub guid: String,
121 pub url: Option<String>,
122 pub title: Option<String>,
123 pub published: Option<String>,
124 /// This DID's read bit. `false` when there is no `entry_state` row at all.
125 pub read: bool,
126 /// This DID's star bit. `false` when there is no `entry_state` row at all.
127 pub starred: bool,
128}
129
130/// Which list [`list_entries`] (and its siblings) is producing.
131#[derive(Debug, Clone, Copy, PartialEq, Eq)]
132pub enum ListView {
133 /// No `entry_state` row for this DID, or one with `read = 0`.
134 Unread,
135 /// An `entry_state` row with `starred = 1`.
136 Starred,
137 /// Every subscribed entry, read or not.
138 All,
139}
140
141impl ListView {
142 /// The `WHERE` fragment that selects this view, given `s` as the per-DID
143 /// `entry_state` LEFT JOIN alias.
144 fn predicate(self) -> &'static str {
145 match self {
146 // An entry with no state row is unread — hence LEFT JOIN + COALESCE
147 // rather than a join that would drop never-touched entries.
148 ListView::Unread => "COALESCE(s.read, 0) = 0",
149 ListView::Starred => "COALESCE(s.starred, 0) = 1",
150 ListView::All => "1 = 1",
151 }
152 }
153}
154
155/// Per-`(did, entry)` read/star state — the fast in-session working copy that the
156/// batched flusher later syncs to the PDS as a per-feed read cursor.
157#[derive(Debug, Clone, FromRow, PartialEq, Eq)]
158pub struct EntryState {
159 pub did: String,
160 pub entry_id: i64,
161 pub read: bool,
162 pub starred: bool,
163 pub updated_at: String,
164}
165
166/// Per-`(did, feed_url)` read cursor — the local mirror of the PDS
167/// `community.lexicon.rss.readState` record plus flush bookkeeping.
168///
169/// `read_ids` / `unread_ids` are stored as JSON arrays of entry ids (the two
170/// bounded exception sets around the `read_through` high-water-mark); `dirty`
171/// marks that local `entry_state` has changed since the last PDS flush.
172#[derive(Debug, Clone, FromRow, PartialEq, Eq)]
173pub struct ReadCursor {
174 pub did: String,
175 pub feed_url: String,
176 /// High-water-mark (RFC3339): every entry seen/published `<=` this is read.
177 pub read_through: Option<String>,
178 /// JSON array of entry ids newer than `read_through` that are also read.
179 pub read_ids: String,
180 /// JSON array of entry ids older than `read_through` explicitly kept unread.
181 pub unread_ids: String,
182 /// Set when `entry_state` changed since the last flush (debounce trigger).
183 pub dirty: bool,
184 /// Whether this cursor's `readState` record has been CREATED in the PDS yet.
185 /// The first flush of a feed must emit an `applyWrites#create` (an `#update`
186 /// errors on a record that does not pre-exist, and applyWrites is atomic
187 /// per-repo, so one not-yet-created cursor would drop the whole DID batch).
188 /// Flipped to `true` on the flush that creates it.
189 #[sqlx(default)]
190 pub pds_created: bool,
191 pub updated_at: String,
192}
193
194/// The `network_stat` key the relay adoption probe writes under.
195///
196/// Lives here, beside [`NetworkStat`], because **both** the writer (the
197/// scheduler's probe, compiled into the binary) and the reader (`web::about`,
198/// compiled into the library) name it — a literal in either place would be two
199/// strings free to drift apart.
200pub const ADOPTION_STAT_KEY: &str = "adoption.subscription";
201
202/// One relay's observation of how many repos hold a collection
203/// (`design/NETWORK-SPEC.md` §4.3). A projection: droppable, rebuildable from
204/// the network, and never read by anything on the reading path.
205#[derive(Debug, Clone, FromRow, PartialEq, Eq)]
206pub struct NetworkStat {
207 /// The metric key, e.g. [`ADOPTION_STAT_KEY`].
208 pub key: String,
209 /// The relay base URL the number came from.
210 pub source: String,
211 /// The observed count.
212 pub value: i64,
213 /// Set when the probe hit its page cap: the value is a floor, not a count.
214 pub truncated: bool,
215 /// When the observation was taken (RFC3339, UTC).
216 pub observed_at: String,
217}
218
219/// New-feed payload for [`upsert_feed`] (id is assigned by SQLite).
220#[derive(Debug, Clone, Default)]
221pub struct NewFeed {
222 pub url: String,
223 pub title: Option<String>,
224 pub site_url: Option<String>,
225 pub etag: Option<String>,
226 pub last_modified: Option<String>,
227 pub last_polled: Option<String>,
228 pub next_poll: Option<String>,
229}
230
231/// New-entry payload for [`insert_entries`] (id is assigned by SQLite,
232/// `fetched_at` defaults to "now" when not supplied).
233#[derive(Debug, Clone, Default)]
234pub struct NewEntry {
235 pub guid: String,
236 pub url: Option<String>,
237 pub title: Option<String>,
238 pub author: Option<String>,
239 pub published: Option<String>,
240 /// Already-sanitized HTML.
241 pub content_html: Option<String>,
242 /// Optional explicit fetch time (RFC3339); defaults to now if `None`.
243 pub fetched_at: Option<String>,
244}
245
246/// The SQLite schema. Idempotent — safe to run on every startup.
247///
248/// `feeds`/`entries` are the shared cache; `entry_state`/`read_cursor` are
249/// per-DID. Indices cover the scheduler's due-feed query, the read/unread list
250/// query, and the flusher's dirty-cursor scan.
251const SCHEMA: &str = r#"
252PRAGMA foreign_keys = ON;
253
254CREATE TABLE IF NOT EXISTS feeds (
255 id INTEGER PRIMARY KEY AUTOINCREMENT,
256 url TEXT NOT NULL UNIQUE,
257 title TEXT,
258 site_url TEXT,
259 etag TEXT,
260 last_modified TEXT,
261 last_polled TEXT,
262 next_poll TEXT,
263 consecutive_errors INTEGER NOT NULL DEFAULT 0,
264 last_error_kind TEXT,
265 last_error TEXT,
266 -- What the poller does with this row; see `feed::FeedKind`. Written by the
267 -- Rust side at insert so SQL never re-derives it from the URL.
268 kind TEXT NOT NULL DEFAULT 'rss'
269);
270CREATE INDEX IF NOT EXISTS idx_feeds_next_poll ON feeds (next_poll);
271-- NOTE: `idx_feeds_kind` is created in `apply_migrations`, AFTER `kind` is
272-- ensured, for the same reason as the `intended_did` indexes below. 0.3.9 put
273-- it here and crash-looped production on its first boot: on an existing volume
274-- the CREATE TABLE above is a no-op, so the column does not exist yet.
275
276CREATE TABLE IF NOT EXISTS entries (
277 id INTEGER PRIMARY KEY AUTOINCREMENT,
278 feed_id INTEGER NOT NULL REFERENCES feeds (id) ON DELETE CASCADE,
279 guid TEXT NOT NULL,
280 url TEXT,
281 title TEXT,
282 author TEXT,
283 published TEXT,
284 content_html TEXT,
285 fetched_at TEXT NOT NULL,
286 UNIQUE (feed_id, guid)
287);
288-- The list and prev/next queries order on `COALESCE(published, fetched_at)`
289-- (#187). Measured on the real query shape (LEFT JOIN entry_state, EXISTS
290-- sub_ref), this index serves them as well as it served bare `published`; a
291-- `(feed_id, published, fetched_at)` replacement was tried and was ~3.8x
292-- slower on the default prev/next query, which never chose it (review of #213).
293CREATE INDEX IF NOT EXISTS idx_entries_feed_published ON entries (feed_id, published);
294
295CREATE TABLE IF NOT EXISTS entry_state (
296 did TEXT NOT NULL,
297 entry_id INTEGER NOT NULL REFERENCES entries (id) ON DELETE CASCADE,
298 read INTEGER NOT NULL DEFAULT 0,
299 starred INTEGER NOT NULL DEFAULT 0,
300 updated_at TEXT NOT NULL,
301 PRIMARY KEY (did, entry_id)
302);
303CREATE INDEX IF NOT EXISTS idx_entry_state_did_read ON entry_state (did, read);
304-- The FK child key. `entry_id` is the TRAILING column of the primary key, so
305-- without this index it is not the leading column of anything and SQLite must
306-- FULL SCAN entry_state for EVERY row deleted from `entries` to service
307-- ON DELETE CASCADE.
308--
309-- That is not theoretical. Measured on 600k entry_state rows: 500 deletes took
310-- 10.3s and 2,000 took 38.3s, against a busy_timeout of 5s — so any retention
311-- sweep removing more than roughly 260 entries made every concurrent writer
312-- (star, mark-read, OAuth session write) fail with SQLITE_BUSY. With this index
313-- the same 32,850-row delete goes from ~10 minutes to 0.7s.
314--
315-- It also fixes the per-feed trim, whose starred-sparing subquery scans
316-- entry_state on every poll of every feed and scales with TOTAL rows across all
317-- users rather than with the feed being trimmed (2ms -> 21ms at 1M rows).
318CREATE INDEX IF NOT EXISTS idx_entry_state_entry_id ON entry_state (entry_id);
319
320-- Per-DID subscription projection. The shared `feeds`/`entries` cache is
321-- deduped by URL and NOT owned by any single DID; `sub_ref` records which
322-- feeds a given DID actually subscribes to (mirrored from the caller's PDS
323-- subscription set on every resolve/sync). Every entry/feed READ and every
324-- read/star MUTATION is scoped through this table so one user can never read
325-- or mutate another user's cached articles. Rows are refreshed by
326-- `replace_sub_refs`.
327CREATE TABLE IF NOT EXISTS sub_ref (
328 did TEXT NOT NULL,
329 feed_id INTEGER NOT NULL REFERENCES feeds (id) ON DELETE CASCADE,
330 PRIMARY KEY (did, feed_id)
331);
332CREATE INDEX IF NOT EXISTS idx_sub_ref_feed ON sub_ref (feed_id);
333
334CREATE TABLE IF NOT EXISTS read_cursor (
335 did TEXT NOT NULL,
336 feed_url TEXT NOT NULL,
337 read_through TEXT,
338 read_ids TEXT NOT NULL DEFAULT '[]',
339 unread_ids TEXT NOT NULL DEFAULT '[]',
340 dirty INTEGER NOT NULL DEFAULT 0,
341 pds_created INTEGER NOT NULL DEFAULT 0,
342 updated_at TEXT NOT NULL,
343 PRIMARY KEY (did, feed_url)
344);
345CREATE INDEX IF NOT EXISTS idx_read_cursor_dirty ON read_cursor (did, dirty);
346-- The (did, feed_url) PRIMARY KEY can't serve a feed_url-only lookup (did is the
347-- leading column). The retention path's orphan-cursor cleanup filters cursors by
348-- feed_url alone, so give it an index.
349CREATE INDEX IF NOT EXISTS idx_read_cursor_feed_url ON read_cursor (feed_url);
350
351CREATE TABLE IF NOT EXISTS beta_access (
352 did TEXT PRIMARY KEY,
353 handle TEXT,
354 granted_by TEXT NOT NULL,
355 granted_at INTEGER NOT NULL,
356 invite_code_used TEXT
357);
358
359CREATE TABLE IF NOT EXISTS invite_codes (
360 code TEXT PRIMARY KEY,
361 creator_did TEXT NOT NULL,
362 status TEXT NOT NULL,
363 invitee_did TEXT,
364 -- The follower DID a bot-minted claim was minted FOR (recorded at mint time,
365 -- distinct from `invitee_did` which is stamped at redeem). This is the
366 -- server-side idempotency key: a second `POST /bot/claims` for a DID that
367 -- already holds an outstanding active code returns the SAME code instead of
368 -- minting a duplicate, so a bot-host state loss cannot re-mint per follower.
369 intended_did TEXT,
370 created_at INTEGER NOT NULL,
371 expires_at INTEGER NOT NULL,
372 redeemed_at INTEGER
373);
374CREATE INDEX IF NOT EXISTS idx_invite_codes_status ON invite_codes (status, expires_at);
375-- NOTE: the `intended_did` indexes are created in `apply_migrations`, AFTER the
376-- `intended_did` column is ensured. They MUST NOT live in this base SCHEMA batch:
377-- on an existing pre-0.2.2 volume the `CREATE TABLE IF NOT EXISTS invite_codes`
378-- above is a no-op (the table already exists without `intended_did`), so a
379-- `CREATE INDEX ... (intended_did, ...)` here would fail with "no such column"
380-- and crash-loop the boot before migrations ever run.
381
382-- Network-observation counters (v0.2.8, design/NETWORK-SPEC.md §4.3). One row
383-- per (metric, relay): the adoption probe records how many repos a given relay
384-- has INDEXED as holding a collection. We store the COUNT, never the DID list —
385-- persisting the DIDs would build a durable register of "accounts that use an
386-- RSS reader" on our disk for a feature whose only output is an integer. This
387-- table is a PROJECTION, not a source of truth: `DROP TABLE` it and the next
388-- probe rebuilds it, and nothing in the reader path reads it. Bounded forever at
389-- (metrics × relays) rows, so it never interacts with the DB-size watermark.
390CREATE TABLE IF NOT EXISTS network_stat (
391 key TEXT NOT NULL, -- e.g. 'adoption.subscription'
392 source TEXT NOT NULL, -- the relay host the number came from
393 value INTEGER NOT NULL,
394 truncated INTEGER NOT NULL DEFAULT 0,
395 observed_at TEXT NOT NULL,
396 PRIMARY KEY (key, source)
397);
398-- Repo-operation timings, for comparing the two backends across a CUTOVER.
399--
400-- Persisted rather than held in memory because flipping the backend requires a
401-- restart, and an in-memory table would lose the outgoing backend's numbers at
402-- exactly the moment they became worth comparing against. These rows are the
403-- only reason a "side by side" table can show two backends at once.
404--
405-- `repo_timing` is a bounded window of recent samples (pruned per backend+op);
406-- `repo_timing_total` carries the all-time counts, which must survive that
407-- pruning or a long-running backend would appear to have served fewer calls
408-- than a fresh one.
409CREATE TABLE IF NOT EXISTS repo_timing (
410 id INTEGER PRIMARY KEY AUTOINCREMENT,
411 backend TEXT NOT NULL,
412 op TEXT NOT NULL,
413 micros INTEGER NOT NULL,
414 ok INTEGER NOT NULL,
415 at INTEGER NOT NULL
416);
417
418CREATE INDEX IF NOT EXISTS idx_repo_timing_key ON repo_timing(backend, op, id);
419
420CREATE TABLE IF NOT EXISTS repo_timing_total (
421 backend TEXT NOT NULL,
422 op TEXT NOT NULL,
423 ok_count INTEGER NOT NULL DEFAULT 0,
424 err_count INTEGER NOT NULL DEFAULT 0,
425 PRIMARY KEY (backend, op)
426);
427
428"#;
429
430/// RFC3339 timestamp for "now" (UTC, seconds precision), used as the default for
431/// `*_at` columns. Uses `chrono` to match the shape written by [`crate::feed`]
432/// and [`crate::web`] (one timestamp format across the whole crate).
433fn now_rfc3339() -> String {
434 chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, true)
435}
436
437/// Open the per-DID SQLite cache described by [`Config`] (its `db_path`), run
438/// schema creation, and return the pool.
439///
440/// This is the entrypoint `main` calls: it derives the sqlx SQLite URL from the
441/// configured filesystem path and delegates to [`init_url`]. Kept separate from
442/// [`init_url`] so tests can open an in-memory database directly.
443pub async fn init(config: &Config) -> Result<Pool> {
444 // sqlx wants a `sqlite://<path>` URL; build it from the configured path.
445 let db_url = format!("sqlite://{}", config.db_path.display());
446 init_url(&db_url).await
447}
448
449/// Open (creating if needed) the SQLite database at `db_url`, run schema
450/// creation, and return a connection pool.
451///
452/// `db_url` is a sqlx SQLite URL, e.g. `sqlite://featherreader.db` or
453/// `sqlite::memory:` for an ephemeral in-memory database. The file is created
454/// if it does not exist; WAL journaling is enabled for on-disk databases and
455/// foreign keys are enforced on every connection.
456/// Ceiling the WAL is truncated back to at each checkpoint.
457///
458/// The WAL lives on the same volume as the database and counts against the same
459/// 1 GB, but nothing bounded it: SQLite grows the WAL to fit the largest
460/// transaction it has ever seen and never shrinks it again without this limit.
461const WAL_SIZE_LIMIT_BYTES: i64 = 64 * 1024 * 1024;
462
463pub async fn init_url(db_url: &str) -> Result<Pool> {
464 // An in-memory DB must run on a SINGLE connection: each `:memory:` connection
465 // is a *separate* database, and a multi-connection in-memory pool can also
466 // deadlock a writer against an idle pooled connection's shared-cache table
467 // read-lock (SQLITE_LOCKED, code 262 — which `busy_timeout` does NOT retry;
468 // seen as a Linux-only flaky failure in redeem_code's UPDATE). On-disk uses
469 // WAL + a 5-connection pool as normal.
470 let is_memory = db_url.contains(":memory:");
471 let mut opts = SqliteConnectOptions::from_str(db_url)
472 .with_context(|| format!("invalid sqlite url: {db_url}"))?
473 .create_if_missing(true)
474 .foreign_keys(true);
475 // WAL is a no-op / unsupported for :memory:, so only request it on-disk.
476 if !is_memory {
477 opts = opts.journal_mode(sqlx::sqlite::SqliteJournalMode::Wal);
478 // **Incremental auto-vacuum, set at CREATION.**
479 //
480 // `auto_vacuum` was read by `reclaim` and never set anywhere, so every
481 // database ran in SQLite's default NONE mode and `reclaim` always took
482 // its full-`VACUUM` branch — daily, and again after every prune. A full
483 // VACUUM needs free disk roughly equal to the live database because it
484 // writes a whole new file, which is exactly what is scarce under the
485 // disk pressure that triggers a sweep; on a ~700 MiB database on a 1 GB
486 // volume it cannot complete at all.
487 //
488 // This pragma only takes effect on a database with no tables yet, so it
489 // fixes NEW instances permanently and does nothing to existing ones —
490 // deliberately. Changing it on a populated database requires running the
491 // very full VACUUM that is unsafe here, so that is a separate,
492 // operator-invoked step: see [`migrate_to_incremental_vacuum`].
493 opts = opts.auto_vacuum(sqlx::sqlite::SqliteAutoVacuum::Incremental);
494 // Truncate the WAL back down at checkpoints. Without a limit, a WAL
495 // grown once by a single large transaction stays that size for the life
496 // of the file — permanently occupying volume the watermark is trying to
497 // protect. The batched retention deletes keep transactions small now, so
498 // in practice the WAL should rarely approach this; the limit is what
499 // makes that a guarantee rather than a hope.
500 opts = opts.pragma("journal_size_limit", WAL_SIZE_LIMIT_BYTES.to_string());
501 }
502 // Under a concurrent write burst (the poller's insert_entries tx racing the
503 // web layer's mark_read / redeem_code tx) SQLite would otherwise return
504 // SQLITE_BUSY the instant a writer holds the lock. `busy_timeout` makes a
505 // blocked connection WAIT (retry) for up to this long before erroring, so
506 // short lock contention resolves transparently instead of surfacing a
507 // spurious failure. Mirrors the OAuth sidecar's `stores.ts`
508 // (`PRAGMA busy_timeout = 5000`). 5 s is comfortably above any single
509 // FeatherReader transaction.
510 opts = opts.busy_timeout(std::time::Duration::from_millis(5000));
511 // Quiet sqlx's per-statement query logging.
512 opts = opts.log_statements(tracing::log::LevelFilter::Debug);
513
514 let pool = SqlitePoolOptions::new()
515 // Keep at least one connection alive so an in-memory DB isn't dropped
516 // (each `:memory:` connection is a *separate* database otherwise).
517 .min_connections(1)
518 .max_connections(if is_memory { 1 } else { 5 })
519 .connect_with(opts)
520 .await
521 .with_context(|| format!("failed to open sqlite pool: {db_url}"))?;
522
523 init_schema(&pool).await?;
524 Ok(pool)
525}
526
527/// Run the idempotent schema creation. Split out so callers/tests can (re)apply
528/// it against an already-open pool.
529pub async fn init_schema(pool: &SqlitePool) -> Result<()> {
530 // `execute` runs the multi-statement batch (sqlite allows this).
531 sqlx::query(SCHEMA)
532 .execute(pool)
533 .await
534 .context("failed to create schema")?;
535 apply_migrations(pool).await?;
536 // The Rust OAuth client's tables live in the same database. Created
537 // UNCONDITIONALLY, not only when that backend is selected: the tables are
538 // empty and harmless under the sidecar, whereas creating them lazily would
539 // make the first request after a cutover flip fail with "no such table" --
540 // at the one moment nobody wants to discover a migration was missed.
541 crate::oauth::store::init_schema(pool)
542 .await
543 .context("failed to create the OAuth schema")?;
544 Ok(())
545}
546
547/// Apply additive, idempotent migrations to bring an EXISTING database up to the
548/// current [`SCHEMA`]. `CREATE TABLE IF NOT EXISTS` never alters a table that
549/// already exists, so a column added to a shipped table must be back-filled here
550/// (SQLite has no `ADD COLUMN IF NOT EXISTS`, so we probe `table_info` first).
551async fn apply_migrations(pool: &SqlitePool) -> Result<()> {
552 // feeds.consecutive_errors — drives the exponential poll backoff. Older DBs
553 // predate the column; add it (defaulting to 0) if it is missing.
554 ensure_column(
555 pool,
556 "PRAGMA table_info(feeds)",
557 "consecutive_errors",
558 "ALTER TABLE feeds ADD COLUMN consecutive_errors INTEGER NOT NULL DEFAULT 0",
559 )
560 .await?;
561 // feeds.last_error_kind / feeds.last_error — WHY a feed is failing, not just
562 // how often. `consecutive_errors` recorded a count and nothing else, which is
563 // how a systematic defect across sixty feeds stayed indistinguishable from
564 // sixty dead blogs until #159: every one of them was our own 304 handling,
565 // and the table could not say so. Nullable, and NULL once a poll succeeds.
566 ensure_column(
567 pool,
568 "PRAGMA table_info(feeds)",
569 "last_error_kind",
570 "ALTER TABLE feeds ADD COLUMN last_error_kind TEXT",
571 )
572 .await?;
573 ensure_column(
574 pool,
575 "PRAGMA table_info(feeds)",
576 "last_error",
577 "ALTER TABLE feeds ADD COLUMN last_error TEXT",
578 )
579 .await?;
580
581 // feeds.kind — what the poller does with a row. Older DBs predate it and
582 // get `'rss'` from the DEFAULT, which is wrong for the at:// rows, so it is
583 // back-filled below.
584 ensure_column(
585 pool,
586 "PRAGMA table_info(feeds)",
587 "kind",
588 "ALTER TABLE feeds ADD COLUMN kind TEXT NOT NULL DEFAULT 'rss'",
589 )
590 .await?;
591 // Here, not in the base SCHEMA batch: it names a column that only exists
592 // after the line above. See the note beside `idx_feeds_next_poll`.
593 sqlx::query("CREATE INDEX IF NOT EXISTS idx_feeds_kind ON feeds (kind)")
594 .execute(pool)
595 .await
596 .context("creating idx_feeds_kind")?;
597
598 // **Re-derived in Rust, every row, every start — not translated once.**
599 //
600 // `kind` is a pure function of `url`, so it is a cache, and a cache that is
601 // only ever written forward goes stale the moment the function changes.
602 // The first version of this was a one-directional SQL `UPDATE` carrying its
603 // own copy of the rule as a string predicate: it agreed with
604 // `FeedKind::of` on the day it was written, translated `rss` to
605 // `publication` and never the reverse, and had no way to notice either
606 // fact. Asking the Rust classifier about every row instead means the column
607 // cannot disagree with the one function that defines it, and a future kind
608 // — or a corrected rule — needs no migration of its own.
609 //
610 // Cheap by shape, not by assumption: it writes only rows that are actually
611 // wrong, so the steady state is a single scan of a table that holds one row
612 // per subscribed feed.
613 let rows = sqlx::query("SELECT id, url, kind FROM feeds")
614 .fetch_all(pool)
615 .await
616 .context("reading feeds to re-derive kind")?;
617 let mut tx = pool.begin().await.context("begin kind re-derivation")?;
618 let (mut to_pollable, mut to_unpollable, mut unreadable) = (0u64, 0u64, 0u64);
619 for row in rows {
620 // **A row we cannot read is skipped, not fatal.** This runs on the boot
621 // path, so anything that returns `Err` here is the difference between a
622 // wedged poller and a site that will not start. A `url` or `kind` that
623 // is not decodable as text takes no opinion from us and keeps whatever
624 // it has; every reader downstream already treats an unknown kind as
625 // unpollable. Nothing sqlx writes produces such a row — it binds `&str`
626 // as TEXT everywhere — so reaching this means the file was edited by
627 // hand, which is exactly when refusing to boot is the least helpful
628 // thing to do.
629 let (Ok(id), Ok(url), Ok(kind)) = (
630 row.try_get::<i64, _>("id"),
631 row.try_get::<String, _>("url"),
632 row.try_get::<String, _>("kind"),
633 ) else {
634 unreadable += 1;
635 continue;
636 };
637 let want = crate::feed::FeedKind::of(&url);
638 if kind == want.as_str() {
639 continue;
640 }
641 sqlx::query("UPDATE feeds SET kind = ?1 WHERE id = ?2")
642 .bind(want.as_str())
643 .bind(id)
644 .execute(&mut *tx)
645 .await
646 .with_context(|| format!("re-deriving kind for feed {id}"))?;
647 if crate::feed::FeedKind::POLLABLE.contains(&want) {
648 to_pollable += 1;
649 } else {
650 // **Declaring a row unpollable orphans its poll state, so clear
651 // it.** A backoff horizon and an error count belong to a feed the
652 // scheduler selects; on a row it will never select again they are
653 // dead, and not inert. They are hidden from `/stats` and the cause
654 // histogram, which filter on kind, so they rot unseen — and if a
655 // later rule change makes the row pollable again it resumes at
656 // `backoff_for(n)` on an `n` earned under a classification that no
657 // longer applies, which for seven prior errors is a first retry ten
658 // hours out instead of five minutes.
659 //
660 // Narrower than the step below, deliberately: that one clears only
661 // rows we never polled, on the grounds that a real feed's history
662 // still means something. This clears rows whose history can no
663 // longer mean anything, because nothing will add to it or act on
664 // it.
665 sqlx::query(
666 "UPDATE feeds SET consecutive_errors = 0, last_error_kind = NULL, \
667 last_error = NULL, next_poll = NULL WHERE id = ?1",
668 )
669 .bind(id)
670 .execute(&mut *tx)
671 .await
672 .with_context(|| format!("clearing orphaned poll state for feed {id}"))?;
673 to_unpollable += 1;
674 }
675 }
676 tx.commit().await.context("commit kind re-derivation")?;
677 // Quiet in the steady state, which is every boot where nothing changed.
678 // Split by direction because the two mean opposite things to an operator:
679 // one puts feeds back in the poller's queue, the other takes them out of
680 // every figure `/stats` reports.
681 if to_pollable > 0 || to_unpollable > 0 {
682 tracing::info!(
683 to_pollable,
684 to_unpollable,
685 "feeds.kind re-derived from the URL"
686 );
687 }
688 if unreadable > 0 {
689 tracing::warn!(
690 unreadable,
691 "feeds rows are not readable as text; their kind was left alone"
692 );
693 }
694
695 // **Clear failure counts on rows we never actually polled.**
696 //
697 // `due_feeds` excludes them by kind (see `feed::FeedKind`) — but rows
698 // subscribed before the scheme check already carry the errors OUR refusal
699 // produced. Left alone they would count as failing forever, since no poll
700 // that could clear them will ever be scheduled.
701 //
702 // A real feed's history is untouched: it still means something. The
703 // recorded reason goes with the count: a row with no errors must carry no
704 // reason, which is what `reset_feed_errors` promises and a test asserts.
705 //
706 // **Idempotent by predicate.** `last_polled` is set only by a successful
707 // poll — `bump_feed_errors` never touches it — so `last_polled IS NULL`
708 // selects exactly the rows whose every error came from our own refusal.
709 // A row a wired standard.site reader has fetched once keeps its later
710 // failures across restarts; a row that only ever failed under the refusal
711 // is cleared at every boot, including after a rollback to a build that
712 // polled it. A version stamp was the first design and left that rollback
713 // case a permanent hole (re-accumulated errors hidden by the filters,
714 // never cleared). Trade-off accepted: a publication that has never once
715 // succeeded restarts its backoff at the floor on every boot.
716 sqlx::query(sqlx::AssertSqlSafe(format!(
717 "UPDATE feeds SET consecutive_errors = 0, last_error_kind = NULL, last_error = NULL \
718 WHERE kind NOT IN ({POLLABLE_KINDS_SQL}) AND last_polled IS NULL \
719 AND consecutive_errors > 0"
720 )))
721 .execute(pool)
722 .await
723 .context("clearing error counts on unpollable at:// feeds")?;
724 // read_cursor.pds_created — tracks whether a feed's readState record has been
725 // created in the PDS, so the first flush emits a `create` (not a bare
726 // `update`, which errors on a not-yet-existing record). Older DBs predate it.
727 ensure_column(
728 pool,
729 "PRAGMA table_info(read_cursor)",
730 "pds_created",
731 "ALTER TABLE read_cursor ADD COLUMN pds_created INTEGER NOT NULL DEFAULT 0",
732 )
733 .await?;
734 // invite_codes.intended_did — the follower DID a bot claim was minted for, the
735 // server-side idempotency key for `POST /bot/claims`. Older DBs (before the
736 // follow→invite bot) predate it; it is nullable (browser/admin-minted codes
737 // leave it NULL).
738 ensure_column(
739 pool,
740 "PRAGMA table_info(invite_codes)",
741 "intended_did",
742 "ALTER TABLE invite_codes ADD COLUMN intended_did TEXT",
743 )
744 .await?;
745 // Indexes on `intended_did` are created HERE (not in the base SCHEMA batch)
746 // because they reference a column that only exists after the migration above.
747 // On an existing pre-0.2.2 DB the `invite_codes` CREATE TABLE is a no-op, so
748 // an index on `intended_did` in SCHEMA would fail before this migration ran
749 // (that was blocker B1). All are `IF NOT EXISTS`, so re-running is a no-op.
750 //
751 // Look up an outstanding active claim by the DID it was minted for (bot dedupe).
752 sqlx::query(
753 "CREATE INDEX IF NOT EXISTS idx_invite_codes_intended \
754 ON invite_codes (intended_did, status)",
755 )
756 .execute(pool)
757 .await
758 .context("creating idx_invite_codes_intended")?;
759 // Enforce at MOST one outstanding active claim per intended DID. This makes
760 // the bot's dedupe check-then-mint race-safe: two concurrent `POST /bot/claims`
761 // for the same follower can no longer both insert an active code (the second
762 // INSERT hits this unique constraint). Partial so it only constrains active
763 // bot-minted rows — redeemed/expired rows and NULL-intended (admin/browser)
764 // codes are unconstrained. (Blocker/should-fix S4.)
765 sqlx::query(
766 "CREATE UNIQUE INDEX IF NOT EXISTS idx_invite_codes_intended_active \
767 ON invite_codes (intended_did) \
768 WHERE intended_did IS NOT NULL AND status = 'active'",
769 )
770 .execute(pool)
771 .await
772 .context("creating idx_invite_codes_intended_active")?;
773
774 // **Re-date rows stored with a future date before ingest refused them**
775 // (#188). An item dated 2999 that has since left its feed is never polled
776 // again to be corrected, so it would stay first in the list and survive the
777 // per-feed cap. Cleared, `fetched_at` dates it. The same bound ingest uses;
778 // a no-op once there are none.
779 let ceiling = (chrono::Utc::now()
780 + chrono::Duration::days(crate::feed::MAX_FUTURE_PUBLISHED_DAYS))
781 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
782 sqlx::query("UPDATE entries SET published = NULL WHERE published > ?1")
783 .bind(&ceiling)
784 .execute(pool)
785 .await
786 .context("clearing stored future publication dates")?;
787 Ok(())
788}
789
790/// Add a column via `alter_sql` iff `info_sql` (a `PRAGMA table_info(<table>)`)
791/// does not already report `column`. All three SQL args are hard-coded internal
792/// literals (never user input), so they are safe `&'static str`s — the table name
793/// can't be a bind parameter in `PRAGMA`, which is why they're passed whole.
794async fn ensure_column(
795 pool: &SqlitePool,
796 info_sql: &'static str,
797 column: &str,
798 alter_sql: &'static str,
799) -> Result<()> {
800 let rows = sqlx::query(info_sql)
801 .fetch_all(pool)
802 .await
803 .with_context(|| format!("{info_sql} failed"))?;
804 let present = rows.iter().any(|r| r.get::<String, _>("name") == column);
805 if !present {
806 sqlx::query(alter_sql)
807 .execute(pool)
808 .await
809 .with_context(|| format!("adding column {column} via {alter_sql}"))?;
810 }
811 Ok(())
812}
813
814/// Insert a feed by URL, or update its metadata if the URL already exists.
815/// Returns the feed's row id (existing or newly assigned).
816///
817/// EVERY updatable column is COALESCE'd, so `None` means "leave alone" for all
818/// of them and a partial upsert cannot clobber a field it never mentioned.
819///
820/// `etag`/`last_modified` were the exception until now, and the exception was
821/// silently disabling conditional GET for the entire instance. `set_next_poll`
822/// in the scheduler supplies only `url` + `next_poll` after every single poll,
823/// which wrote both validators back to NULL — so `304 Not Modified` was
824/// unreachable and every feed was re-downloaded, re-parsed, re-sanitised and
825/// re-inserted in full, hourly, forever. `feed::touch_polled` had discovered the
826/// same trap earlier and worked around it in its own caller by re-reading the
827/// row first; that local fix is what let the next caller walk into it.
828///
829/// A stale validator is not a hazard: if the origin no longer issues one it
830/// ignores our `If-None-Match` and returns `200`, and if it still matches then
831/// `304` was the correct answer anyway.
832pub async fn upsert_feed(pool: &SqlitePool, feed: &NewFeed) -> Result<i64> {
833 let row = sqlx::query(
834 r#"
835 INSERT INTO feeds (url, title, site_url, etag, last_modified, last_polled, next_poll, kind)
836 VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8)
837 ON CONFLICT (url) DO UPDATE SET
838 title = COALESCE(excluded.title, feeds.title),
839 site_url = COALESCE(excluded.site_url, feeds.site_url),
840 etag = COALESCE(excluded.etag, feeds.etag),
841 last_modified = COALESCE(excluded.last_modified, feeds.last_modified),
842 last_polled = COALESCE(excluded.last_polled, feeds.last_polled),
843 next_poll = COALESCE(excluded.next_poll, feeds.next_poll),
844 -- Not COALESCE: `kind` is derived from the URL, and `excluded`
845 -- always carries the current answer. Preserving the stored value
846 -- would make a row's classification a function of when it was
847 -- first subscribed rather than of what it is.
848 kind = excluded.kind
849 RETURNING id
850 "#,
851 )
852 .bind(&feed.url)
853 .bind(&feed.title)
854 .bind(&feed.site_url)
855 .bind(&feed.etag)
856 .bind(&feed.last_modified)
857 .bind(&feed.last_polled)
858 .bind(&feed.next_poll)
859 // Decided once, in Rust, and never re-derived from the URL by SQL.
860 .bind(crate::feed::FeedKind::of(&feed.url).as_str())
861 .fetch_one(pool)
862 .await
863 .with_context(|| format!("upsert_feed failed for {}", feed.url))?;
864
865 Ok(row.get::<i64, _>("id"))
866}
867
868/// Fetch a feed by its URL, if present.
869pub async fn get_feed_by_url(pool: &SqlitePool, url: &str) -> Result<Option<Feed>> {
870 let feed = sqlx::query_as::<_, Feed>("SELECT * FROM feeds WHERE url = ?1")
871 .bind(url)
872 .fetch_optional(pool)
873 .await
874 .with_context(|| format!("get_feed_by_url failed for {url}"))?;
875 Ok(feed)
876}
877
878/// The `kind` values the scheduler may select, as a SQL list.
879///
880/// Pinned against [`crate::feed::FeedKind::POLLABLE`] by
881/// `the_sql_kind_list_matches_the_rust_one` — a literal here and a slice there
882/// is exactly the drift the column was introduced to end, so the two are
883/// asserted equal rather than trusted. Wiring the standard.site reader means
884/// changing both, and that test is what makes forgetting one a failure.
885pub(crate) const POLLABLE_KINDS_SQL: &str = "'rss', 'publication'";
886
887/// The `kind` values the retention **window** applies to, as a SQL list.
888///
889/// Pinned against [`crate::feed::FeedKind::AGED`] by
890/// `the_sql_aged_kind_list_matches_the_rust_one`, for the same reason
891/// [`POLLABLE_KINDS_SQL`] is pinned against `POLLABLE`.
892///
893/// Why a publication is not in it: see `FeedKind::AGED`. Measured — a 14-day
894/// window stored zero rows from every real publication tried, because their
895/// newest documents were 109 to 241 days old.
896pub(crate) const AGED_KINDS_SQL: &str = "'rss'";
897
898/// How many rows the poller will never select — the capacity consumed by feeds
899/// that cannot be fetched.
900///
901/// Rendered on `/admin/metrics` because the global ceiling counts these rows
902/// (see [`count_feeds`]) while `/stats` does not, so without this the cap could
903/// be reached with every public number saying otherwise.
904pub async fn unpollable_feeds(pool: &SqlitePool) -> Result<i64> {
905 sqlx::query_scalar(sqlx::AssertSqlSafe(format!(
906 "SELECT COUNT(*) FROM feeds WHERE kind NOT IN ({POLLABLE_KINDS_SQL})"
907 )))
908 .fetch_one(pool)
909 .await
910 .context("counting unpollable feeds")
911}
912
913/// How many rows the poller will never select. Test-only: the assertion the
914/// at:// tests make, spelled once, against the predicate the code uses.
915#[cfg(test)]
916pub(crate) async fn count_unpollable_feeds(pool: &SqlitePool) -> Result<i64> {
917 sqlx::query_scalar(sqlx::AssertSqlSafe(format!(
918 "SELECT COUNT(*) FROM feeds WHERE kind NOT IN ({POLLABLE_KINDS_SQL})"
919 )))
920 .fetch_one(pool)
921 .await
922 .context("counting unpollable feeds")
923}
924
925/// The scheduler's hot query: feeds whose `next_poll` is due (`<= as_of`, or
926/// never polled), oldest-due first. `as_of` is an RFC3339 timestamp.
927pub async fn due_feeds(pool: &SqlitePool, as_of: &str, limit: i64) -> Result<Vec<Feed>> {
928 let sql = format!(
929 r#"
930 SELECT * FROM feeds
931 WHERE (next_poll IS NULL OR next_poll <= ?1)
932 -- Only POLLABLE kinds are due; an `unsupported` row is skipped, not
933 -- failed. The why lives on `feed::FeedKind::POLLABLE`.
934 AND kind IN ({POLLABLE_KINDS_SQL})
935 ORDER BY next_poll IS NOT NULL, next_poll ASC
936 LIMIT ?2
937 "#
938 );
939 let feeds = sqlx::query_as::<_, Feed>(sqlx::AssertSqlSafe(sql))
940 .bind(as_of)
941 .bind(limit)
942 .fetch_all(pool)
943 .await
944 .context("due_feeds failed")?;
945 Ok(feeds)
946}
947
948/// One failing feed, named, for the ADMIN view only.
949///
950/// The public `/stats` histogram is counts by cause and nothing else, by that
951/// page's own stated promise. This is the other half: the coarse bucket
952/// `fetch` covers DNS failure, timeout, SSRF refusal and — as #159 proved —
953/// this reader's own bugs, so a count alone cannot separate "the publishers are
954/// gone" from "we are broken". The detail can, and it lives behind the
955/// `ALLOWED_DIDS` gate where per-feed data is already permitted.
956#[derive(Debug, Clone, PartialEq, Eq)]
957pub struct FailingFeed {
958 pub url: String,
959 pub consecutive_errors: i64,
960 /// `None` for a row that predates the column — see the `unknown` bucket.
961 pub kind: Option<String>,
962 pub detail: Option<String>,
963}
964
965/// Every currently-failing feed with its recorded cause, worst first.
966///
967/// **Admin-gated callers only.** Bounded because this renders into one response
968/// and a large instance should not be able to make that response unbounded.
969pub async fn failing_feeds(pool: &SqlitePool, limit: i64) -> Result<Vec<FailingFeed>> {
970 // The same exclusion as `poll_health`: a row the poller never selects
971 // can never have its errors cleared, so listing it here would pin it to
972 // the top of the operator's page for good.
973 let sql = format!(
974 r#"
975 SELECT url, consecutive_errors, last_error_kind, last_error
976 FROM feeds
977 WHERE consecutive_errors > 0 AND kind IN ({POLLABLE_KINDS_SQL})
978 ORDER BY consecutive_errors DESC, url ASC
979 LIMIT ?1
980 "#
981 );
982 let rows: Vec<(String, i64, Option<String>, Option<String>)> =
983 sqlx::query_as(sqlx::AssertSqlSafe(sql))
984 .bind(limit)
985 .fetch_all(pool)
986 .await
987 .context("listing failing feeds")?;
988 Ok(rows
989 .into_iter()
990 .map(|(url, consecutive_errors, kind, detail)| FailingFeed {
991 url,
992 consecutive_errors,
993 kind,
994 detail,
995 })
996 .collect())
997}
998
999/// Cap on the stored `last_error` detail. Remote text on an unattended path.
1000const MAX_ERROR_DETAIL_CHARS: usize = 300;
1001
1002/// Record a poll FAILURE for a feed: bump its `consecutive_errors` by one and
1003/// return the NEW count. The count drives the exponential poll backoff, so a
1004/// persistently-failing feed spaces its retries out toward the ceiling instead of
1005/// hammering the 5-minute floor forever. Reset to 0 by [`reset_feed_errors`] on
1006/// any success/304.
1007pub async fn bump_feed_errors(
1008 pool: &SqlitePool,
1009 url: &str,
1010 kind: crate::feed::FailureKind,
1011 detail: &str,
1012) -> Result<i64> {
1013 let row = sqlx::query(
1014 "UPDATE feeds SET consecutive_errors = consecutive_errors + 1, \
1015 last_error_kind = ?2, last_error = ?3 \
1016 WHERE url = ?1 RETURNING consecutive_errors",
1017 )
1018 .bind(url)
1019 .bind(kind.as_str())
1020 // **Truncated.** This is a remote server's error text on an unattended path;
1021 // an upstream that returns a megabyte of prose should cost a bounded row,
1022 // not an unbounded one.
1023 .bind(
1024 detail
1025 .chars()
1026 .take(MAX_ERROR_DETAIL_CHARS)
1027 .collect::<String>(),
1028 )
1029 .fetch_optional(pool)
1030 .await
1031 .with_context(|| format!("bump_feed_errors failed for {url}"))?;
1032 // If the feed row somehow vanished, treat it as the first error.
1033 Ok(row
1034 .map(|r| r.get::<i64, _>("consecutive_errors"))
1035 .unwrap_or(1))
1036}
1037
1038/// Schedule a feed's next poll `delay` from now.
1039///
1040/// Lived as a private fn in the scheduler until `web::add_subscription`
1041/// needed it too: a poll taken off the scheduler settled the error columns but
1042/// never rescheduled, so a re-subscribed working feed stayed parked on its stale
1043/// backoff horizon for up to 24h. One implementation, two callers.
1044///
1045/// `upsert_feed` COALESCEs unset fields, so supplying only url + next_poll bumps
1046/// the schedule without clobbering title/validators/last_polled.
1047pub async fn set_next_poll(pool: &SqlitePool, url: &str, delay: std::time::Duration) -> Result<()> {
1048 let next = chrono::Utc::now()
1049 + chrono::Duration::from_std(delay).unwrap_or_else(|_| chrono::Duration::hours(1));
1050 let next_poll = next.to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
1051 let nf = NewFeed {
1052 url: url.to_string(),
1053 next_poll: Some(next_poll),
1054 ..Default::default()
1055 };
1056 upsert_feed(pool, &nf).await.map(|_| ())
1057}
1058
1059/// [`due_feeds`] for one kind only. The RSS poller and the publication poller
1060/// each select their own, so neither can be held by the other's reads.
1061pub async fn due_feeds_of_kind(
1062 pool: &SqlitePool,
1063 as_of: &str,
1064 kind: crate::feed::FeedKind,
1065 limit: i64,
1066) -> Result<Vec<Feed>> {
1067 sqlx::query_as::<_, Feed>(
1068 "SELECT * FROM feeds WHERE (next_poll IS NULL OR next_poll <= ?1) AND kind = ?2 \
1069 ORDER BY next_poll IS NOT NULL, next_poll ASC LIMIT ?3",
1070 )
1071 .bind(as_of)
1072 .bind(kind.as_str())
1073 .bind(limit)
1074 .fetch_all(pool)
1075 .await
1076 .context("due_feeds_of_kind failed")
1077}
1078
1079/// Spread the first polls of never-polled `kind` rows across `spread`.
1080///
1081/// **Admitting a kind to the poller makes every row of it due at once.**
1082/// `due_feeds` sorts `next_poll IS NULL` ahead of every dated row, and rows that
1083/// were never pollable have no schedule, so the boot that admits them hands the
1084/// poller a block that outranks every regular feed — including an overdue one —
1085/// until it drains (`feed::FeedKind::POLLABLE` documents the measurement). This
1086/// gives each such row its own slot in `[now, now + spread)`, in id order, so
1087/// the block arrives as a trickle. Rows that have been polled, or already carry
1088/// a schedule, are untouched. Returns how many rows were scheduled.
1089pub async fn stagger_unscheduled(
1090 pool: &SqlitePool,
1091 kind: crate::feed::FeedKind,
1092 spread: std::time::Duration,
1093) -> Result<u64> {
1094 let ids: Vec<i64> = sqlx::query_scalar(
1095 "SELECT id FROM feeds WHERE kind = ?1 AND next_poll IS NULL AND last_polled IS NULL \
1096 ORDER BY id",
1097 )
1098 .bind(kind.as_str())
1099 .fetch_all(pool)
1100 .await
1101 .context("listing unscheduled feeds to stagger")?;
1102 if ids.is_empty() {
1103 return Ok(0);
1104 }
1105 let now = chrono::Utc::now();
1106 let spread = chrono::Duration::from_std(spread).unwrap_or_else(|_| chrono::Duration::hours(1));
1107 let n = ids.len() as i32;
1108 let mut tx = pool.begin().await.context("begin stagger")?;
1109 for (i, id) in ids.iter().enumerate() {
1110 // Evenly spaced, the first due now: slot i of n across the spread.
1111 let at = now + spread * i as i32 / n;
1112 let next_poll = at.to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
1113 sqlx::query("UPDATE feeds SET next_poll = ?1 WHERE id = ?2 AND next_poll IS NULL")
1114 .bind(next_poll)
1115 .bind(id)
1116 .execute(&mut *tx)
1117 .await
1118 .context("staggering a feed's first poll")?;
1119 }
1120 tx.commit().await.context("commit stagger")?;
1121 Ok(ids.len() as u64)
1122}
1123
1124/// Reset a feed's `consecutive_errors` to 0 after a successful poll (or a 304).
1125/// A no-op UPDATE if the row is missing.
1126pub async fn reset_feed_errors(pool: &SqlitePool, url: &str) -> Result<()> {
1127 // **Clears the reason too.** A stale `last_error` on a feed that is now
1128 // succeeding is worse than none: it is the aggregate below reporting a cause
1129 // that stopped applying, which is the failure this column exists to end.
1130 sqlx::query(
1131 "UPDATE feeds SET consecutive_errors = 0, last_error_kind = NULL, last_error = NULL \
1132 WHERE url = ?1",
1133 )
1134 .bind(url)
1135 .execute(pool)
1136 .await
1137 .with_context(|| format!("reset_feed_errors failed for {url}"))?;
1138 Ok(())
1139}
1140
1141/// The feeds a `did` currently subscribes to, per its `sub_ref` projection.
1142/// Used by the PDS-unreachable fallback in `resolve_subscriptions` to render
1143/// the sidebar from the caller's OWN last-known subscriptions (fail closed)
1144/// rather than every cached feed.
1145pub async fn feeds_for_did(pool: &SqlitePool, did: &str) -> Result<Vec<Feed>> {
1146 let feeds = sqlx::query_as::<_, Feed>(
1147 r#"
1148 SELECT f.* FROM feeds f
1149 JOIN sub_ref sr ON sr.feed_id = f.id AND sr.did = ?1
1150 ORDER BY f.title IS NULL, f.title, f.url
1151 "#,
1152 )
1153 .bind(did)
1154 .fetch_all(pool)
1155 .await
1156 .with_context(|| format!("feeds_for_did failed for {did}"))?;
1157 Ok(feeds)
1158}
1159
1160/// The feed ids a `did` currently subscribes to (its `sub_ref` rows).
1161///
1162/// **Not bounded by `max_subs_per_did`.** This comment used to claim it was, and
1163/// callers leaned on that: the cap is enforced on the ADD and OPML paths only,
1164/// never on read, and `sub_ref` is rebuilt from whatever the PDS returns — which
1165/// any client can write to, bounded only by the list-pages ceiling at 20,000
1166/// records. A claim in a comment is not a bound.
1167///
1168/// Callers must therefore not assume a small result. The one that cared — the
1169/// list views' scope filter — no longer does: it passes the whole set as a
1170/// single `json_each` bind rather than one SQL placeholder per feed.
1171pub async fn subscribed_feed_ids(pool: &SqlitePool, did: &str) -> Result<Vec<i64>> {
1172 let ids: Vec<i64> = sqlx::query_scalar("SELECT feed_id FROM sub_ref WHERE did = ?1")
1173 .bind(did)
1174 .fetch_all(pool)
1175 .await
1176 .with_context(|| format!("subscribed_feed_ids failed for {did}"))?;
1177 Ok(ids)
1178}
1179
1180/// The number of feeds a `did` currently subscribes to (its `sub_ref` rows).
1181/// Backs the per-DID subscription cap enforced at the add/import paths.
1182pub async fn count_subscriptions_for_did(pool: &SqlitePool, did: &str) -> Result<i64> {
1183 let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM sub_ref WHERE did = ?1")
1184 .bind(did)
1185 .fetch_one(pool)
1186 .await
1187 .with_context(|| format!("count_subscriptions_for_did failed for {did}"))?;
1188 Ok(n)
1189}
1190
1191/// The number of distinct feeds in the shared cache. Backs the global feeds
1192/// ceiling checked before a brand-new feed is inserted.
1193pub async fn count_feeds(pool: &SqlitePool) -> Result<i64> {
1194 let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM feeds")
1195 .fetch_one(pool)
1196 .await
1197 .context("count_feeds failed")?;
1198 Ok(n)
1199}
1200
1201/// The **used** size of the SQLite database, in bytes, computed as
1202/// `(page_count - freelist_count) * page_size`. Backs the DB-size watermark that
1203/// disables new polling.
1204///
1205/// Subtracting the freelist is what keeps the watermark from latching the poller
1206/// off: `page_count` counts pages the file has *allocated*, including ones freed
1207/// by a `DELETE` but not yet returned to the OS (SQLite keeps them on a freelist
1208/// for reuse and never shrinks the file without a VACUUM). Counting only the
1209/// live pages means a retention prune (which frees pages, see [`reclaim`]) is
1210/// actually reflected here, so the watermark can drop back below its threshold
1211/// and polling resumes. Cheap (three `PRAGMA` reads); works for file + `:memory:`.
1212///
1213/// **The WAL counts too.** This is the number the DB-size watermark compares
1214/// against a VOLUME size, and in WAL mode the `-wal` sidecar sits on that same
1215/// volume — so leaving it out understated exactly the quantity the watermark
1216/// exists to bound. It is added back below, best-effort: a WAL that cannot be
1217/// stat'd contributes zero rather than failing the check, since a watermark that
1218/// errors is worse than one that is slightly optimistic.
1219pub async fn db_size_bytes(pool: &SqlitePool) -> Result<i64> {
1220 let page_count: i64 = sqlx::query_scalar("PRAGMA page_count")
1221 .fetch_one(pool)
1222 .await
1223 .context("PRAGMA page_count failed")?;
1224 let freelist_count: i64 = sqlx::query_scalar("PRAGMA freelist_count")
1225 .fetch_one(pool)
1226 .await
1227 .context("PRAGMA freelist_count failed")?;
1228 let page_size: i64 = sqlx::query_scalar("PRAGMA page_size")
1229 .fetch_one(pool)
1230 .await
1231 .context("PRAGMA page_size failed")?;
1232 let used_pages = page_count.saturating_sub(freelist_count).max(0);
1233 Ok(used_pages
1234 .saturating_mul(page_size)
1235 .saturating_add(wal_bytes(pool).await))
1236}
1237
1238/// Bytes the write-ahead log currently occupies on the database's volume, or 0
1239/// when there is no WAL (`:memory:`, non-WAL journal modes) or it cannot be
1240/// stat'd. Best-effort by design — see [`db_size_bytes`].
1241async fn wal_bytes(pool: &SqlitePool) -> i64 {
1242 let Some(path) = main_db_path(pool).await else {
1243 return 0;
1244 };
1245 std::fs::metadata(format!("{path}-wal"))
1246 .map(|m| i64::try_from(m.len()).unwrap_or(i64::MAX))
1247 .unwrap_or(0)
1248}
1249
1250/// The main database's file path, or `None` for `:memory:`.
1251async fn main_db_path(pool: &SqlitePool) -> Option<String> {
1252 sqlx::query_scalar("SELECT file FROM pragma_database_list WHERE name = 'main' AND file <> ''")
1253 .fetch_optional(pool)
1254 .await
1255 .ok()
1256 .flatten()
1257}
1258
1259/// Freelist pages returned to the OS per `incremental_vacuum` step. At a 4 KiB
1260/// page that is ~8 MiB per batch — a short lock hold, and few enough steps that
1261/// a large reclaim is tens of statements rather than thousands.
1262const RECLAIM_BATCH_PAGES: i64 = 2_000;
1263
1264/// Backstop on the reclaim loop. `freelist_count == 0` and the no-progress check
1265/// are the real terminators; at [`RECLAIM_BATCH_PAGES`] this is 2M pages (~8 GiB),
1266/// far past anything a 1 GB volume holds.
1267const RECLAIM_MAX_BATCHES: usize = 1_000;
1268
1269/// Reclaim freed pages so the database file (and its used-page accounting) can
1270/// actually shrink after a retention/prune sweep DELETEs rows.
1271///
1272/// Without this, a `DELETE` moves pages onto the freelist but never shrinks the
1273/// file — so once the DB-size watermark trips and retention deletes rows,
1274/// `page_count` stays put and [`db_size_bytes`] (well, its raw `page_count`
1275/// form) would never fall back below the watermark, latching the poller off
1276/// forever. Call this AFTER a prune. It uses incremental vacuum when the database
1277/// is in `auto_vacuum = INCREMENTAL` mode (cheap, no full rewrite), and otherwise
1278/// falls back to a full `VACUUM`.
1279pub async fn reclaim(pool: &SqlitePool) -> Result<()> {
1280 match auto_vacuum_mode(pool).await? {
1281 AutoVacuum::Incremental => {
1282 // **Bounded, like the deletes that precede it.**
1283 //
1284 // With no page argument this reclaims the ENTIRE freelist in one
1285 // transaction — handing straight back the write-lock hold that
1286 // batching the retention deletes had just won, immediately after the
1287 // sweep that created the freelist in the first place. Same shape as
1288 // `delete_in_batches`: a bounded unit of work, then an explicit
1289 // hand-off so a waiting writer actually gets in.
1290 // **Both early exits are LOUD.** Failing to reclaim is the failure
1291 // this function exists to prevent: `db_size_bytes` stays high,
1292 // `poll_due_once` keeps polling paused, and `/stats` says "paused"
1293 // with nothing anywhere saying reclaim gave up. Exiting silently
1294 // makes that indistinguishable from a sweep that had nothing to do.
1295 let mut drained = true;
1296 for batch in 0..RECLAIM_MAX_BATCHES {
1297 let before: i64 = sqlx::query_scalar("PRAGMA freelist_count")
1298 .fetch_one(pool)
1299 .await
1300 .context("PRAGMA freelist_count failed")?;
1301 if before == 0 {
1302 break;
1303 }
1304 // A PRAGMA argument cannot be a bind parameter, and this one is
1305 // a `const i64` declared in this file — nothing external reaches
1306 // it.
1307 sqlx::query(sqlx::AssertSqlSafe(format!(
1308 "PRAGMA incremental_vacuum({RECLAIM_BATCH_PAGES})"
1309 )))
1310 .execute(pool)
1311 .await
1312 .context("PRAGMA incremental_vacuum failed")?;
1313 let after: i64 = sqlx::query_scalar("PRAGMA freelist_count")
1314 .fetch_one(pool)
1315 .await
1316 .context("PRAGMA freelist_count failed")?;
1317 // No progress: either nothing more can be freed, or a
1318 // concurrent retention delete pushed `after` back up. Both leave
1319 // pages allocated, which is what an operator needs to know.
1320 //
1321 // This comment previously also claimed "a long-lived WAL read
1322 // snapshot pins freelist pages". MEASURED AND FALSE: with a
1323 // reader holding a snapshot taken BEFORE the delete, the
1324 // freelist still drained 2000 → 0 and `page_count` halved. A
1325 // reader blocks the CHECKPOINT, not the incremental vacuum — so
1326 // that case exits this loop through the SUCCESS path and is
1327 // reported below, not here.
1328 if after >= before {
1329 tracing::warn!(
1330 freelist_pages = after,
1331 batches_run = batch + 1,
1332 "reclaim stopped making progress with pages still on the \
1333 freelist; the file will not shrink and the DB-size watermark \
1334 may stay engaged until the next sweep"
1335 );
1336 drained = false;
1337 break;
1338 }
1339 tokio::time::sleep(std::time::Duration::from_millis(10)).await;
1340 // `after > 0` matters: the final batch can drain the freelist
1341 // completely, in which case the loop reaches here having
1342 // SUCCEEDED and would otherwise log "with pages still on the
1343 // freelist" for an empty one — and suppress the success line.
1344 // This is the same guard `delete_in_batches` carries, and the
1345 // same defect it already had; reproduced here verbatim by
1346 // copying the loop's shape without its condition.
1347 if batch + 1 == RECLAIM_MAX_BATCHES && after > 0 {
1348 tracing::warn!(
1349 batches_run = batch + 1,
1350 freelist_pages = after,
1351 "reclaim hit its batch backstop with pages still on the \
1352 freelist; the rest waits for the next sweep"
1353 );
1354 drained = false;
1355 }
1356 }
1357 if drained {
1358 tracing::debug!("reclaim: freelist drained");
1359 }
1360 }
1361 // SQLite already returns freed pages at every commit in this mode.
1362 // Nothing to do, and a VACUUM would be pure cost.
1363 AutoVacuum::Full => {}
1364 // **Deliberately a no-op, where this used to run a full VACUUM.**
1365 //
1366 // Nothing ever set `auto_vacuum`, so NONE was the mode every database
1367 // actually ran in — which made the full-VACUUM branch the one that
1368 // always executed, daily and after every prune. A full VACUUM writes a
1369 // complete second copy of the database, so it needs free disk roughly
1370 // equal to the live file; that is precisely what is missing under the
1371 // disk pressure that triggers a retention sweep. `poll_due_once` already
1372 // carries a comment explaining this danger and removed VACUUM from the
1373 // poll path — while leaving it in the retention path that runs under the
1374 // same pressure.
1375 //
1376 // Skipping it does NOT latch the DB-size watermark, which is the failure
1377 // this branch was written to prevent: `db_size_bytes` subtracts the
1378 // freelist, so a DELETE lowers the measured size with no VACUUM at all.
1379 // What is lost is the FILE shrinking, and the fix for that is to get the
1380 // database into INCREMENTAL mode — see `migrate_to_incremental_vacuum`,
1381 // which is operator-invoked precisely because it needs the one operation
1382 // that is unsafe to attempt automatically.
1383 AutoVacuum::None => {
1384 tracing::warn!(
1385 "auto_vacuum=NONE: skipping reclaim. Freed pages stay allocated and \
1386 the file will not shrink. Run `featherreader --migrate-auto-vacuum` \
1387 once, while the volume has headroom, to move this database to \
1388 INCREMENTAL mode."
1389 );
1390 }
1391 }
1392
1393 // Truncate the WAL as well. It lives on the same volume and is counted by
1394 // `db_size_bytes`, so reclaiming database pages while leaving a WAL grown by
1395 // the sweep that just ran would give back part of the space and hold the
1396 // rest. Worth doing even in the NONE branch above, where it is the only
1397 // space this function can return at all.
1398 //
1399 // **A blocked checkpoint is the real way the file stays big, so it warns.**
1400 //
1401 // Measured: with a reader holding an open snapshot, `incremental_vacuum`
1402 // still drains the freelist and `page_count` halves — but the main file
1403 // stayed at 16.4 MB until the reader released and the checkpoint could
1404 // truncate it to 8.2 MB. So a reader does not stop the reclaim; it stops the
1405 // SHRINK. That is the operator-visible outcome (`db_size_bytes` counts the
1406 // WAL, and the watermark is compared against a volume), and it used to be
1407 // reported at `debug!` — below any realistic filter — while the loop above
1408 // warned loudly about a mechanism that does not actually occur.
1409 //
1410 // Not an error: the next sweep checkpoints again once the reader is gone.
1411 match checkpoint_wal(pool).await {
1412 Ok(true) => {}
1413 Ok(false) => tracing::warn!(
1414 "the WAL could not be truncated after reclaim (busy: a concurrent reader \
1415 OR writer held it); the freed pages are gone but the file has not \
1416 shrunk yet, and the DB-size watermark may stay engaged until the next \
1417 sweep"
1418 ),
1419 Err(err) => tracing::warn!(%err, "wal checkpoint after reclaim failed"),
1420 }
1421 Ok(())
1422}
1423
1424/// Run a truncating WAL checkpoint. `Ok(false)` means SQLite declined because a
1425/// reader held the WAL.
1426///
1427/// **The busy case is a ROW, not an error.** `PRAGMA wal_checkpoint` returns
1428/// `(busy, log_frames, checkpointed_frames)` and sets `busy = 1` when it could
1429/// not run — measured: `(1, 3, 3)` with one open read transaction versus
1430/// `(0, 0, 0)` without. So `if let Err(..)` never fires on the case it was
1431/// written for, and a caller that depends on the WAL actually being truncated
1432/// (the migration's size report does) would silently get the untruncated one.
1433async fn checkpoint_wal<'e, E>(conn: E) -> Result<bool>
1434where
1435 E: sqlx::Executor<'e, Database = sqlx::Sqlite>,
1436{
1437 let row: (i64, i64, i64) = sqlx::query_as("PRAGMA wal_checkpoint(TRUNCATE)")
1438 .fetch_one(conn)
1439 .await
1440 .context("PRAGMA wal_checkpoint(TRUNCATE) failed")?;
1441 Ok(row.0 == 0)
1442}
1443
1444/// A database's `auto_vacuum` mode.
1445#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1446pub enum AutoVacuum {
1447 /// 0 — freed pages stay on the freelist; only a full `VACUUM` returns them.
1448 None,
1449 /// 1 — SQLite returns freed pages at every commit.
1450 Full,
1451 /// 2 — freed pages are returned on demand by `PRAGMA incremental_vacuum`.
1452 Incremental,
1453}
1454
1455/// Read the database's `auto_vacuum` mode.
1456pub async fn auto_vacuum_mode(pool: &SqlitePool) -> Result<AutoVacuum> {
1457 let mode: i64 = sqlx::query_scalar("PRAGMA auto_vacuum")
1458 .fetch_one(pool)
1459 .await
1460 .context("PRAGMA auto_vacuum failed")?;
1461 Ok(match mode {
1462 1 => AutoVacuum::Full,
1463 2 => AutoVacuum::Incremental,
1464 _ => AutoVacuum::None,
1465 })
1466}
1467
1468/// What [`migrate_to_incremental_vacuum`] did.
1469#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1470pub enum VacuumMigration {
1471 /// Already in a mode that reclaims; nothing was run.
1472 NotNeeded(AutoVacuum),
1473 /// Refused: not enough free space on the volume to hold the rebuilt file.
1474 ///
1475 /// `file_bytes` is the on-disk size, reported alongside the live-page figure
1476 /// the requirement is computed from, because on exactly this population
1477 /// (`NONE` mode, large freelist) the two differ a lot and only one of them
1478 /// matches what `ls -l` says.
1479 RefusedNoHeadroom {
1480 needed: u64,
1481 available: u64,
1482 file_bytes: Option<u64>,
1483 },
1484 /// Ran the pragma + full VACUUM; the database is now INCREMENTAL.
1485 Migrated {
1486 bytes_before: i64,
1487 bytes_after: i64,
1488 file_before: Option<u64>,
1489 file_after: Option<u64>,
1490 },
1491}
1492
1493/// Move a populated database from `auto_vacuum = NONE` to `INCREMENTAL`.
1494///
1495/// **Why this cannot happen at boot.** SQLite ignores `PRAGMA auto_vacuum` on a
1496/// database that already has tables unless it is followed by a full `VACUUM`,
1497/// which rebuilds the file. So the migration off the dangerous mode requires the
1498/// exact operation that is dangerous — a genuine chicken-and-egg, and the reason
1499/// this is an explicit operator step run when the volume has headroom rather
1500/// than something attempted lazily on a machine that is already under pressure.
1501///
1502/// Doing it automatically would also reintroduce the failure shape T2.1 just
1503/// removed: a boot-time VACUUM that cannot complete on a full volume, on a
1504/// supervisor that restarts the machine whenever a child exits, is a crash loop.
1505///
1506/// `available_bytes` is the caller's measurement of free space on the database's
1507/// volume (`None` where the platform cannot report it). The check is a refusal,
1508/// not a warning: starting a VACUUM that cannot finish wastes I/O on a box that
1509/// has none to spare. `VACUUM` itself is atomic — an interrupted one leaves the
1510/// original database intact — so the risk being managed here is wasted work and
1511/// a long write-lock hold, not corruption.
1512pub async fn migrate_to_incremental_vacuum(
1513 pool: &SqlitePool,
1514 available_bytes: Option<u64>,
1515) -> Result<VacuumMigration> {
1516 let mode = auto_vacuum_mode(pool).await?;
1517 if mode != AutoVacuum::None {
1518 return Ok(VacuumMigration::NotNeeded(mode));
1519 }
1520
1521 // **The on-disk file, not the live-page count.** `db_size_bytes` subtracts
1522 // the freelist, and the population this migration exists for is precisely
1523 // `auto_vacuum = NONE` with a large freelist — so the live size can be far
1524 // smaller than the file, and an operator comparing the refusal message to
1525 // `ls -l` would not trust either number. The rebuild is sized by the LIVE
1526 // pages (that is what gets copied), but the report shows both.
1527 let bytes_before = db_size_bytes(pool).await?;
1528 let file_before = main_db_file_bytes(pool).await;
1529 // Resolved BEFORE a connection is acquired below. Asking the pool for
1530 // anything while holding one of its connections deadlocks a saturated pool —
1531 // and a single-connection pool is always saturated. The first version of the
1532 // temp-directory block did exactly that, and because `main_db_path` swallows
1533 // errors into `None` it did not even fail loudly: it stalled for the full
1534 // acquire timeout and then silently skipped setting the directory, which is
1535 // the one thing it exists to do.
1536 let temp_dir = main_db_path(pool).await.and_then(|p| {
1537 std::path::Path::new(&p)
1538 .parent()
1539 .map(std::path::Path::to_path_buf)
1540 });
1541 let needed = (bytes_before.max(0) as u64).saturating_mul(2);
1542 if let Some(available) = available_bytes {
1543 if available < needed {
1544 return Ok(VacuumMigration::RefusedNoHeadroom {
1545 needed,
1546 available,
1547 file_bytes: file_before,
1548 });
1549 }
1550 }
1551
1552 // **One connection for both statements.**
1553 //
1554 // `PRAGMA auto_vacuum` on a populated database is connection-scoped INTENT
1555 // that only takes effect when the SAME connection runs the VACUUM. Issued
1556 // against the pool they can land on different connections, and the rebuild
1557 // then happens in NONE mode — caught by the `ensure!` below, so loud rather
1558 // than silent, but the operator has paid a whole-file rewrite for nothing on
1559 // a box chosen for being short of disk.
1560 let mut conn = pool
1561 .acquire()
1562 .await
1563 .context("acquiring a connection for the auto_vacuum migration")?;
1564
1565 // **Put the temp copy on the DATABASE's volume.**
1566 //
1567 // A VACUUM rebuilds through a temporary database, and the headroom check
1568 // above measures the data volume. `temp_store = FILE` alone only chooses
1569 // file-over-memory; it does NOT choose which filesystem, so the temp copy
1570 // resolved via `SQLITE_TMPDIR`/`TMPDIR`/`/var/tmp`/`/tmp` — the container
1571 // rootfs. The check could pass on `/data` and the VACUUM still hit
1572 // `SQLITE_FULL`, or fill the rootfs out from under Caddy.
1573 //
1574 // `temp_store_directory` is the pragma that actually decides — measured:
1575 // setting it alone moves the file, setting `temp_store = FILE` alone does
1576 // not. It is deprecated but fully functional in the bundled SQLite (3.51.3,
1577 // built without `SQLITE_OMIT_DEPRECATED`), and there is no non-deprecated
1578 // equivalent reachable from a connection.
1579 //
1580 // `temp_store = FILE` is kept as belt-and-braces rather than because it is
1581 // needed: this build's compile-time default is already FILE, but a build
1582 // defaulting to MEMORY would silently ignore the directory entirely.
1583 //
1584 // Note it sets the PROCESS-GLOBAL `sqlite3_temp_directory`, not connection
1585 // state — visible on other connections and other pools. Harmless because
1586 // this function is only reachable from the one-shot `--migrate-auto-vacuum`
1587 // CLI path, which does nothing else.
1588 sqlx::query("PRAGMA temp_store = FILE")
1589 .execute(&mut *conn)
1590 .await
1591 .context("PRAGMA temp_store = FILE failed")?;
1592 if let Some(dir) = temp_dir.clone() {
1593 // The path comes from SQLite's own `database_list`, not from a caller.
1594 let quoted = dir.display().to_string().replace('\'', "''");
1595 if let Err(err) = sqlx::query(sqlx::AssertSqlSafe(format!(
1596 "PRAGMA temp_store_directory = '{quoted}'"
1597 )))
1598 .execute(&mut *conn)
1599 .await
1600 {
1601 // Not fatal: the VACUUM can still succeed if the default temp
1602 // location happens to have room. But the headroom check is then
1603 // measuring the wrong filesystem, so say so.
1604 tracing::warn!(
1605 %err, dir = %dir.display(),
1606 "could not point SQLite's temp storage at the database volume; the \
1607 headroom check may not cover where the VACUUM actually writes"
1608 );
1609 }
1610 }
1611
1612 // Order matters: the pragma records the INTENT, and the VACUUM is what
1613 // actually rewrites the file in the new mode. Reversed, the VACUUM would
1614 // rebuild in NONE mode and the pragma would then be ignored again.
1615 sqlx::query("PRAGMA auto_vacuum = INCREMENTAL")
1616 .execute(&mut *conn)
1617 .await
1618 .context("PRAGMA auto_vacuum = INCREMENTAL failed")?;
1619 sqlx::query("VACUUM")
1620 .execute(&mut *conn)
1621 .await
1622 .context("VACUUM failed during the auto_vacuum migration")?;
1623
1624 // Fold the WAL back in BEFORE measuring. A VACUUM in WAL mode writes the
1625 // entire rebuilt database through the WAL, which keeps that high-water size
1626 // until a truncating checkpoint — and `db_size_bytes` now counts the WAL. So
1627 // the one number this command reports read as "the migration doubled my
1628 // database", which is the opposite of what it did.
1629 match checkpoint_wal(&mut *conn).await {
1630 Ok(true) => {}
1631 // Reported, because the size this function returns is computed straight
1632 // after and would otherwise read as "the migration doubled my database"
1633 // with nothing saying why.
1634 Ok(false) => tracing::warn!(
1635 "the WAL could not be truncated (a concurrent reader OR writer held it), \
1636 so the reported size below includes it"
1637 ),
1638 Err(err) => tracing::warn!(%err, "post-migration wal checkpoint failed"),
1639 }
1640
1641 // Verified on the HELD connection, then released before anything that goes
1642 // back to the pool. The test pool is single-connection, and so is a
1643 // production pool that happens to be saturated — reaching for a second one
1644 // while still holding the first is a deadlock waiting for a busy moment.
1645 let after_raw: i64 = sqlx::query_scalar("PRAGMA auto_vacuum")
1646 .fetch_one(&mut *conn)
1647 .await
1648 .context("PRAGMA auto_vacuum failed after the migration")?;
1649 drop(conn);
1650 let after = match after_raw {
1651 1 => AutoVacuum::Full,
1652 2 => AutoVacuum::Incremental,
1653 _ => AutoVacuum::None,
1654 };
1655 anyhow::ensure!(
1656 after == AutoVacuum::Incremental,
1657 "the auto_vacuum migration ran but the database is still in {after:?} mode"
1658 );
1659 Ok(VacuumMigration::Migrated {
1660 bytes_before,
1661 bytes_after: db_size_bytes(pool).await?,
1662 file_before,
1663 file_after: main_db_file_bytes(pool).await,
1664 })
1665}
1666
1667/// Size of the main database FILE on disk, or `None` for `:memory:` / an
1668/// unstattable path. Distinct from [`db_size_bytes`], which reports live pages.
1669async fn main_db_file_bytes(pool: &SqlitePool) -> Option<u64> {
1670 let path = main_db_path(pool).await?;
1671 std::fs::metadata(path).ok().map(|m| m.len())
1672}
1673
1674/// Insert a batch of entries for `feed_id`, deduping on `(feed_id, guid)`, then
1675/// trim the feed to at most [`crate::config`]-configured `max_entries_per_feed`
1676/// rows (newest by published date) so one firehose feed can't fill the disk.
1677///
1678/// On a GUID collision the existing entry is updated in place (title/url/body
1679/// may have changed on re-fetch) rather than duplicated. Runs in one
1680/// transaction. Returns the number of rows processed.
1681///
1682/// `max_entries_per_feed <= 0` disables the per-feed trim.
1683pub async fn insert_entries(
1684 pool: &SqlitePool,
1685 feed_id: i64,
1686 entries: &[NewEntry],
1687 max_entries_per_feed: i64,
1688) -> Result<u64> {
1689 let mut tx = pool.begin().await.context("begin insert_entries tx")?;
1690 let mut count: u64 = 0;
1691 for e in entries {
1692 let fetched_at = e.fetched_at.clone().unwrap_or_else(now_rfc3339);
1693 let res = sqlx::query(
1694 r#"
1695 INSERT INTO entries
1696 (feed_id, guid, url, title, author, published, content_html, fetched_at)
1697 VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8)
1698 ON CONFLICT (feed_id, guid) DO UPDATE SET
1699 url = excluded.url,
1700 title = excluded.title,
1701 author = excluded.author,
1702 published = excluded.published,
1703 content_html = excluded.content_html
1704 "#,
1705 )
1706 .bind(feed_id)
1707 .bind(&e.guid)
1708 .bind(&e.url)
1709 .bind(&e.title)
1710 .bind(&e.author)
1711 .bind(&e.published)
1712 .bind(&e.content_html)
1713 .bind(&fetched_at)
1714 .execute(&mut *tx)
1715 .await
1716 .with_context(|| format!("insert entry {} failed", e.guid))?;
1717 count += res.rows_affected();
1718 }
1719
1720 // Entries-per-feed cap: keep only the newest `max_entries_per_feed` rows for
1721 // this feed, deleting the overflow in the same transaction. "Newest" is
1722 // COALESCE(published, fetched_at) so an UNDATED entry (NULL published) sorts
1723 // by when we fetched it (NOT NULL) rather than always sorting LAST and being
1724 // evicted first — otherwise a feed of undated items would trim its freshest
1725 // rows. This bounds a single firehose/misbehaving feed's storage footprint
1726 // independent of the global retention sweep. `<= 0` disables it.
1727 //
1728 // The bound is `2 * max_entries_per_feed`, not `max_entries_per_feed`: the
1729 // newest N by date, plus up to N starred. See the sparing subquery below.
1730 if max_entries_per_feed > 0 {
1731 sqlx::query(
1732 r#"
1733 DELETE FROM entries
1734 WHERE feed_id = ?1
1735 AND id NOT IN (
1736 SELECT id FROM entries
1737 WHERE feed_id = ?1
1738 ORDER BY COALESCE(published, fetched_at) DESC, id DESC
1739 LIMIT ?2
1740 )
1741 -- Starred entries survive the per-feed trim, exactly as they
1742 -- survive the retention sweep. This predicate was added to the
1743 -- sweep and NOT here, which left the documented guarantee
1744 -- ("starred entries are never evicted") false — and made this
1745 -- path, which runs on every poll of every feed rather than daily,
1746 -- the main producer of the very "starred but not cached" case the
1747 -- saved-record rendering exists to paper over.
1748 --
1749 -- The sparing is BOUNDED and SCOPED, and both matter:
1750 --
1751 -- Bounded, because the first version spared every starred row
1752 -- without limit, which did not weaken the cap so much as remove
1753 -- it — measured at cap=5 with 50 starred rows, 55 survived, 11x
1754 -- the cap. That is the same unbounded-sparing mistake the
1755 -- retention hard ceiling was added to fix, reintroduced in the
1756 -- other sweep. Worst case is now cap + cap.
1757 --
1758 -- Scoped, because `SELECT entry_id FROM entry_state WHERE
1759 -- starred = 1` reads EVERY starred row on the instance, for every
1760 -- poll of every feed — cost scaling with total users rather than
1761 -- with the feed being trimmed.
1762 AND id NOT IN (
1763 SELECT e2.id FROM entries e2
1764 WHERE e2.feed_id = ?1
1765 AND EXISTS (
1766 SELECT 1 FROM entry_state s
1767 WHERE s.entry_id = e2.id AND s.starred = 1
1768 )
1769 ORDER BY COALESCE(e2.published, e2.fetched_at) DESC, e2.id DESC
1770 LIMIT ?2
1771 )
1772 "#,
1773 )
1774 .bind(feed_id)
1775 .bind(max_entries_per_feed)
1776 .execute(&mut *tx)
1777 .await
1778 .with_context(|| format!("trimming feed {feed_id} to {max_entries_per_feed} entries"))?;
1779 }
1780
1781 // Per-feed trim above may have DELETEd entries; their ids can linger in the
1782 // read_cursor exception sets (read_ids/unread_ids have no FK to entries), so
1783 // scrub the orphaned ids out of THIS feed's cursors in the same transaction.
1784 // Bounds id-set growth and keeps the flushed PDS record from referencing
1785 // entries that no longer exist. Scoped to the one feed for cheapness.
1786 if max_entries_per_feed > 0 {
1787 prune_orphan_cursor_ids_tx(&mut tx, Some(feed_id)).await?;
1788 }
1789
1790 tx.commit().await.context("commit insert_entries tx")?;
1791 Ok(count)
1792}
1793
1794/// Make a feed due for polling on the next tick.
1795///
1796/// Used when a saved article is missing from the cache: if the reader still
1797/// subscribes to the feed, the poller may be able to bring the article back on
1798/// its own. Clearing `next_poll` is the whole mechanism — `due_feeds` treats
1799/// NULL as due — so this adds no synthetic rows and no special-case fetch path.
1800///
1801/// **Rate-limited by `not_polled_since`**, and that is not a nicety.
1802///
1803/// `due_feeds` treats a NULL `next_poll` as due immediately, so clearing it
1804/// unconditionally from a page handler meant every reload of the starred view
1805/// made those feeds due again — bypassing the poll interval entirely. That is
1806/// outbound amplification against third-party feed origins, and it lets one
1807/// reader's feeds monopolise a poll budget that is shared and already the
1808/// binding constraint on how many readers an instance can serve.
1809///
1810/// A feed polled within the window is left alone: if the article was not in the
1811/// feed a minute ago, another fetch now will not find it either. The nudge is
1812/// therefore worth at most one extra poll per feed per interval, which is the
1813/// cadence the poller already targets.
1814///
1815/// A no-op if the URL is not a known feed.
1816pub async fn mark_feed_due(
1817 pool: &SqlitePool,
1818 feed_url: &str,
1819 not_polled_since: &str,
1820) -> Result<()> {
1821 sqlx::query(
1822 "UPDATE feeds SET next_poll = NULL \
1823 WHERE url = ?1 AND (last_polled IS NULL OR last_polled < ?2)",
1824 )
1825 .bind(feed_url)
1826 .bind(not_polled_since)
1827 .execute(pool)
1828 .await
1829 .context("marking a feed due")?;
1830 Ok(())
1831}
1832
1833/// Delete entries whose age exceeds the retention window — the shared cache's
1834/// **rolling window** — except those a reader has starred or not yet read. "Age" is `COALESCE(published, fetched_at)` so an UNDATED
1835/// entry falls back to when it was fetched (never NULL) rather than being treated
1836/// as infinitely old. `entry_state` cascades via its `ON DELETE CASCADE` FK.
1837///
1838/// After the delete, orphaned entry ids are scrubbed out of every affected feed's
1839/// `read_cursor` exception sets (which have no FK to `entries`) so the id-sets do
1840/// not grow without bound and the flushed PDS record never references a vanished
1841/// entry. The caller (the retention sweep) should follow a non-zero return with
1842/// [`reclaim`] so freed pages return to the OS.
1843///
1844/// The two knobs are **independent**. `days == 0` disables the rolling window and
1845/// nothing else; `hard_days == 0` disables the ceiling and nothing else. Only
1846/// when both are off is this a no-op. Returns the number of entry rows deleted.
1847pub async fn prune_old_entries(
1848 pool: &SqlitePool,
1849 days: i64,
1850 hard_days: i64,
1851 publication_days: i64,
1852) -> Result<u64> {
1853 let now = chrono::Utc::now();
1854 // **A window too large to be a date disables that pass; it must not panic.**
1855 //
1856 // `chrono::Duration::days` and `DateTime - TimeDelta` both panic out of
1857 // range, and every knob here parses from a `u32` with no upper bound — so
1858 // `FEATHERREADER_RETENTION_DAYS=1000000000` (a plausible unit slip: seconds or
1859 // milliseconds typed into a days field) panicked this function. Measured:
1860 // anything past roughly 96 million days overflows, and `u32::MAX` does.
1861 //
1862 // The consequence was not a crash an operator would notice. This runs in a
1863 // spawned task, so tokio catches the panic and the retention sweeper simply
1864 // stops for the life of the process — silently, permanently, and taking the
1865 // release valve for `db_size_watermark_bytes` with it, which is the one thing
1866 // that stops polling for every reader.
1867 //
1868 // Disabled-not-panicking is also the answer `standard_site::ingest_floor`
1869 // already gives for the same input, and the two are supposed to mirror each
1870 // other — `Config::retention_for` exists to keep them agreeing. An
1871 // unrepresentable window meant "store everything" there and "panic" here.
1872 let at = |d: i64, knob: &str| -> Option<String> {
1873 let cutoff = chrono::Duration::try_days(d).and_then(|w| now.checked_sub_signed(w));
1874 if cutoff.is_none() {
1875 tracing::warn!(
1876 days = d,
1877 knob,
1878 "retention window is too large to express as a date; treating it as \
1879 disabled for this sweep rather than failing the sweeper"
1880 );
1881 }
1882 cutoff.map(|t| t.to_rfc3339_opts(chrono::SecondsFormat::Secs, true))
1883 };
1884
1885 let cutoff = (days > 0).then(|| at(days, "retention_days")).flatten();
1886 // **The third window, for the kinds age does not bound.** See
1887 // [`AGED_KINDS_SQL`] and `FeedKind::AGED`: a publication's entries are
1888 // bounded by COUNT (the per-feed trim), because a 14-day window stored zero
1889 // rows from every real publication measured. This is the backstop that keeps
1890 // "not aged out" from meaning "immortal" — the per-feed trim only runs when a
1891 // poll stores something, so rows belonging to a feed nobody polls any more
1892 // have nothing else to reap them.
1893 let publication_cutoff = (publication_days > 0)
1894 .then(|| at(publication_days, "publication_retention_days"))
1895 .flatten();
1896 // The ceiling only means anything if it is STRICTLY OLDER than the window.
1897 // At `0 < hard_days <= days` the two cutoffs coincide, and since the hard
1898 // delete spares nothing, it would delete exactly the rows the soft delete
1899 // exists to spare — turning the whole starred/unread exception into a no-op.
1900 // With no window at all (`days <= 0`) there is nothing to be inside of, so a
1901 // positive ceiling stands on its own.
1902 //
1903 // This used to be `hard_days.max(days)`, which clamps the wrong way: it made
1904 // `0` — the value an operator reaches for to turn a ceiling OFF, and the
1905 // documented "disabled" value for `RETENTION_DAYS` one line above it in the
1906 // same table — the single most destructive setting available, silently
1907 // purging starred and unread entries at the soft window. Measured: with
1908 // `days=14`, `hard=0` deleted a 30-day starred entry and a 30-day unread one.
1909 //
1910 // `<= 0` now means disabled, consistently with `days`. A contradictory
1911 // positive value is refused rather than reinterpreted downward.
1912 //
1913 // The ceiling is deliberately NOT gated on the window being enabled. It used
1914 // to be — this function returned on `days <= 0` before the ceiling was even
1915 // computed — which made `RETENTION_DAYS=0` mean "no window AND no ceiling":
1916 // the one configuration with no bound on the shared cache whatsoever. That
1917 // became load-bearing when the per-feed trim started sparing starred entries.
1918 // Before, the trim was a backstop for them; now nothing was. "I don't want a
1919 // rolling window" and "I don't want any ceiling at all" are different
1920 // statements, and are now configured separately.
1921 let hard_cutoff = if hard_days > 0 && (days <= 0 || hard_days > days) {
1922 at(hard_days, "retention_hard_days")
1923 } else {
1924 if hard_days > 0 {
1925 tracing::warn!(
1926 hard_days,
1927 days,
1928 "retention hard ceiling is not older than the retention window; \
1929 ignoring it — set it above the window or to 0 to disable"
1930 );
1931 }
1932 None
1933 };
1934
1935 if cutoff.is_none() && hard_cutoff.is_none() && publication_cutoff.is_none() {
1936 return Ok(0);
1937 }
1938
1939 // **The hard ceiling — the bound that sparing would otherwise remove.**
1940 //
1941 // Sparing `read = 0` is not a small exception: "mark unread" is a one-click
1942 // UI control, and `entries` is SHARED across every reader on the instance.
1943 // Without a ceiling, one person can pin unbounded rows, and the pins are
1944 // permanent.
1945 //
1946 // That matters beyond disk. `poll_due_once` stops ALL polling once the
1947 // database crosses `db_size_watermark_bytes`, and the retention DELETE is
1948 // the documented release valve. Pinned rows can hold the valve shut
1949 // forever, so the failure mode is: one reader pins enough content, the DB
1950 // latches above the watermark, and polling stops for EVERY reader with no
1951 // self-healing path. The window used to be an unconditional bound; sparing
1952 // removed it, and this restores it.
1953 //
1954 // Starred entries go too at this age, and that is now safe: a saved record
1955 // whose entry is gone renders from the PDS record as a link card, so the
1956 // reader keeps the article's identity even when the cache does not keep its
1957 // text.
1958 let hard_deleted = match &hard_cutoff {
1959 Some(cutoff) => {
1960 delete_in_batches(
1961 pool,
1962 // Scoped to the kinds the window applies to. A publication's
1963 // entries answer to `publication_cutoff` below instead, which is
1964 // generous where this is tight — an archive read is not a cache
1965 // of the last few days.
1966 &format!(
1967 "SELECT id FROM entries WHERE COALESCE(published, fetched_at) < ?1 \
1968 AND feed_id IN (SELECT id FROM feeds WHERE kind IN ({AGED_KINDS_SQL}))"
1969 ),
1970 cutoff,
1971 "hard ceiling",
1972 )
1973 .await?
1974 }
1975 None => 0,
1976 };
1977 // **Entries a reader has DELIBERATELY marked are kept, whatever their age.**
1978 //
1979 // Precisely: an entry is spared when some DID has an `entry_state` row for
1980 // it with `starred = 1` or `read = 0`. An entry nobody has ever touched has
1981 // no `entry_state` row at all and is NOT spared, even though every read path
1982 // treats "no row" as unread.
1983 //
1984 // That asymmetry is deliberate and load-bearing. Sparing every never-touched
1985 // entry would spare essentially the whole table — almost no entry is ever
1986 // interacted with — which would make the window a no-op and leave the hard
1987 // ceiling as the only bound. The window is for evicting cache nobody claimed;
1988 // the exception is for the things a reader acted on.
1989 //
1990 // This comment used to read "starred and unread entries are kept", which is
1991 // the reading that would motivate exactly that change.
1992 //
1993 // The window is a cache eviction policy, not a data-retention policy. The
1994 // PDS is the source of truth for what a reader CHOSE — subscriptions,
1995 // folders, stars, read-state — but the entry CONTENT was never there. It
1996 // exists here and at the origin feed, and a feed typically serves only its
1997 // last few dozen items, so a pruned article is usually unrecoverable.
1998 //
1999 // Deleting indiscriminately therefore lost two things a reader would notice:
2000 // a starred article vanished from the starred view entirely (the view joins
2001 // `entries`, and `entry_state` cascades on the delete, so the star went with
2002 // it), and anything still unread disappeared before it was ever read. Both
2003 // are the opposite of a cache.
2004 //
2005 // This is what the documentation has always described; the query did not
2006 // implement it.
2007 let soft_deleted = match &cutoff {
2008 Some(cutoff) => {
2009 delete_in_batches(
2010 pool,
2011 // **`NOT EXISTS`, not `id NOT IN (…)`.**
2012 //
2013 // The list form materialises the ENTIRE pinned set on every
2014 // batch, and that set scales with total users rather than with
2015 // the feed being swept; this probes `idx_entry_state_entry_id`
2016 // per candidate row instead. Measured on 1M entries with 600k
2017 // `entry_state` rows of which 10% are pinned: **64.8 s as a list,
2018 // 43.6 s as a correlated exists — 1.49x, for no disk and no write
2019 // amplification.**
2020 //
2021 // **An earlier version of this comment claimed 2.4x, and that a
2022 // partial index on the pinned predicate "changed the time by
2023 // nothing at all". Both were artifacts of a bad fixture.** It
2024 // made every `entry_state` row match `starred = 1 OR read = 0` —
2025 // no "read and not starred" rows at all, which is the commonest
2026 // state a reader leaves behind. That inflated the list form's
2027 // cost (the materialised set was the whole table) and made a
2028 // PARTIAL index on that predicate cover 100% of rows, so it could
2029 // not be selective and duly did nothing.
2030 //
2031 // On a realistic distribution the review's proposed index is NOT
2032 // useless: it takes the list form from 64.8 s to 44.0 s, most of
2033 // the way to the rewrite. The rewrite is still the better change
2034 // because it costs no disk and no insert throughput — but it wins
2035 // by less than claimed, against an alternative that was dismissed
2036 // on a measurement of the wrong thing.
2037 //
2038 // Indexes are still declined, now on honest numbers: the pinned
2039 // index buys 12% (43.6 → 38.5 s) for 6.9 MiB, the age index 22%
2040 // (→ 33.9 s) for 27.9 MiB, both with write amplification on a
2041 // poller that inserts constantly, against a daily sweep that is
2042 // already batched and interruptible. See
2043 // `store::tests::r6_measure_retention_sweep`.
2044 //
2045 // Also strictly safer. `NOT IN` against a subquery containing a
2046 // NULL evaluates to NULL for every row, which would silently
2047 // delete nothing. `entry_state.entry_id` is `NOT NULL` today, so
2048 // the two are equivalent — but the equivalence depends on a
2049 // column constraint somewhere else, and `NOT EXISTS` does not.
2050 // `sparing_honours_every_did_not_just_one` pins the multi-DID
2051 // case, which is the only one where the forms could diverge.
2052 &format!(
2053 "SELECT e.id FROM entries e \
2054 WHERE COALESCE(e.published, e.fetched_at) < ?1 \
2055 AND e.feed_id IN \
2056 (SELECT id FROM feeds WHERE kind IN ({AGED_KINDS_SQL})) \
2057 AND NOT EXISTS ( \
2058 SELECT 1 FROM entry_state s \
2059 WHERE s.entry_id = e.id \
2060 AND (s.starred = 1 OR s.read = 0) \
2061 )"
2062 ),
2063 cutoff,
2064 "window",
2065 )
2066 .await?
2067 }
2068 None => 0,
2069 };
2070 // **The archive ceiling, for every kind the window does not cover.**
2071 //
2072 // `kind NOT IN` rather than `kind = 'publication'` deliberately: a kind added
2073 // later and left out of `FeedKind::AGED` inherits a bound here rather than
2074 // inheriting immortality. Spares nothing, for the reason the hard ceiling
2075 // spares nothing — a saved record whose entry is gone still renders from the
2076 // PDS record as a link card, so the reader keeps the article's identity.
2077 let publication_deleted = match &publication_cutoff {
2078 Some(cutoff) => {
2079 delete_in_batches(
2080 pool,
2081 &format!(
2082 "SELECT id FROM entries WHERE COALESCE(published, fetched_at) < ?1 \
2083 AND feed_id IN (SELECT id FROM feeds WHERE kind NOT IN ({AGED_KINDS_SQL}))"
2084 ),
2085 cutoff,
2086 "archive ceiling",
2087 )
2088 .await?
2089 }
2090 None => 0,
2091 };
2092 let deleted = soft_deleted + hard_deleted + publication_deleted;
2093
2094 // Only touch cursors when rows actually went away — and OUTSIDE the deletes.
2095 //
2096 // This used to run inside the one transaction that wrapped both deletes,
2097 // which made the whole sweep a single write-lock hold: load every
2098 // `read_cursor` row, then issue a fresh per-cursor `SELECT … JOIN … WHERE
2099 // f.url = ?` returning up to `max_entries_per_feed` ids, all before the
2100 // commit. SQLite is single-writer and `busy_timeout` is 5 s, so for that
2101 // whole span every mark-read, every login write and every cursor flush
2102 // failed.
2103 //
2104 // Correctness survives the move because the scrub is idempotent — it
2105 // computes each cursor's surviving ids from what is in `entries` NOW, and
2106 // rewrites only cursors that actually change. If the process dies between
2107 // the deletes and the scrub, the next sweep finishes the job, and in the
2108 // meantime a stale id in an exception set is inert: the flusher sends it,
2109 // and it names an entry nobody can reach.
2110 if deleted > 0 {
2111 if let Err(err) = prune_orphan_cursor_ids(pool, None).await {
2112 // The deletes already committed and are the point of this call.
2113 // A failed scrub leaves stale ids to be cleaned up next sweep.
2114 tracing::warn!(%err, "retention sweep: cursor id scrub failed after the deletes");
2115 }
2116 }
2117
2118 Ok(deleted)
2119}
2120
2121/// Rows deleted per statement by [`delete_in_batches`].
2122///
2123/// Small enough that one batch — including its `entry_state` FK cascade — is a
2124/// short lock hold, large enough that a big sweep is tens of statements rather
2125/// than thousands.
2126const PRUNE_BATCH: i64 = 1_000;
2127
2128/// Backstop against a delete loop that never drains. `rows_affected == 0` is the
2129/// real terminator; this only bounds the damage if a future predicate change
2130/// makes that untrue. At [`PRUNE_BATCH`] this is 10M rows, far past anything a
2131/// 1 GB volume holds.
2132const PRUNE_MAX_BATCHES: usize = 10_000;
2133
2134/// How long [`delete_in_batches`] stands down between batches, so a writer
2135/// waiting on the SQLite write lock actually gets it rather than losing the race
2136/// to the loop's next statement.
2137///
2138/// Named because it is the one thing that makes batching a fix rather than
2139/// bookkeeping, and because `a_writer_gets_through_while_the_sweep_runs` derives
2140/// its "was this sweep long enough to measure" floor from it. A sweep that is
2141/// genuinely batched cannot finish faster than one hand-off per batch; that is a
2142/// structural lower bound, not a number calibrated against a particular machine.
2143const PRUNE_BATCH_HANDOFF: std::time::Duration = std::time::Duration::from_millis(10);
2144
2145/// Delete every entry matched by `select_ids` (a `SELECT id FROM entries …`
2146/// bound to one `?1` cutoff), in bounded batches, **one implicit transaction per
2147/// batch**.
2148///
2149/// The retention sweep used to be a single `DELETE` inside one explicit
2150/// transaction. On a populated instance that is one unbroken write-lock hold
2151/// covering tens of thousands of row deletes plus their `entry_state` cascades —
2152/// measured at ~10 minutes before `idx_entry_state_entry_id` existed, and still
2153/// a single indivisible span after it. Everything else that writes (mark-read,
2154/// login, cursor flush) has a 5 s `busy_timeout` and simply fails for the
2155/// duration.
2156///
2157/// Batching does not make the total work smaller; it makes it INTERRUPTIBLE. A
2158/// writer waiting on the lock gets in between batches instead of timing out, and
2159/// the short sleep below guarantees that window actually exists rather than
2160/// leaving it to chance against a tight loop.
2161///
2162/// A partial sweep is safe: each batch commits on its own, and the predicate is
2163/// a fixed cutoff, so a crash mid-sweep leaves fewer rows deleted and the next
2164/// run finishes the job.
2165async fn delete_in_batches(
2166 pool: &SqlitePool,
2167 select_ids: &str,
2168 cutoff: &str,
2169 label: &str,
2170) -> Result<u64> {
2171 let sql = format!("DELETE FROM entries WHERE id IN ({select_ids} LIMIT {PRUNE_BATCH})");
2172 let mut total: u64 = 0;
2173 for batch in 0..PRUNE_MAX_BATCHES {
2174 let n = sqlx::query(sqlx::AssertSqlSafe(sql.clone()))
2175 .bind(cutoff)
2176 .execute(pool)
2177 .await
2178 .with_context(|| format!("prune_old_entries {label} (cutoff {cutoff})"))?
2179 .rows_affected();
2180 total += n;
2181 if n == 0 {
2182 return Ok(total);
2183 }
2184 // Hand the write lock over, so the loop cannot re-acquire it the instant
2185 // it commits and leave a waiting writer to fight for the gap between two
2186 // statements. At `PRUNE_BATCH` rows per batch this adds one
2187 // `PRUNE_BATCH_HANDOFF` per 1,000 deleted rows to a sweep that runs once
2188 // a day.
2189 //
2190 // This comment has twice carried a number it could not support. It first
2191 // said a writer "still starves" without the hand-off; that was replaced
2192 // with "roughly 3x writer throughput", quoting one sample from each of
2193 // two runs. Repeated, the two distributions overlap heavily (medians
2194 // ~1.4 writes/ms with the sleep against ~1.0 without, and several
2195 // sleep-less runs beat the median with it), so 3x is not a figure this
2196 // comment can assert.
2197 //
2198 // What is defensible without a benchmark: removing it lets the loop
2199 // re-acquire immediately, so a waiting writer is left racing the gap
2200 // between two statements instead of being handed a window. Writers do
2201 // still get through either way. `a_writer_gets_through_while_the_sweep_runs`
2202 // catches the removal about three runs in five — see the note there; the
2203 // rest of the time the loop still looks batched, because it is.
2204 tokio::time::sleep(PRUNE_BATCH_HANDOFF).await;
2205 // Only warn if the backstop actually cut the sweep short. A final batch
2206 // that happened to drain the last rows would otherwise log "the rest
2207 // waits for the next run" with nothing left — and an operator who reads
2208 // that during an incident would go looking for a backlog that is not
2209 // there. `n < PRUNE_BATCH` means this batch found fewer rows than it
2210 // asked for, so there are none behind it.
2211 // Still a 1-in-`PRUNE_BATCH` false positive when the final batch drains
2212 // exactly a full batch with nothing behind it — distinguishing that
2213 // needs another COUNT per sweep, which is not worth paying to make a
2214 // backstop message that has never fired slightly more precise.
2215 if batch + 1 == PRUNE_MAX_BATCHES && n == PRUNE_BATCH as u64 {
2216 tracing::warn!(
2217 label,
2218 total,
2219 "retention sweep hit its batch backstop; the rest waits for the next run"
2220 );
2221 }
2222 }
2223 Ok(total)
2224}
2225
2226/// Scrub entry ids that no longer exist out of `read_cursor.read_ids` /
2227/// `unread_ids`. `read_cursor` is keyed by `(did, feed_url)` and its id-sets have
2228/// NO foreign key to `entries`, so a prune/trim that deletes entries would
2229/// otherwise leave dangling ids that (a) grow the sets without bound and (b) get
2230/// flushed to the PDS as references to vanished entries.
2231///
2232/// When `feed_id` is `Some`, only that feed's cursors are examined (the cheap
2233/// path used right after a per-feed trim); `None` scans every cursor (the
2234/// retention sweep, which can delete across many feeds at once). A cursor whose
2235/// sets actually change is rewritten and marked `dirty` so the flusher resyncs
2236/// it; unchanged cursors are left untouched (no spurious dirtying / PDS writes).
2237/// Returns the number of cursor rows modified.
2238///
2239/// This is the TRANSACTIONAL variant, used by the per-feed trim inside
2240/// `insert_entries`: it is scoped to one feed, examines that feed's cursors
2241/// only, and genuinely wants to land atomically with the trim that created the
2242/// orphans. The retention sweep uses [`prune_orphan_cursor_ids`] instead —
2243/// global scope inside one transaction is what made the sweep a multi-minute
2244/// write-lock hold.
2245async fn prune_orphan_cursor_ids_tx(
2246 tx: &mut sqlx::Transaction<'_, sqlx::Sqlite>,
2247 feed_id: Option<i64>,
2248) -> Result<u64> {
2249 // The set of live entry ids we prune against. Scope to the feed's URL when a
2250 // feed_id is given so we filter only that feed's cursors against that feed's
2251 // entries; otherwise consider all cursors / all entries.
2252 let feed_url = match feed_id {
2253 Some(fid) => match feed_url_for_id_tx(tx, fid).await? {
2254 Some(u) => Some(u),
2255 None => return Ok(0), // feed vanished mid-tx; nothing to prune
2256 },
2257 None => None,
2258 };
2259
2260 // Load the (did, feed_url, read_ids, unread_ids) of the candidate cursors.
2261 let cursors: Vec<(String, String, String, String)> = match &feed_url {
2262 Some(url) => sqlx::query(
2263 "SELECT did, feed_url, read_ids, unread_ids FROM read_cursor WHERE feed_url = ?1",
2264 )
2265 .bind(url)
2266 .fetch_all(&mut **tx)
2267 .await
2268 .context("prune_orphan_cursor_ids: load feed cursors")?,
2269 None => sqlx::query("SELECT did, feed_url, read_ids, unread_ids FROM read_cursor")
2270 .fetch_all(&mut **tx)
2271 .await
2272 .context("prune_orphan_cursor_ids: load all cursors")?,
2273 }
2274 .into_iter()
2275 .map(|r| {
2276 (
2277 r.get::<String, _>("did"),
2278 r.get::<String, _>("feed_url"),
2279 r.get::<String, _>("read_ids"),
2280 r.get::<String, _>("unread_ids"),
2281 )
2282 })
2283 .collect();
2284
2285 if cursors.is_empty() {
2286 return Ok(0);
2287 }
2288
2289 let now = now_rfc3339();
2290 let mut changed: u64 = 0;
2291 for (did, curl, read_ids, unread_ids) in cursors {
2292 // The live entry ids for THIS cursor's feed (join by URL — the cursor key).
2293 let live: std::collections::HashSet<i64> = sqlx::query_scalar::<_, i64>(
2294 "SELECT e.id FROM entries e JOIN feeds f ON f.id = e.feed_id WHERE f.url = ?1",
2295 )
2296 .bind(&curl)
2297 .fetch_all(&mut **tx)
2298 .await
2299 .with_context(|| format!("prune_orphan_cursor_ids: live ids for {curl}"))?
2300 .into_iter()
2301 .collect();
2302
2303 let new_read = filter_id_set_to_live(&read_ids, &live);
2304 let new_unread = filter_id_set_to_live(&unread_ids, &live);
2305 if new_read == read_ids && new_unread == unread_ids {
2306 continue; // nothing orphaned — leave the cursor (and its dirty flag) alone
2307 }
2308 sqlx::query(
2309 "UPDATE read_cursor SET read_ids = ?3, unread_ids = ?4, dirty = 1, updated_at = ?5 \
2310 WHERE did = ?1 AND feed_url = ?2",
2311 )
2312 .bind(&did)
2313 .bind(&curl)
2314 .bind(&new_read)
2315 .bind(&new_unread)
2316 .bind(&now)
2317 .execute(&mut **tx)
2318 .await
2319 .with_context(|| format!("prune_orphan_cursor_ids: rewrite cursor {did}/{curl}"))?;
2320 changed += 1;
2321 }
2322 Ok(changed)
2323}
2324
2325/// [`prune_orphan_cursor_ids_tx`] over the pool — **no enclosing transaction**.
2326///
2327/// Same result, different locking. Each statement commits on its own, so the
2328/// single write lock is taken for one cursor rewrite at a time and released
2329/// between them, and the reads in between block nothing at all in WAL mode.
2330/// That matters because this is the global pass: the retention sweep's version
2331/// loads EVERY `read_cursor` row and then issues one live-ids query per cursor,
2332/// and holding all of that inside a transaction is what made a daily sweep look
2333/// like an outage to every writer on the instance.
2334///
2335/// **Each cursor's read-modify-write is one short transaction**, and that is not
2336/// optional. The first version of this loaded every cursor into a snapshot, then
2337/// walked them issuing an unguarded `UPDATE` per cursor from that snapshot. A
2338/// `mark_read` landing during the walk — seconds, on a global pass — had its new
2339/// id silently overwritten by the stale set, and the rewrite set `dirty = 1`, so
2340/// the flusher then pushed the truncated set to the PDS as authoritative. Local
2341/// `entry_state` still said read, so the loss was invisible here and visible
2342/// only in every OTHER atproto client. The transactional predecessor did not
2343/// have that bug: it held the write lock across the whole pass, so a concurrent
2344/// `mark_read` blocked and applied on top.
2345///
2346/// So the lock is not eliminated, it is SCOPED: one cursor's live-ids query plus
2347/// its update, rather than every cursor's. That keeps what T2.2 was for (a daily
2348/// sweep must not look like an outage) without trading it for lost writes.
2349///
2350/// Re-running is still safe — surviving ids are recomputed from the current
2351/// contents of `entries` — so dying partway just means the next sweep finishes.
2352///
2353/// `feed_id = Some(..)` scopes to one feed; `None` scans every cursor. Returns
2354/// the number of cursor rows modified.
2355async fn prune_orphan_cursor_ids(pool: &SqlitePool, feed_id: Option<i64>) -> Result<u64> {
2356 let feed_url = match feed_id {
2357 Some(fid) => match sqlx::query_scalar::<_, String>("SELECT url FROM feeds WHERE id = ?1")
2358 .bind(fid)
2359 .fetch_optional(pool)
2360 .await
2361 .context("prune_orphan_cursor_ids: feed url")?
2362 {
2363 Some(u) => Some(u),
2364 None => return Ok(0),
2365 },
2366 None => None,
2367 };
2368
2369 // Only the KEYS come from this snapshot. The id-sets are deliberately not
2370 // read here — they are re-read inside each cursor's own transaction below,
2371 // because anything read out here is stale by the time it is written back.
2372 let keys: Vec<(String, String)> = match &feed_url {
2373 Some(url) => sqlx::query_as("SELECT did, feed_url FROM read_cursor WHERE feed_url = ?1")
2374 .bind(url)
2375 .fetch_all(pool)
2376 .await
2377 .context("prune_orphan_cursor_ids: load feed cursors")?,
2378 None => sqlx::query_as("SELECT did, feed_url FROM read_cursor")
2379 .fetch_all(pool)
2380 .await
2381 .context("prune_orphan_cursor_ids: load all cursors")?,
2382 };
2383
2384 let mut changed: u64 = 0;
2385 for (did, curl) in keys {
2386 // A cursor that vanished between the key snapshot and now is simply
2387 // skipped; a cursor that APPEARED is missed until the next sweep. Both
2388 // are fine — the scrub is housekeeping, not a correctness barrier.
2389 match scrub_one_cursor(pool, &did, &curl).await {
2390 Ok(true) => changed += 1,
2391 Ok(false) => {}
2392 // One bad cursor must not abandon the rest of the pass.
2393 Err(err) => tracing::warn!(%err, %did, feed = %curl, "cursor id scrub failed"),
2394 }
2395 }
2396 Ok(changed)
2397}
2398
2399/// Scrub one cursor's id-sets inside its own transaction. Returns whether the
2400/// row changed.
2401///
2402/// The read of the id-sets, the live-ids query and the write all happen under
2403/// one transaction, so a `mark_read` that lands mid-sweep either goes first (and
2404/// is included) or waits (and applies on top). Reading the sets outside and
2405/// writing them back later is the lost-update shape this function exists to
2406/// avoid — see [`prune_orphan_cursor_ids`].
2407async fn scrub_one_cursor(pool: &SqlitePool, did: &str, feed_url: &str) -> Result<bool> {
2408 let mut tx = pool.begin().await.context("begin scrub_one_cursor tx")?;
2409
2410 let (_, read_ids, unread_ids) = cursor_sets(&mut tx, did, feed_url).await?;
2411 // An empty exception set has nothing to orphan, and skipping it avoids the
2412 // live-ids query entirely — the dominant cost of this pass, and the common
2413 // case for a cursor sitting at its high-water mark.
2414 if is_empty_id_set(&read_ids) && is_empty_id_set(&unread_ids) {
2415 return Ok(false);
2416 }
2417
2418 let live: std::collections::HashSet<i64> = sqlx::query_scalar::<_, i64>(
2419 "SELECT e.id FROM entries e JOIN feeds f ON f.id = e.feed_id WHERE f.url = ?1",
2420 )
2421 .bind(feed_url)
2422 .fetch_all(&mut *tx)
2423 .await
2424 .with_context(|| format!("prune_orphan_cursor_ids: live ids for {feed_url}"))?
2425 .into_iter()
2426 .collect();
2427
2428 let new_read = filter_id_set_to_live(&read_ids, &live);
2429 let new_unread = filter_id_set_to_live(&unread_ids, &live);
2430 if new_read == read_ids && new_unread == unread_ids {
2431 return Ok(false); // nothing orphaned — leave the cursor (and its dirty flag) alone
2432 }
2433 sqlx::query(
2434 "UPDATE read_cursor SET read_ids = ?3, unread_ids = ?4, dirty = 1, updated_at = ?5 \
2435 WHERE did = ?1 AND feed_url = ?2",
2436 )
2437 .bind(did)
2438 .bind(feed_url)
2439 .bind(&new_read)
2440 .bind(&new_unread)
2441 .bind(now_rfc3339())
2442 .execute(&mut *tx)
2443 .await
2444 .with_context(|| format!("prune_orphan_cursor_ids: rewrite cursor {did}/{feed_url}"))?;
2445 tx.commit().await.context("commit scrub_one_cursor tx")?;
2446 Ok(true)
2447}
2448
2449/// Whether a stored id-set is *textually* empty — `[]` or blank.
2450///
2451/// Deliberately NOT a parse: this is a fast pre-filter, and
2452/// [`filter_id_set_to_live`] remains the authority on what a set contains. An
2453/// unparseable value returns `false` here, so it goes through the full path and
2454/// gets canonicalised to `[]` rather than being skipped — the pre-filter fails
2455/// toward doing the work, which is the safe direction.
2456fn is_empty_id_set(raw: &str) -> bool {
2457 let t = raw.trim();
2458 t.is_empty() || t == "[]"
2459}
2460
2461/// Filter a JSON id-array string down to only ids present in `live`, returning
2462/// the canonical JSON-array-of-strings form (matching [`json_id_set_toggle`]). A
2463/// malformed input yields `[]`.
2464fn filter_id_set_to_live(raw: &str, live: &std::collections::HashSet<i64>) -> String {
2465 let ids: Vec<i64> = serde_json::from_str::<Vec<serde_json::Value>>(raw)
2466 .ok()
2467 .map(|vals| {
2468 vals.into_iter()
2469 .filter_map(|v| match v {
2470 serde_json::Value::Number(n) => n.as_i64(),
2471 serde_json::Value::String(s) => s.parse::<i64>().ok(),
2472 _ => None,
2473 })
2474 .filter(|id| live.contains(id))
2475 .collect()
2476 })
2477 .unwrap_or_default();
2478 let as_strings: Vec<String> = ids.iter().map(|i| i.to_string()).collect();
2479 serde_json::to_string(&as_strings).unwrap_or_else(|_| "[]".to_string())
2480}
2481
2482/// Replace the per-DID subscription projection (`sub_ref`) for `did` with
2483/// exactly `feed_ids`, in one transaction.
2484///
2485/// Called from the web layer's subscription-resolve/sync path so `sub_ref`
2486/// always mirrors the caller's *current* PDS subscription set. This is the
2487/// authority every scoped read/mutation checks against — a feed the caller no
2488/// longer subscribes to drops out of their read surface immediately.
2489pub async fn replace_sub_refs(pool: &SqlitePool, did: &str, feed_ids: &[i64]) -> Result<()> {
2490 let mut tx = pool.begin().await.context("begin replace_sub_refs tx")?;
2491 sqlx::query("DELETE FROM sub_ref WHERE did = ?1")
2492 .bind(did)
2493 .execute(&mut *tx)
2494 .await
2495 .with_context(|| format!("clear sub_ref for {did}"))?;
2496 for &feed_id in feed_ids {
2497 sqlx::query("INSERT OR IGNORE INTO sub_ref (did, feed_id) VALUES (?1, ?2)")
2498 .bind(did)
2499 .bind(feed_id)
2500 .execute(&mut *tx)
2501 .await
2502 .with_context(|| format!("insert sub_ref {did}/{feed_id}"))?;
2503 }
2504 tx.commit().await.context("commit replace_sub_refs tx")?;
2505 Ok(())
2506}
2507
2508/// Whether `did` currently subscribes to the feed `feed_id` owns
2509/// (i.e. a `sub_ref` row exists). The authorization primitive behind every
2510/// per-DID scoped read/mutation.
2511pub async fn did_subscribes_to_entry(pool: &SqlitePool, did: &str, entry_id: i64) -> Result<bool> {
2512 let found: Option<i64> = sqlx::query_scalar(
2513 r#"
2514 SELECT 1
2515 FROM entries e
2516 JOIN sub_ref sr ON sr.feed_id = e.feed_id AND sr.did = ?1
2517 WHERE e.id = ?2
2518 "#,
2519 )
2520 .bind(did)
2521 .bind(entry_id)
2522 .fetch_optional(pool)
2523 .await
2524 .with_context(|| format!("did_subscribes_to_entry failed for {did}/{entry_id}"))?;
2525 Ok(found.is_some())
2526}
2527
2528/// The exact `(sql, bind_count)` `list_entries` runs, for a view and scope.
2529///
2530/// **One path, so a test cannot assert on something the query is free to
2531/// ignore.** A named `LIST_PROJECTION` constant was not enough: the test read
2532/// the constant while `list_entries` passed `list_query_sql` whatever it liked,
2533/// so swapping in an inline literal containing `e.content_html` still shipped
2534/// green. The test now calls this.
2535fn list_entries_sql(view: ListView, feed_ids: Option<&[i64]>) -> (String, usize) {
2536 list_query_sql(Projection::EntryList, view, feed_ids)
2537}
2538
2539/// Which columns a list query may select.
2540///
2541/// **A closed type, not a `&str`.** A named constant was not enough and neither
2542/// was a helper function: both left `list_query_sql` taking an arbitrary string,
2543/// so a call site could pass an inline literal containing `e.content_html` and
2544/// ship green — twice over, which is how this ended up as an enum. The article
2545/// body is up to 20 KB per row and the list renders 50 at a time, so reading it
2546/// is the difference between a bounded response and a megabyte per page.
2547#[derive(Debug, Clone, Copy, PartialEq, Eq)]
2548enum Projection {
2549 /// The list view. Deliberately omits `content_html`.
2550 EntryList,
2551 Count,
2552 Ids,
2553 FeedCounts,
2554 StarredUrls,
2555}
2556
2557impl Projection {
2558 const fn columns(self) -> &'static str {
2559 match self {
2560 Projection::EntryList => {
2561 "e.id, e.feed_id, e.guid, e.url, e.title, e.published, \
2562 COALESCE(s.read, 0) AS read, COALESCE(s.starred, 0) AS starred"
2563 }
2564 Projection::Count => "COUNT(*)",
2565 Projection::Ids => "e.id",
2566 Projection::FeedCounts => "e.feed_id, COUNT(*)",
2567 Projection::StarredUrls => "e.url, e.guid",
2568 }
2569 }
2570}
2571
2572/// The shared body of every list query: the per-DID `entry_state` LEFT JOIN, the
2573/// `sub_ref` authorization predicate, the view predicate and the optional
2574/// feed-id restriction. `projection` is spliced in as the `SELECT` list.
2575///
2576/// Returns the SQL plus the number of feed-id placeholders emitted, so the
2577/// caller knows where its own `LIMIT`/`OFFSET` placeholders start. `?1` is
2578/// always the DID; feed ids are `?2..`.
2579///
2580/// **Why the callers may assert this is SQL-safe.** Only three things vary, and
2581/// none is caller data: `projection` and [`ListView::predicate`] are `&'static
2582/// str` written in this file, and the feed-id restriction contributes only a
2583/// COUNT — the ids themselves are bound, never formatted in. Every runtime value
2584/// (the DID, the ids, the limit, the offset) reaches SQLite as a bind parameter.
2585fn list_query_sql(
2586 projection: Projection,
2587 view: ListView,
2588 feed_ids: Option<&[i64]>,
2589) -> (String, usize) {
2590 let cols = projection.columns();
2591 let scoped = feed_ids.is_some();
2592 let mut sql = format!(
2593 "SELECT {cols} \
2594 FROM entries e \
2595 LEFT JOIN entry_state s ON s.entry_id = e.id AND s.did = ?1 \
2596 WHERE {} \
2597 AND EXISTS ( \
2598 SELECT 1 FROM sub_ref sr \
2599 WHERE sr.did = ?1 AND sr.feed_id = e.feed_id \
2600 )",
2601 view.predicate()
2602 );
2603 if scoped {
2604 // **ONE bind parameter for any scope size.**
2605 //
2606 // This used to emit one placeholder per feed id, so the SQL string and
2607 // the bind list both grew with the reader's subscription count — which
2608 // is PDS-supplied and bounded only by the 20,000-record list ceiling.
2609 //
2610 // That was reachable-broken, not merely ugly: `SQLITE_LIMIT_VARIABLE_NUMBER`
2611 // is 32766 on the bundled build, and the ids were bound TWICE per render
2612 // (the count query and the page query), so the effective ceiling was
2613 // ~16,383 feeds — below the list ceiling. Past it, `prepare` fails with
2614 // "too many SQL variables" and the reader's page 500s. Measured: 20,000
2615 // ids through `json_each` is a 108 KB bind that runs in 9.9 ms; 32,767
2616 // placeholders does not prepare at all.
2617 // The first attempt at bounding it truncated the subscription list
2618 // instead, which traded a query-shape problem for an access problem:
2619 // `sync_sub_refs` writes `sub_ref` from that list, so dropped feeds
2620 // became unreadable AND unmutatable. `json_each` removes the need to
2621 // choose — the whole set rides in as one JSON text bind.
2622 sql.push_str(" AND e.feed_id IN (SELECT value FROM json_each(?2))");
2623 }
2624 (sql, usize::from(scoped))
2625}
2626
2627/// Bind the DID and the optional feed-id restriction, in the order
2628/// [`list_query_sql`] emits them — `?1` the DID, `?2` the scope JSON when there
2629/// is one.
2630fn bind_list_scope<'q, O>(
2631 q: sqlx::query::QueryAs<'q, sqlx::Sqlite, O, sqlx::sqlite::SqliteArguments>,
2632 did: &'q str,
2633 feed_ids: Option<&[i64]>,
2634) -> sqlx::query::QueryAs<'q, sqlx::Sqlite, O, sqlx::sqlite::SqliteArguments> {
2635 let q = q.bind(did);
2636 match feed_ids {
2637 // Serialising i64s cannot fail; the fallback is an empty array, which
2638 // matches nothing — the fail-closed direction for a scope filter.
2639 Some(ids) => q.bind(serde_json::to_string(ids).unwrap_or_else(|_| "[]".to_string())),
2640 None => q,
2641 }
2642}
2643
2644/// One page of a list view, newest-published first, scoped to `did`'s
2645/// subscriptions (`sub_ref`) and optionally narrowed to `feed_ids`.
2646///
2647/// **`limit` is a required parameter, not a convenience.** This function
2648/// replaced three `SELECT e.*` queries that had no `LIMIT` at all and pulled the
2649/// article body they never used; leaving an unbounded variant next to the
2650/// bounded one would just be the same trap with a longer name. If a caller wants
2651/// "everything", it has to say how much everything is allowed to be. See
2652/// [`EntryListRow`] for what the projection deliberately omits and why.
2653///
2654/// `feed_ids = Some(&[])` means "no feeds in scope" and returns empty without
2655/// touching the database — distinct from `None`, which means "every feed this
2656/// DID subscribes to".
2657pub async fn list_entries(
2658 pool: &SqlitePool,
2659 did: &str,
2660 view: ListView,
2661 feed_ids: Option<&[i64]>,
2662 limit: i64,
2663 offset: i64,
2664) -> Result<Vec<EntryListRow>> {
2665 if feed_ids.is_some_and(<[i64]>::is_empty) || limit <= 0 {
2666 return Ok(Vec::new());
2667 }
2668 let (mut sql, n) = list_entries_sql(view, feed_ids);
2669 sql.push_str(&format!(
2670 " ORDER BY COALESCE(e.published, e.fetched_at) DESC, e.id DESC LIMIT ?{} OFFSET ?{}",
2671 n + 2,
2672 n + 3
2673 ));
2674 let q = sqlx::query_as::<_, EntryListRow>(sqlx::AssertSqlSafe(sql));
2675 let rows = bind_list_scope(q, did, feed_ids)
2676 .bind(limit)
2677 .bind(offset.max(0))
2678 .fetch_all(pool)
2679 .await
2680 .with_context(|| format!("list_entries({view:?}) failed for {did}"))?;
2681 Ok(rows)
2682}
2683
2684/// How many entries the same scope + view would return, unpaged. Used for the
2685/// "N entries" heading and to decide whether a next-page link is warranted —
2686/// both of which used to read `entries.len()` off a fully materialized list.
2687pub async fn count_entries_for_view(
2688 pool: &SqlitePool,
2689 did: &str,
2690 view: ListView,
2691 feed_ids: Option<&[i64]>,
2692) -> Result<i64> {
2693 if feed_ids.is_some_and(<[i64]>::is_empty) {
2694 return Ok(0);
2695 }
2696 let (sql, _) = list_query_sql(Projection::Count, view, feed_ids);
2697 // `query_as` over a 1-tuple keeps one binding helper for both shapes.
2698 let q = sqlx::query_as::<_, (i64,)>(sqlx::AssertSqlSafe(sql));
2699 let (n,) = bind_list_scope(q, did, feed_ids)
2700 .fetch_one(pool)
2701 .await
2702 .with_context(|| format!("count_entries_for_view({view:?}) failed for {did}"))?;
2703 Ok(n)
2704}
2705
2706/// The ordered entry ids for a scope + view — the same ordering [`list_entries`]
2707/// renders, used for the reader's prev/next links.
2708///
2709/// Ids only: this one genuinely spans the whole list rather than a page (prev/next
2710/// needs the reader's position in it), so it is the one query where row COUNT can
2711/// still be large. An id is 8 bytes against the 11.9 KB row this used to fetch,
2712/// and `limit` bounds it regardless. Past the limit, prev/next simply stops
2713/// finding neighbours — the article still opens.
2714pub async fn list_entry_ids(
2715 pool: &SqlitePool,
2716 did: &str,
2717 view: ListView,
2718 feed_ids: Option<&[i64]>,
2719 limit: i64,
2720) -> Result<Vec<i64>> {
2721 if feed_ids.is_some_and(<[i64]>::is_empty) || limit <= 0 {
2722 return Ok(Vec::new());
2723 }
2724 let (mut sql, n) = list_query_sql(Projection::Ids, view, feed_ids);
2725 sql.push_str(&format!(
2726 " ORDER BY COALESCE(e.published, e.fetched_at) DESC, e.id DESC LIMIT ?{}",
2727 n + 2
2728 ));
2729 let q = sqlx::query_as::<_, (i64,)>(sqlx::AssertSqlSafe(sql));
2730 let rows = bind_list_scope(q, did, feed_ids)
2731 .bind(limit)
2732 .fetch_all(pool)
2733 .await
2734 .with_context(|| format!("list_entry_ids({view:?}) failed for {did}"))?;
2735 Ok(rows.into_iter().map(|(id,)| id).collect())
2736}
2737
2738/// Unread counts per `feed_id` for a DID — the sidebar's per-feed badges.
2739///
2740/// Counted in SQL. The sidebar used to fetch every unread entry (bodies and all)
2741/// and count them in Rust, on every page with chrome, which is the single most
2742/// frequent instance of the projection problem [`EntryListRow`] describes.
2743pub async fn unread_counts_by_feed(
2744 pool: &SqlitePool,
2745 did: &str,
2746) -> Result<std::collections::HashMap<i64, i64>> {
2747 let (sql, _) = list_query_sql(Projection::FeedCounts, ListView::Unread, None);
2748 let rows =
2749 sqlx::query_as::<_, (i64, i64)>(sqlx::AssertSqlSafe(format!("{sql} GROUP BY e.feed_id")))
2750 .bind(did)
2751 .fetch_all(pool)
2752 .await
2753 .with_context(|| format!("unread_counts_by_feed failed for {did}"))?;
2754 Ok(rows.into_iter().collect())
2755}
2756
2757/// The `(url, guid)` identity pairs of every cached starred entry for a DID.
2758///
2759/// The starred view matches PDS saved records against these to decide which
2760/// records the cache can render itself. It must span the whole starred set, not
2761/// the visible page: a record that looks uncached gets an un-save button that
2762/// deletes the PDS RECORD rather than un-starring the entry, so narrowing this
2763/// set changes what a click destroys. Identity strings only — no bodies.
2764///
2765/// **Truncation is reported, not absorbed.** The `limit` is a memory backstop,
2766/// but hitting it violates the invariant above — and the first version had no
2767/// way to say so and no `ORDER BY`, so it silently returned an ARBITRARY subset
2768/// and every starred article outside it rendered with a record-destroying
2769/// button. `Truncated` lets the caller fail closed instead, and the ordering
2770/// makes the subset at least deterministic across renders rather than
2771/// whatever the query planner felt like returning.
2772pub enum StarredIdentities {
2773 /// The complete set for this DID.
2774 All(Vec<(Option<String>, String)>),
2775 /// `limit` was reached, so this is a partial set and MUST NOT be used to
2776 /// decide that a record is uncached.
2777 Truncated,
2778}
2779
2780pub async fn starred_identities(
2781 pool: &SqlitePool,
2782 did: &str,
2783 limit: i64,
2784) -> Result<StarredIdentities> {
2785 let (mut sql, n) = list_query_sql(Projection::StarredUrls, ListView::Starred, None);
2786 // One past the limit, so reaching it is distinguishable from landing on it
2787 // exactly. Ordered by id so the rows are stable; `url`/`guid` are not
2788 // guaranteed unique or non-NULL, and the id is both.
2789 //
2790 // The placeholder index comes from `list_query_sql` rather than being
2791 // hardcoded: it was `?2` only because this call passes `None` for the scope,
2792 // which is the kind of coupling that breaks silently when the shared builder
2793 // changes shape — as it just did.
2794 sql.push_str(&format!(" ORDER BY e.id LIMIT ?{}", n + 2));
2795 let rows = sqlx::query_as::<_, (Option<String>, String)>(sqlx::AssertSqlSafe(sql))
2796 .bind(did)
2797 .bind(limit.saturating_add(1))
2798 .fetch_all(pool)
2799 .await
2800 .with_context(|| format!("starred_identities failed for {did}"))?;
2801 if rows.len() as i64 > limit {
2802 return Ok(StarredIdentities::Truncated);
2803 }
2804 Ok(StarredIdentities::All(rows))
2805}
2806
2807/// Mark a single entry read/unread for a DID, upserting the per-DID state row
2808/// and stamping `updated_at`. Preserves any existing `starred` bit. Also
2809/// projects the change into the per-`(did, feed_url)` [`ReadCursor`] and marks
2810/// it `dirty` so the batched flusher pushes it to the PDS (see
2811/// `project_entry_into_cursor`).
2812///
2813/// AUTHORIZED per-DID: the upsert only touches an entry the caller subscribes
2814/// to (`sub_ref`). Returns `true` if a row was written, `false` if `did` does
2815/// not subscribe to the entry's feed (the web layer maps that to a 404 —
2816/// a non-subscriber can never mutate another user's state).
2817pub async fn mark_read(pool: &SqlitePool, did: &str, entry_id: i64, read: bool) -> Result<bool> {
2818 let now = now_rfc3339();
2819 let mut tx = pool.begin().await.context("begin mark_read tx")?;
2820 let res = sqlx::query(
2821 r#"
2822 INSERT INTO entry_state (did, entry_id, read, starred, updated_at)
2823 SELECT ?1, e.id, ?3, 0, ?4
2824 FROM entries e
2825 WHERE e.id = ?2
2826 AND EXISTS (
2827 SELECT 1 FROM sub_ref sr
2828 WHERE sr.did = ?1 AND sr.feed_id = e.feed_id
2829 )
2830 ON CONFLICT (did, entry_id) DO UPDATE SET
2831 read = excluded.read,
2832 updated_at = excluded.updated_at
2833 "#,
2834 )
2835 .bind(did)
2836 .bind(entry_id)
2837 .bind(read)
2838 .bind(&now)
2839 .execute(&mut *tx)
2840 .await
2841 .with_context(|| format!("mark_read failed for {did}/{entry_id}"))?;
2842
2843 if res.rows_affected() == 0 {
2844 // Not authorized (no `sub_ref`) — nothing written, no cursor to dirty.
2845 tx.rollback().await.ok();
2846 return Ok(false);
2847 }
2848
2849 // Project the read/unread into this feed's read cursor (dirty=1) so the
2850 // flusher syncs it to the PDS. Same tx as the state write so a crash can't
2851 // leave the two out of step.
2852 project_entry_into_cursor(&mut tx, did, entry_id, read, &now).await?;
2853
2854 tx.commit().await.context("commit mark_read tx")?;
2855 Ok(true)
2856}
2857
2858/// Star/unstar a single entry for a DID (upsert, preserving `read`).
2859///
2860/// AUTHORIZED per-DID like [`mark_read`]: only touches an entry the caller
2861/// subscribes to. Returns `true` if a row was written, `false` if `did` does
2862/// not subscribe (→ 404 at the web layer).
2863pub async fn mark_starred(
2864 pool: &SqlitePool,
2865 did: &str,
2866 entry_id: i64,
2867 starred: bool,
2868) -> Result<bool> {
2869 let res = sqlx::query(
2870 r#"
2871 INSERT INTO entry_state (did, entry_id, read, starred, updated_at)
2872 SELECT ?1, e.id, 0, ?3, ?4
2873 FROM entries e
2874 WHERE e.id = ?2
2875 AND EXISTS (
2876 SELECT 1 FROM sub_ref sr
2877 WHERE sr.did = ?1 AND sr.feed_id = e.feed_id
2878 )
2879 ON CONFLICT (did, entry_id) DO UPDATE SET
2880 starred = excluded.starred,
2881 updated_at = excluded.updated_at
2882 "#,
2883 )
2884 .bind(did)
2885 .bind(entry_id)
2886 .bind(starred)
2887 .bind(now_rfc3339())
2888 .execute(pool)
2889 .await
2890 .with_context(|| format!("mark_starred failed for {did}/{entry_id}"))?;
2891 Ok(res.rows_affected() > 0)
2892}
2893
2894/// Fold ids already covered by a high-water-mark into `read_through`, so the
2895/// exception set stops growing. Returns the new `read_through` when it advanced.
2896///
2897/// **What was wrong.** `read_through` was never COMPUTED — `project_entry_into_cursor`
2898/// only carried an existing value through, and it starts NULL, so in practice it
2899/// was always NULL. That left `read_ids` as the sole mechanism, growing one id
2900/// per article read, bounded only by `max_entries_per_feed` (2000) — while the
2901/// flusher caps the record at `ReadState::MAX_IDS` (1000) keeping the TAIL, with
2902/// no log line. Past 1000 read articles in one feed, the oldest read-state
2903/// silently stopped syncing, and those articles came back UNREAD in any other
2904/// atproto reader. The `cap` helper's own comment assumed "the exception sets
2905/// are expected to stay well under the cap in normal use"; against a 2000-entry
2906/// per-feed ceiling that does not hold.
2907///
2908/// **The rule.** `read_through` means "every entry at or before this time is
2909/// read". So it may advance only to a point with no unread entry at or before
2910/// it. That point is computed here as the newest entry timestamp STRICTLY OLDER
2911/// than the oldest unread entry — strictly, because entries can share a
2912/// timestamp, and a watermark equal to an unread entry's time would assert that
2913/// entry is read.
2914///
2915/// Once the watermark moves, every `read_ids` entry at or before it is
2916/// redundant and is dropped — that is the compaction. `unread_ids` is filtered
2917/// the same way; by construction nothing unread sits at or below the new
2918/// watermark, so it empties, but the filter is written rather than assumed so it
2919/// stays correct if that invariant ever shifts.
2920///
2921/// Timestamps compare lexicographically because every writer normalises to UTC
2922/// `...Z` at seconds precision (`feed::fmt_time`, `now_rfc3339`) — the same
2923/// assumption `poll_health` and the retention window already make.
2924pub async fn compact_cursor(
2925 pool: &SqlitePool,
2926 did: &str,
2927 feed_url: &str,
2928) -> Result<Option<String>> {
2929 let mut tx = pool.begin().await.context("begin compact_cursor tx")?;
2930 let (read_through, read_ids, unread_ids) = cursor_sets(&mut tx, did, feed_url).await?;
2931
2932 // The oldest entry on this feed that `did` has NOT read. `NULL` = nothing
2933 // unread, in which case the watermark can cover the whole feed.
2934 let oldest_unread: Option<String> = sqlx::query_scalar(
2935 r#"
2936 SELECT MIN(COALESCE(e.published, e.fetched_at))
2937 FROM entries e
2938 JOIN feeds f ON f.id = e.feed_id
2939 LEFT JOIN entry_state s ON s.entry_id = e.id AND s.did = ?1
2940 WHERE f.url = ?2 AND COALESCE(s.read, 0) = 0
2941 "#,
2942 )
2943 .bind(did)
2944 .bind(feed_url)
2945 .fetch_one(&mut *tx)
2946 .await
2947 .with_context(|| format!("compact_cursor: oldest unread for {did}/{feed_url}"))?;
2948
2949 let watermark: Option<String> = match &oldest_unread {
2950 Some(oldest) => sqlx::query_scalar(
2951 r#"
2952 SELECT MAX(COALESCE(e.published, e.fetched_at))
2953 FROM entries e JOIN feeds f ON f.id = e.feed_id
2954 WHERE f.url = ?1 AND COALESCE(e.published, e.fetched_at) < ?2
2955 "#,
2956 )
2957 .bind(feed_url)
2958 .bind(oldest)
2959 .fetch_one(&mut *tx)
2960 .await
2961 .with_context(|| format!("compact_cursor: watermark for {did}/{feed_url}"))?,
2962 None => sqlx::query_scalar(
2963 r#"
2964 SELECT MAX(COALESCE(e.published, e.fetched_at))
2965 FROM entries e JOIN feeds f ON f.id = e.feed_id
2966 WHERE f.url = ?1
2967 "#,
2968 )
2969 .bind(feed_url)
2970 .fetch_one(&mut *tx)
2971 .await
2972 .with_context(|| format!("compact_cursor: watermark for {did}/{feed_url}"))?,
2973 };
2974
2975 // Nothing to cover, or the watermark is already at least this far along.
2976 // Never move it BACKWARDS: that would re-assert articles as unread.
2977 let Some(watermark) = watermark else {
2978 return Ok(None);
2979 };
2980 if read_through
2981 .as_deref()
2982 .is_some_and(|rt| rt >= &watermark[..])
2983 {
2984 return Ok(None);
2985 }
2986
2987 let keep_above = ids_published_after(&mut tx, feed_url, &read_ids, &watermark).await?;
2988 let keep_unread =
2989 ids_published_at_or_before(&mut tx, feed_url, &unread_ids, &watermark).await?;
2990
2991 write_cursor_sets(
2992 &mut tx,
2993 did,
2994 feed_url,
2995 Some(&watermark),
2996 &keep_above,
2997 &keep_unread,
2998 &now_rfc3339(),
2999 )
3000 .await?;
3001 tx.commit().await.context("commit compact_cursor tx")?;
3002 Ok(Some(watermark))
3003}
3004
3005/// The subset of `ids` whose entries are published strictly AFTER `watermark`,
3006/// as the canonical JSON array-of-strings the cursor stores.
3007async fn ids_published_after(
3008 tx: &mut sqlx::Transaction<'_, sqlx::Sqlite>,
3009 feed_url: &str,
3010 ids: &str,
3011 watermark: &str,
3012) -> Result<String> {
3013 let live = ids_matching_watermark(tx, feed_url, watermark, true).await?;
3014 Ok(filter_id_set_to_live(ids, &live))
3015}
3016
3017/// The subset of `ids` whose entries are published at or BEFORE `watermark`.
3018async fn ids_published_at_or_before(
3019 tx: &mut sqlx::Transaction<'_, sqlx::Sqlite>,
3020 feed_url: &str,
3021 ids: &str,
3022 watermark: &str,
3023) -> Result<String> {
3024 let live = ids_matching_watermark(tx, feed_url, watermark, false).await?;
3025 Ok(filter_id_set_to_live(ids, &live))
3026}
3027
3028/// Entry ids on `feed_url` on one side of `watermark`. `after = true` selects
3029/// strictly newer; `false` selects at-or-older.
3030async fn ids_matching_watermark(
3031 tx: &mut sqlx::Transaction<'_, sqlx::Sqlite>,
3032 feed_url: &str,
3033 watermark: &str,
3034 after: bool,
3035) -> Result<std::collections::HashSet<i64>> {
3036 let sql = if after {
3037 "SELECT e.id FROM entries e JOIN feeds f ON f.id = e.feed_id \
3038 WHERE f.url = ?1 AND COALESCE(e.published, e.fetched_at) > ?2"
3039 } else {
3040 "SELECT e.id FROM entries e JOIN feeds f ON f.id = e.feed_id \
3041 WHERE f.url = ?1 AND COALESCE(e.published, e.fetched_at) <= ?2"
3042 };
3043 Ok(sqlx::query_scalar::<_, i64>(sql)
3044 .bind(feed_url)
3045 .bind(watermark)
3046 .fetch_all(&mut **tx)
3047 .await
3048 .context("compact_cursor: ids on one side of the watermark")?
3049 .into_iter()
3050 .collect())
3051}
3052
3053/// Clear `did`'s star on any cached entry matching `url` or `guid`, **ignoring
3054/// the subscription projection**. Returns the number of `entry_state` rows
3055/// changed.
3056///
3057/// This closes a desync between the two places a star lives. The starred view
3058/// matches PDS saved records against cached entries through `sub_ref`, so an
3059/// entry that is cached AND starred in a feed the reader has since UNSUBSCRIBED
3060/// from does not match: it renders as an uncached row whose button is
3061/// `POST /saved/{rkey}/delete`. That deletes the PDS record and used to leave
3062/// `entry_state.starred = 1` behind — invisible, because the starred list is
3063/// `sub_ref`-scoped too, until the reader resubscribes and the star reappears
3064/// with no record backing it.
3065///
3066/// **Why omitting `sub_ref` is safe here, when it is the per-DID isolation hook
3067/// everywhere else.** Every row this can touch is keyed by `did` and this writes
3068/// only `starred = 0`. The worst a caller can do with it is clear one of their
3069/// OWN stars — which is what they just asked for. The predicate that matters for
3070/// isolation is the `did` in the `WHERE`, and it is not optional.
3071///
3072/// Matching on `url` OR `guid` mirrors how the view decides a record is already
3073/// cached, so the removal path and the render path agree on what "the same
3074/// article" means.
3075pub async fn clear_star_by_identity(
3076 pool: &SqlitePool,
3077 did: &str,
3078 url: Option<&str>,
3079 guid: Option<&str>,
3080) -> Result<u64> {
3081 // Neither identifier present: nothing to match on. Running the statement
3082 // would compare NULL to NULL and match nothing, but returning early says so.
3083 if url.is_none_or(str::is_empty) && guid.is_none_or(str::is_empty) {
3084 return Ok(0);
3085 }
3086 let res = sqlx::query(
3087 r#"
3088 UPDATE entry_state
3089 SET starred = 0, updated_at = ?4
3090 WHERE did = ?1
3091 AND starred = 1
3092 AND entry_id IN (
3093 SELECT id FROM entries
3094 WHERE (?2 IS NOT NULL AND url = ?2)
3095 OR (?3 IS NOT NULL AND guid = ?3)
3096 )
3097 "#,
3098 )
3099 .bind(did)
3100 .bind(url.filter(|u| !u.is_empty()))
3101 .bind(guid.filter(|g| !g.is_empty()))
3102 .bind(now_rfc3339())
3103 .execute(pool)
3104 .await
3105 .with_context(|| format!("clear_star_by_identity failed for {did}"))?;
3106 Ok(res.rows_affected())
3107}
3108
3109/// Mark every entry of a feed read (or unread) for a DID in one statement —
3110/// backs the "mark-all-read (per feed)" action. Also projects the change into
3111/// the feed's per-DID [`ReadCursor`] (dirty=1) so the batched flusher syncs the
3112/// new read-state to the PDS.
3113pub async fn mark_feed_read(pool: &SqlitePool, did: &str, feed_id: i64, read: bool) -> Result<u64> {
3114 let now = now_rfc3339();
3115 let mut tx = pool.begin().await.context("begin mark_feed_read tx")?;
3116 let res = sqlx::query(
3117 r#"
3118 INSERT INTO entry_state (did, entry_id, read, starred, updated_at)
3119 SELECT ?1, e.id, ?2, 0, ?3 FROM entries e
3120 WHERE e.feed_id = ?4
3121 AND EXISTS (
3122 SELECT 1 FROM sub_ref sr
3123 WHERE sr.did = ?1 AND sr.feed_id = e.feed_id
3124 )
3125 ON CONFLICT (did, entry_id) DO UPDATE SET
3126 read = excluded.read,
3127 updated_at = excluded.updated_at
3128 "#,
3129 )
3130 .bind(did)
3131 .bind(read)
3132 .bind(&now)
3133 .bind(feed_id)
3134 .execute(&mut *tx)
3135 .await
3136 .with_context(|| format!("mark_feed_read failed for {did}/feed {feed_id}"))?;
3137
3138 if res.rows_affected() > 0 {
3139 // Project every affected entry into this feed's read cursor. `feed_id`
3140 // maps to exactly one feed URL, so this is a single per-feed cursor —
3141 // batched, not per-article. Only runs when the caller was authorized
3142 // (some rows changed), so an unsubscribed feed leaves no cursor behind.
3143 project_feed_into_cursor(&mut tx, did, feed_id, read, &now).await?;
3144 }
3145
3146 tx.commit().await.context("commit mark_feed_read tx")?;
3147 Ok(res.rows_affected())
3148}
3149
3150// ---------------------------------------------------------------------------
3151// Read-cursor projection (wires the local read/unread mutation into the
3152// PDS-bound `read_cursor`, so the batched flusher actually pushes read-state)
3153// ---------------------------------------------------------------------------
3154
3155/// Add or remove an entry id from a JSON id-array string, returning the new JSON.
3156/// Membership is set-like (no duplicates) and order-stable (append on add). A
3157/// malformed input is treated as empty so a cosmetic parse issue never blocks a
3158/// projection.
3159fn json_id_set_toggle(raw: &str, id: i64, present: bool) -> String {
3160 let mut ids: Vec<i64> = serde_json::from_str::<Vec<serde_json::Value>>(raw)
3161 .ok()
3162 .map(|vals| {
3163 vals.into_iter()
3164 .filter_map(|v| match v {
3165 serde_json::Value::Number(n) => n.as_i64(),
3166 serde_json::Value::String(s) => s.parse::<i64>().ok(),
3167 _ => None,
3168 })
3169 .collect()
3170 })
3171 .unwrap_or_default();
3172 if present {
3173 if !ids.contains(&id) {
3174 ids.push(id);
3175 }
3176 } else {
3177 ids.retain(|&x| x != id);
3178 }
3179 // Serialize as a JSON array of strings (the shape the flusher / lexicon
3180 // expect — `community.lexicon.rss.readState.readIds` is a string array).
3181 let as_strings: Vec<String> = ids.iter().map(|i| i.to_string()).collect();
3182 serde_json::to_string(&as_strings).unwrap_or_else(|_| "[]".to_string())
3183}
3184
3185/// The feed URL owning `feed_id`, if the row exists (cursors are keyed by URL,
3186/// not feed id — they mirror the PDS-side `readState.feedUrl`).
3187async fn feed_url_for_id_tx(
3188 tx: &mut sqlx::Transaction<'_, sqlx::Sqlite>,
3189 feed_id: i64,
3190) -> Result<Option<String>> {
3191 let url: Option<String> = sqlx::query_scalar("SELECT url FROM feeds WHERE id = ?1")
3192 .bind(feed_id)
3193 .fetch_optional(&mut **tx)
3194 .await
3195 .with_context(|| format!("feed_url_for_id_tx failed for feed {feed_id}"))?;
3196 Ok(url)
3197}
3198
3199/// Fetch the (read_through, read_ids, unread_ids) of an existing cursor, or the
3200/// empty defaults if there is none yet.
3201async fn cursor_sets(
3202 tx: &mut sqlx::Transaction<'_, sqlx::Sqlite>,
3203 did: &str,
3204 feed_url: &str,
3205) -> Result<(Option<String>, String, String)> {
3206 let row = sqlx::query(
3207 "SELECT read_through, read_ids, unread_ids FROM read_cursor \
3208 WHERE did = ?1 AND feed_url = ?2",
3209 )
3210 .bind(did)
3211 .bind(feed_url)
3212 .fetch_optional(&mut **tx)
3213 .await
3214 .with_context(|| format!("cursor_sets failed for {did}/{feed_url}"))?;
3215 Ok(match row {
3216 Some(r) => (
3217 r.get::<Option<String>, _>("read_through"),
3218 r.get::<String, _>("read_ids"),
3219 r.get::<String, _>("unread_ids"),
3220 ),
3221 None => (None, "[]".to_string(), "[]".to_string()),
3222 })
3223}
3224
3225/// Upsert the cursor row for `(did, feed_url)` with the given exception sets,
3226/// stamping `updated_at` and marking it `dirty` so `dirty_cursors` returns it.
3227async fn write_cursor_sets(
3228 tx: &mut sqlx::Transaction<'_, sqlx::Sqlite>,
3229 did: &str,
3230 feed_url: &str,
3231 read_through: Option<&str>,
3232 read_ids: &str,
3233 unread_ids: &str,
3234 now: &str,
3235) -> Result<()> {
3236 sqlx::query(
3237 r#"
3238 INSERT INTO read_cursor
3239 (did, feed_url, read_through, read_ids, unread_ids, dirty, updated_at)
3240 VALUES (?1, ?2, ?3, ?4, ?5, 1, ?6)
3241 ON CONFLICT (did, feed_url) DO UPDATE SET
3242 read_through = excluded.read_through,
3243 read_ids = excluded.read_ids,
3244 unread_ids = excluded.unread_ids,
3245 dirty = 1,
3246 updated_at = excluded.updated_at
3247 "#,
3248 )
3249 .bind(did)
3250 .bind(feed_url)
3251 .bind(read_through)
3252 .bind(read_ids)
3253 .bind(unread_ids)
3254 .bind(now)
3255 .execute(&mut **tx)
3256 .await
3257 .with_context(|| format!("write_cursor_sets failed for {did}/{feed_url}"))?;
3258 Ok(())
3259}
3260
3261/// Project a single entry's read/unread flip into its feed's read cursor.
3262///
3263/// The cursor mirrors `community.lexicon.rss.readState`: a `read_through`
3264/// high-water-mark plus two bounded exception sets. A per-article flip is
3265/// recorded in those sets (`read_ids` when read, `unread_ids` when unread), the
3266/// opposite set is cleared of the id, and the cursor is stamped + marked dirty.
3267/// This keeps the write batched by touching only the ONE per-feed cursor. (Note:
3268/// there is no compaction step yet that folds covered ids back into
3269/// `read_through`; the exception sets are expected to stay well under the cap.)
3270async fn project_entry_into_cursor(
3271 tx: &mut sqlx::Transaction<'_, sqlx::Sqlite>,
3272 did: &str,
3273 entry_id: i64,
3274 read: bool,
3275 now: &str,
3276) -> Result<()> {
3277 // The entry's feed id → feed URL (the cursor key).
3278 let feed_id: Option<i64> = sqlx::query_scalar("SELECT feed_id FROM entries WHERE id = ?1")
3279 .bind(entry_id)
3280 .fetch_optional(&mut **tx)
3281 .await
3282 .with_context(|| format!("project_entry_into_cursor: feed_id for entry {entry_id}"))?;
3283 let feed_id = match feed_id {
3284 Some(f) => f,
3285 None => return Ok(()), // entry vanished mid-tx; nothing to project
3286 };
3287 let feed_url = match feed_url_for_id_tx(tx, feed_id).await? {
3288 Some(u) => u,
3289 None => return Ok(()),
3290 };
3291
3292 let (read_through, read_ids, unread_ids) = cursor_sets(tx, did, &feed_url).await?;
3293 // read=true: id joins read_ids, leaves unread_ids. read=false: the inverse.
3294 let read_ids = json_id_set_toggle(&read_ids, entry_id, read);
3295 let unread_ids = json_id_set_toggle(&unread_ids, entry_id, !read);
3296 write_cursor_sets(
3297 tx,
3298 did,
3299 &feed_url,
3300 read_through.as_deref(),
3301 &read_ids,
3302 &unread_ids,
3303 now,
3304 )
3305 .await
3306}
3307
3308/// Project a mark-all-feed-read/unread into that feed's single read cursor.
3309///
3310/// Every entry the caller subscribes to on `feed_id` is folded into the cursor
3311/// in one write: on mark-all-READ each id joins `read_ids` (and leaves
3312/// `unread_ids`); on mark-all-UNREAD the inverse. Still ONE per-feed cursor row
3313/// (batched), stamped + dirtied for the flusher.
3314async fn project_feed_into_cursor(
3315 tx: &mut sqlx::Transaction<'_, sqlx::Sqlite>,
3316 did: &str,
3317 feed_id: i64,
3318 read: bool,
3319 now: &str,
3320) -> Result<()> {
3321 let feed_url = match feed_url_for_id_tx(tx, feed_id).await? {
3322 Some(u) => u,
3323 None => return Ok(()),
3324 };
3325
3326 // The entry ids on this feed the caller is authorized for (subscribes to).
3327 let ids: Vec<i64> = sqlx::query_scalar(
3328 r#"
3329 SELECT e.id FROM entries e
3330 WHERE e.feed_id = ?2
3331 AND EXISTS (
3332 SELECT 1 FROM sub_ref sr
3333 WHERE sr.did = ?1 AND sr.feed_id = e.feed_id
3334 )
3335 "#,
3336 )
3337 .bind(did)
3338 .bind(feed_id)
3339 .fetch_all(&mut **tx)
3340 .await
3341 .with_context(|| format!("project_feed_into_cursor: entry ids for {did}/feed {feed_id}"))?;
3342
3343 let (read_through, mut read_ids, mut unread_ids) = cursor_sets(tx, did, &feed_url).await?;
3344 for id in ids {
3345 read_ids = json_id_set_toggle(&read_ids, id, read);
3346 unread_ids = json_id_set_toggle(&unread_ids, id, !read);
3347 }
3348 write_cursor_sets(
3349 tx,
3350 did,
3351 &feed_url,
3352 read_through.as_deref(),
3353 &read_ids,
3354 &unread_ids,
3355 now,
3356 )
3357 .await
3358}
3359
3360/// Test-only unbounded convenience wrappers over [`list_entries`].
3361///
3362/// Production code passes an explicit `limit`, because that is the whole point
3363/// of the change these replaced. Fixtures hold a handful of rows and asserting
3364/// on "the whole list" is what the tests actually mean, so they get a helper
3365/// with a stated ceiling instead of each spelling one out — and the ceiling is
3366/// high enough that a test hitting it is a broken fixture, not a truncation.
3367#[cfg(test)]
3368mod test_helpers {
3369 use super::*;
3370
3371 /// Far above any fixture; a test that reaches it has a bug of its own.
3372 const FIXTURE_MAX: i64 = 10_000;
3373
3374 pub(crate) async fn entries_for_feed(
3375 pool: &SqlitePool,
3376 did: &str,
3377 feed_id: i64,
3378 ) -> Result<Vec<EntryListRow>> {
3379 list_entries(pool, did, ListView::All, Some(&[feed_id]), FIXTURE_MAX, 0).await
3380 }
3381
3382 pub(crate) async fn get_unread_for_did(
3383 pool: &SqlitePool,
3384 did: &str,
3385 ) -> Result<Vec<EntryListRow>> {
3386 list_entries(pool, did, ListView::Unread, None, FIXTURE_MAX, 0).await
3387 }
3388
3389 pub(crate) async fn get_starred_for_did(
3390 pool: &SqlitePool,
3391 did: &str,
3392 ) -> Result<Vec<EntryListRow>> {
3393 list_entries(pool, did, ListView::Starred, None, FIXTURE_MAX, 0).await
3394 }
3395}
3396
3397#[cfg(test)]
3398pub(crate) use test_helpers::{entries_for_feed, get_starred_for_did, get_unread_for_did};
3399
3400/// Insert or update a per-`(did, feed_url)` read cursor, stamping `updated_at`.
3401/// The write path for local mark-read updates (and the seam a login-time PDS
3402/// merge would use, once that is wired).
3403pub async fn upsert_cursor(pool: &SqlitePool, cursor: &ReadCursor) -> Result<()> {
3404 sqlx::query(
3405 r#"
3406 INSERT INTO read_cursor
3407 (did, feed_url, read_through, read_ids, unread_ids, dirty, updated_at)
3408 VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7)
3409 ON CONFLICT (did, feed_url) DO UPDATE SET
3410 read_through = excluded.read_through,
3411 read_ids = excluded.read_ids,
3412 unread_ids = excluded.unread_ids,
3413 dirty = excluded.dirty,
3414 updated_at = excluded.updated_at
3415 "#,
3416 )
3417 .bind(&cursor.did)
3418 .bind(&cursor.feed_url)
3419 .bind(&cursor.read_through)
3420 .bind(&cursor.read_ids)
3421 .bind(&cursor.unread_ids)
3422 .bind(cursor.dirty)
3423 .bind(&cursor.updated_at)
3424 .execute(pool)
3425 .await
3426 .with_context(|| {
3427 format!(
3428 "upsert_cursor failed for {}/{}",
3429 cursor.did, cursor.feed_url
3430 )
3431 })?;
3432 Ok(())
3433}
3434
3435/// Fetch a single read cursor, if present.
3436pub async fn get_cursor(
3437 pool: &SqlitePool,
3438 did: &str,
3439 feed_url: &str,
3440) -> Result<Option<ReadCursor>> {
3441 let cursor = sqlx::query_as::<_, ReadCursor>(
3442 "SELECT * FROM read_cursor WHERE did = ?1 AND feed_url = ?2",
3443 )
3444 .bind(did)
3445 .bind(feed_url)
3446 .fetch_optional(pool)
3447 .await
3448 .context("get_cursor failed")?;
3449 Ok(cursor)
3450}
3451
3452/// The flusher's hot query: every cursor with `dirty = 1` for a DID — the ones
3453/// whose read-state changed since the last batched PDS flush.
3454/// How many DIDs hold read-state that cannot currently be flushed: dirty
3455/// cursors with no OAuth session to send them with.
3456///
3457/// **The visible form of the parked state (#117).** The flusher deliberately
3458/// stops warning about these every round, and quiet-and-invisible would be a
3459/// worse bug than the noisy loop it replaces — so the count is surfaced on
3460/// `/admin/metrics`. A non-zero number is not itself an alarm: it is the normal
3461/// state of anyone signed out with unsynced reads. A number that only ever
3462/// grows is the thing to look at.
3463///
3464/// Rust-backend shaped: it asks about `oauth_session`, which is the Rust
3465/// backend's store. On the sidecar backend it over-reports, since those
3466/// sessions live in the sidecar's own database. Prod runs `rust` and the
3467/// sidecar is removed by #18.
3468pub async fn parked_readstate_dids(pool: &SqlitePool) -> Result<i64> {
3469 let row: (i64,) = sqlx::query_as(
3470 r#"
3471 SELECT COUNT(DISTINCT rc.did)
3472 FROM read_cursor rc
3473 WHERE rc.dirty = 1
3474 AND NOT EXISTS (SELECT 1 FROM oauth_session s WHERE s.sub = rc.did)
3475 "#,
3476 )
3477 .fetch_one(pool)
3478 .await
3479 .context("counting parked read-state DIDs")?;
3480 Ok(row.0)
3481}
3482
3483pub async fn dirty_cursors(pool: &SqlitePool, did: &str) -> Result<Vec<ReadCursor>> {
3484 let cursors =
3485 sqlx::query_as::<_, ReadCursor>("SELECT * FROM read_cursor WHERE did = ?1 AND dirty = 1")
3486 .bind(did)
3487 .fetch_all(pool)
3488 .await
3489 .with_context(|| format!("dirty_cursors failed for {did}"))?;
3490 Ok(cursors)
3491}
3492
3493// ---------------------------------------------------------------------------
3494// Network observations (the adoption probe's projection)
3495// ---------------------------------------------------------------------------
3496
3497/// Record one relay's observation, keyed by `(key, source)` so each relay's
3498/// number is kept separately (non-archival relays legitimately disagree).
3499///
3500/// An upsert: the table is bounded forever at (metrics × relays) rows — two
3501/// today — so this can never grow the DB. It must stay an upsert and never
3502/// become a per-DID insert.
3503///
3504/// **A truncated observation never lowers a stored count.** A truncated walk
3505/// saw only part of the network, so a smaller number is evidence about the
3506/// *walk*, not about adoption. Without the guard, one slow run that managed a
3507/// single 500-repo page would overwrite a complete 2 000 and drag the published
3508/// "at least N" down — and because `latest_network_stat` takes the max across
3509/// sources, two relays behind the same operator degrade together, so `/about`
3510/// would sit at the lower figure until a full walk succeeded again. A COMPLETE
3511/// observation always wins, even when smaller (repos genuinely can disappear);
3512/// a truncated one may only ever raise the floor — and an EQUAL count raises
3513/// nothing, so it is rejected too. That is why the guard reads `<=` and not
3514/// `<`: the strict form let a truncated walk that merely matched the stored
3515/// number rewrite the row and flip `truncated` on, degrading "2 000" to "at
3516/// least 2 000" with no change in adoption.
3517pub async fn record_network_stat(pool: &SqlitePool, stat: &NetworkStat) -> Result<()> {
3518 sqlx::query(
3519 r#"
3520 INSERT INTO network_stat (key, source, value, truncated, observed_at)
3521 VALUES (?1, ?2, ?3, ?4, ?5)
3522 ON CONFLICT (key, source) DO UPDATE SET
3523 value = excluded.value,
3524 truncated = excluded.truncated,
3525 observed_at = excluded.observed_at
3526 WHERE NOT (excluded.truncated = 1 AND excluded.value <= network_stat.value)
3527 "#,
3528 )
3529 .bind(&stat.key)
3530 .bind(&stat.source)
3531 .bind(stat.value)
3532 .bind(stat.truncated)
3533 .bind(&stat.observed_at)
3534 .execute(pool)
3535 .await
3536 .with_context(|| {
3537 format!(
3538 "record_network_stat failed for {}/{}",
3539 stat.key, stat.source
3540 )
3541 })?;
3542 Ok(())
3543}
3544
3545/// The highest observation for `key` across every relay — the number to surface
3546/// (`design/NETWORK-SPEC.md` §4.1: relays disagree; show the max). `None` when no
3547/// probe has ever succeeded.
3548pub async fn latest_network_stat(pool: &SqlitePool, key: &str) -> Result<Option<NetworkStat>> {
3549 let stat = sqlx::query_as::<_, NetworkStat>(
3550 "SELECT key, source, value, truncated, observed_at FROM network_stat \
3551 WHERE key = ?1 ORDER BY value DESC, observed_at DESC LIMIT 1",
3552 )
3553 .bind(key)
3554 .fetch_optional(pool)
3555 .await
3556 .with_context(|| format!("latest_network_stat failed for {key}"))?;
3557 Ok(stat)
3558}
3559
3560/// Mark a cursor's PDS `readState` record as CREATED after the flush that first
3561/// created it, so subsequent flushes emit an `update` instead of another
3562/// `create`. Idempotent; a no-op if the row is gone.
3563pub async fn mark_cursor_pds_created(pool: &SqlitePool, did: &str, feed_url: &str) -> Result<()> {
3564 sqlx::query("UPDATE read_cursor SET pds_created = 1 WHERE did = ?1 AND feed_url = ?2")
3565 .bind(did)
3566 .bind(feed_url)
3567 .execute(pool)
3568 .await
3569 .with_context(|| format!("mark_cursor_pds_created failed for {did}/{feed_url}"))?;
3570 Ok(())
3571}
3572
3573/// Clear the `dirty` flag on a cursor after a successful PDS flush — but ONLY if
3574/// the row still carries the exact `flushed_updated_at` snapshot we flushed.
3575///
3576/// The flusher reads a cursor, sends it to the PDS (a network round-trip), then
3577/// clears `dirty`. A concurrent [`upsert_cursor`] (a fresh mark-read) can land
3578/// DURING that in-flight write, bumping `updated_at` and re-setting `dirty = 1`
3579/// for reads that were NOT in the flushed snapshot. An unconditional
3580/// `SET dirty = 0` would silently drop those reads. Guarding on the snapshot's
3581/// `updated_at` makes this a compare-and-swap: if `updated_at` changed under us,
3582/// zero rows update, the row stays dirty, and it re-flushes next round.
3583pub async fn clear_cursor_dirty(
3584 pool: &SqlitePool,
3585 did: &str,
3586 feed_url: &str,
3587 flushed_updated_at: &str,
3588) -> Result<()> {
3589 sqlx::query(
3590 "UPDATE read_cursor SET dirty = 0 \
3591 WHERE did = ?1 AND feed_url = ?2 AND updated_at = ?3",
3592 )
3593 .bind(did)
3594 .bind(feed_url)
3595 .bind(flushed_updated_at)
3596 .execute(pool)
3597 .await
3598 .context("clear_cursor_dirty failed")?;
3599 Ok(())
3600}
3601
3602// ---------------------------------------------------------------------------
3603// Closed-beta invite gate (beta_access + invite_codes)
3604// ---------------------------------------------------------------------------
3605//
3606// Ported in SHAPE from a prior Go beta-gate (RedeemCode / CreateInviteCode /
3607// code_gen) but deliberately trimmed for FeatherReader's before-public
3608// experiment: NO viral invite-budget tree, NO generation cap, NO waitlist /
3609// invite-request table, and SQLite instead of Mongo. A code is minted by an
3610// existing member (or admin), and redeeming it grants a seat while seats remain
3611// under the configured cap.
3612
3613/// Unix-epoch seconds for "now" — the integer time base for the beta tables.
3614pub(crate) fn now_unix() -> i64 {
3615 chrono::Utc::now().timestamp()
3616}
3617
3618/// The invite-code alphabet: uppercase letters + digits with the
3619/// visually-ambiguous glyphs removed (`I`, `O`, `0`, `1`) so a code read aloud
3620/// or copied by hand is unambiguous.
3621const CODE_ALPHABET: &[u8] = b"ABCDEFGHJKLMNPQRSTUVWXYZ23456789";
3622
3623/// Human-facing prefix so a FeatherReader invite code is recognisable at a
3624/// glance.
3625const CODE_PREFIX: &str = "FEATHER-";
3626
3627/// Number of random characters after the prefix.
3628const CODE_BODY_LEN: usize = 8;
3629
3630/// Generate a random, unguessable invite code of the form `FEATHER-XXXXXXXX`.
3631///
3632/// Draws from the OS CSPRNG (`getrandom`) and maps each byte onto
3633/// `CODE_ALPHABET` via rejection sampling so the alphabet distribution is
3634/// uniform (no modulo bias). Infallible in practice; a `getrandom` failure
3635/// (no entropy source) propagates as an error rather than a weak code.
3636pub fn generate_invite_code() -> Result<String> {
3637 let n = CODE_ALPHABET.len() as u16; // 31
3638 // Largest multiple of `n` that fits in a byte; bytes at or above it are
3639 // rejected so every accepted byte maps uniformly onto the alphabet.
3640 let limit = 256 / n * n; // 256 - (256 % n)
3641 let mut out = String::with_capacity(CODE_PREFIX.len() + CODE_BODY_LEN);
3642 out.push_str(CODE_PREFIX);
3643 let mut got = 0;
3644 let mut buf = [0u8; 1];
3645 while got < CODE_BODY_LEN {
3646 getrandom::fill(&mut buf).context("getrandom failed while minting invite code")?;
3647 let b = buf[0] as u16;
3648 if b < limit {
3649 out.push(CODE_ALPHABET[(b % n) as usize] as char);
3650 got += 1;
3651 }
3652 }
3653 Ok(out)
3654}
3655
3656/// Whether a DID currently holds a beta seat.
3657pub async fn has_beta_access(pool: &SqlitePool, did: &str) -> Result<bool> {
3658 let row = sqlx::query("SELECT 1 FROM beta_access WHERE did = ?1")
3659 .bind(did)
3660 .fetch_optional(pool)
3661 .await
3662 .with_context(|| format!("has_beta_access failed for {did}"))?;
3663 Ok(row.is_some())
3664}
3665
3666/// Count the beta seats currently granted — the numerator checked against the
3667/// configured cap on redeem.
3668pub async fn count_beta_access(pool: &SqlitePool) -> Result<i64> {
3669 let row = sqlx::query("SELECT COUNT(*) AS n FROM beta_access")
3670 .fetch_one(pool)
3671 .await
3672 .context("count_beta_access failed")?;
3673 Ok(row.get::<i64, _>("n"))
3674}
3675
3676/// Count `active`, unexpired invite codes — the outstanding-but-unredeemed seats
3677/// a bot has already promised. Added to [`count_beta_access`] this is the "seats
3678/// committed" figure the bot mint path (`POST /bot/claims`) checks against the
3679/// cap, so it doesn't over-promise more claims than seats remain (the redeem-time
3680/// cap in [`redeem_code`] is the hard backstop; this avoids telling a follower
3681/// "you're in" for a seat that will be full by the time they claim it).
3682pub async fn count_active_codes(pool: &SqlitePool) -> Result<i64> {
3683 let now = now_unix();
3684 let row = sqlx::query(
3685 "SELECT COUNT(*) AS n FROM invite_codes WHERE status = 'active' AND expires_at >= ?1",
3686 )
3687 .bind(now)
3688 .fetch_one(pool)
3689 .await
3690 .context("count_active_codes failed")?;
3691 Ok(row.get::<i64, _>("n"))
3692}
3693
3694/// Grant a beta seat directly (admin / seed path — no code consumed). Idempotent
3695/// on `did` (re-granting updates the row rather than erroring).
3696pub async fn grant_access(
3697 pool: &SqlitePool,
3698 did: &str,
3699 handle: Option<&str>,
3700 granted_by: &str,
3701 invite_code_used: Option<&str>,
3702) -> Result<()> {
3703 sqlx::query(
3704 r#"
3705 INSERT INTO beta_access (did, handle, granted_by, granted_at, invite_code_used)
3706 VALUES (?1, ?2, ?3, ?4, ?5)
3707 ON CONFLICT (did) DO UPDATE SET
3708 handle = COALESCE(excluded.handle, beta_access.handle),
3709 granted_by = excluded.granted_by,
3710 invite_code_used = COALESCE(excluded.invite_code_used, beta_access.invite_code_used)
3711 "#,
3712 )
3713 .bind(did)
3714 .bind(handle)
3715 .bind(granted_by)
3716 .bind(now_unix())
3717 .bind(invite_code_used)
3718 .execute(pool)
3719 .await
3720 .with_context(|| format!("grant_access failed for {did}"))?;
3721 Ok(())
3722}
3723
3724/// Mint a new `active` invite code owned by `creator_did`, expiring `ttl_secs`
3725/// from now. Returns the generated code string. The browser/admin path leaves the
3726/// bot idempotency key (`intended_did`) NULL; see [`mint_code_for_did`] for the
3727/// bot path that records the target follower.
3728pub async fn mint_code(pool: &SqlitePool, creator_did: &str, ttl_secs: i64) -> Result<String> {
3729 mint_code_inner(pool, creator_did, ttl_secs, None).await
3730}
3731
3732/// Like [`mint_code`] but records the follower `intended_did` the code is minted
3733/// FOR, so a later `POST /bot/claims` for the same DID can return the SAME code
3734/// (see [`find_active_code_for_did`]) rather than minting a duplicate. This is the
3735/// app-side idempotency backstop that survives a bot-host state loss.
3736pub async fn mint_code_for_did(
3737 pool: &SqlitePool,
3738 creator_did: &str,
3739 ttl_secs: i64,
3740 intended_did: &str,
3741) -> Result<String> {
3742 mint_code_inner(pool, creator_did, ttl_secs, Some(intended_did)).await
3743}
3744
3745async fn mint_code_inner(
3746 pool: &SqlitePool,
3747 creator_did: &str,
3748 ttl_secs: i64,
3749 intended_did: Option<&str>,
3750) -> Result<String> {
3751 let code = generate_invite_code()?;
3752 let now = now_unix();
3753 let expires_at = now.saturating_add(ttl_secs.max(0));
3754 sqlx::query(
3755 r#"
3756 INSERT INTO invite_codes
3757 (code, creator_did, status, invitee_did, intended_did, created_at, expires_at, redeemed_at)
3758 VALUES (?1, ?2, 'active', NULL, ?3, ?4, ?5, NULL)
3759 "#,
3760 )
3761 .bind(&code)
3762 .bind(creator_did)
3763 .bind(intended_did)
3764 .bind(now)
3765 .bind(expires_at)
3766 .execute(pool)
3767 .await
3768 .with_context(|| format!("mint_code failed for creator {creator_did}"))?;
3769 Ok(code)
3770}
3771
3772/// Does this error chain represent the partial-unique-index conflict raised when
3773/// a SECOND active claim is minted for a DID that already has one
3774/// (`idx_invite_codes_intended_active`)? The web layer uses this to recover from a
3775/// lost mint race (S4): on a conflict it re-reads the winner's code instead of
3776/// 500-ing. Matches on the sqlx `Database` error's UNIQUE-constraint code (SQLite
3777/// 2067 / primary 19) AND the offending COLUMN in the message
3778/// (`invite_codes.intended_did` — SQLite names the column(s), not the index), so an
3779/// unrelated constraint violation (e.g. the `code` PRIMARY KEY) is NOT swallowed.
3780pub fn is_intended_active_conflict(err: &anyhow::Error) -> bool {
3781 for cause in err.chain() {
3782 if let Some(sqlx::Error::Database(db)) = cause.downcast_ref::<sqlx::Error>() {
3783 let msg = db.message();
3784 // SQLite reports UNIQUE violations with (primary) code 19 /
3785 // (extended) 2067; the message names the offending column(s), e.g.
3786 // "UNIQUE constraint failed: invite_codes.intended_did".
3787 let is_unique = db.code().as_deref() == Some("2067")
3788 || db.code().as_deref() == Some("19")
3789 || msg.contains("UNIQUE constraint failed");
3790 // Scope to the intended_did index specifically. Only that index and the
3791 // `code` PRIMARY KEY can raise a UNIQUE error here; the partial unique
3792 // index is the only one over `intended_did`, so the column reference
3793 // uniquely identifies it.
3794 if is_unique && msg.contains("invite_codes.intended_did") {
3795 return true;
3796 }
3797 }
3798 }
3799 false
3800}
3801
3802/// The `code` of an outstanding (`active`, unexpired) invite minted FOR the
3803/// follower `intended_did`, if one exists — the app-side idempotency lookup for
3804/// `POST /bot/claims`. `Some(code)` means "return this existing code, do NOT mint
3805/// a second"; `None` means "no live code for this DID — mint one".
3806///
3807/// S3 — this lookup ONLY returns `active`, UNEXPIRED codes; once a code passes
3808/// `expires_at` (or `expire_old_codes` flips it to `expired`) this returns `None`,
3809/// so the next `POST /bot/claims` MINTS A FRESH code for the DID. There is no
3810/// in-place "refresh" of an expired code (the partial-unique index only constrains
3811/// `active` rows, so a fresh mint after expiry is allowed). The bot then re-posts:
3812/// its record rkey is deterministic per DID, so the existing skeet is UPDATED in
3813/// place with the new claim URL (see the bot's `reconcile_stale_record`, S1) rather
3814/// than a second skeet being posted. NOTE: a bot-`delivered` follower whose link
3815/// expired UNCLAIMED is only re-minted if the bot re-processes that DID (a re-seen
3816/// follow, a `waitlisted` retry, or a bot-store reset); manual recovery is to clear
3817/// the bot's `handled` row for that DID so the next cycle re-mints + re-posts.
3818/// If several live codes somehow exist (a race), the soonest-expiring is returned.
3819pub async fn find_active_code_for_did(
3820 pool: &SqlitePool,
3821 intended_did: &str,
3822) -> Result<Option<String>> {
3823 let now = now_unix();
3824 let row = sqlx::query(
3825 "SELECT code FROM invite_codes
3826 WHERE intended_did = ?1 AND status = 'active' AND expires_at >= ?2
3827 ORDER BY expires_at ASC
3828 LIMIT 1",
3829 )
3830 .bind(intended_did)
3831 .bind(now)
3832 .fetch_optional(pool)
3833 .await
3834 .with_context(|| format!("find_active_code_for_did failed for {intended_did}"))?;
3835 Ok(row.map(|r| r.get::<String, _>("code")))
3836}
3837
3838/// Atomically redeem an invite code for `did`, granting a beta seat.
3839///
3840/// Runs entirely in one transaction so the capacity check and the seat grant
3841/// cannot race (two redeems can't both slip past a `cap - 1` count). Steps:
3842/// 1. verify the code exists, is `active`, and is not past `expires_at`;
3843/// 2. verify the current seat count is `< cap`;
3844/// 3. flip the code `active`→`redeemed` (stamping `invitee_did` + `redeemed_at`);
3845/// 4. insert the `beta_access` row.
3846///
3847/// On a policy failure returns the matching [`RedeemError`] (the tx rolls back);
3848/// a real SQLite error propagates as the outer [`anyhow::Error`].
3849pub async fn redeem_code(
3850 pool: &SqlitePool,
3851 code: &str,
3852 did: &str,
3853 handle: Option<&str>,
3854 cap: i64,
3855) -> Result<std::result::Result<(), RedeemError>> {
3856 let now = now_unix();
3857 let mut tx = pool.begin().await.context("begin redeem_code tx")?;
3858
3859 // Take the write lock at the START of the transaction. sqlx issues a plain
3860 // deferred BEGIN, so without this the capacity SELECT below runs under a read
3861 // snapshot: two concurrent redeems could both pass the gate, and the loser's
3862 // later UPDATE would fail with SQLITE_BUSY_SNAPSHOT (which busy_timeout does
3863 // NOT retry) — an opaque error instead of a clean CapacityFull. A leading
3864 // no-op write against the target row acquires the RESERVED lock immediately
3865 // (SQLite locks on any write statement, even one matching zero rows), so the
3866 // second redeem blocks on the first, then reads the post-commit seat count
3867 // and returns CapacityFull. (The cap already held via snapshot isolation;
3868 // this upgrades the failure mode from a hard error to the right one.)
3869 sqlx::query("UPDATE invite_codes SET status = status WHERE code = ?1")
3870 .bind(code)
3871 .execute(&mut *tx)
3872 .await
3873 .context("redeem_code: acquire write lock")?;
3874
3875 // 1. Look the code up.
3876 let row =
3877 sqlx::query("SELECT status, expires_at, intended_did FROM invite_codes WHERE code = ?1")
3878 .bind(code)
3879 .fetch_optional(&mut *tx)
3880 .await
3881 .context("redeem_code: lookup")?;
3882 let row = match row {
3883 Some(r) => r,
3884 None => return Ok(Err(RedeemError::NotFound)),
3885 };
3886 let status: String = row.get("status");
3887 let expires_at: i64 = row.get("expires_at");
3888 let intended_did: Option<String> = row.get("intended_did");
3889
3890 // DID-binding gate (blocker B2). A bot-minted claim link is posted PUBLICLY
3891 // with a non-confidential token, so anyone who sees a follower's reply could
3892 // redeem it with a throwaway account — defeating the follow-gate, the daily
3893 // sybil budget, and the rate limit. When the code was minted FOR a specific
3894 // follower (`intended_did IS NOT NULL`), only that DID may redeem it; anyone
3895 // else gets a `NotFound` (indistinguishable from a bad code — no oracle).
3896 // Codes with a NULL `intended_did` (admin/browser-minted) stay open, as
3897 // before — those are meant to be sharable.
3898 if let Some(bound) = intended_did.as_deref() {
3899 if bound != did {
3900 return Ok(Err(RedeemError::NotFound));
3901 }
3902 }
3903
3904 // Status gate: only an `active` code is redeemable. Anything already
3905 // redeemed/revoked is "already redeemed" from the redeemer's view; an
3906 // `expired` status (or a past expiry) is "expired".
3907 if status == "expired" || now > expires_at {
3908 return Ok(Err(RedeemError::Expired));
3909 }
3910 if status != "active" {
3911 return Ok(Err(RedeemError::AlreadyRedeemed));
3912 }
3913
3914 // 2. Capacity gate (inside the tx so it can't race a concurrent redeem).
3915 let count: i64 = sqlx::query("SELECT COUNT(*) AS n FROM beta_access")
3916 .fetch_one(&mut *tx)
3917 .await
3918 .context("redeem_code: count")?
3919 .get("n");
3920 if count >= cap {
3921 return Ok(Err(RedeemError::CapacityFull));
3922 }
3923
3924 // 3. Flip the code active→redeemed. The `status = 'active'` guard in the
3925 // WHERE makes this a compare-and-swap: if a concurrent tx already flipped it
3926 // (despite the read above), zero rows change and we treat it as redeemed.
3927 let flipped = sqlx::query(
3928 r#"
3929 UPDATE invite_codes
3930 SET status = 'redeemed', invitee_did = ?2, redeemed_at = ?3
3931 WHERE code = ?1 AND status = 'active'
3932 "#,
3933 )
3934 .bind(code)
3935 .bind(did)
3936 .bind(now)
3937 .execute(&mut *tx)
3938 .await
3939 .context("redeem_code: flip")?;
3940 if flipped.rows_affected() == 0 {
3941 return Ok(Err(RedeemError::AlreadyRedeemed));
3942 }
3943
3944 // 4. Grant the seat.
3945 sqlx::query(
3946 r#"
3947 INSERT INTO beta_access (did, handle, granted_by, granted_at, invite_code_used)
3948 VALUES (?1, ?2, ?3, ?4, ?5)
3949 ON CONFLICT (did) DO UPDATE SET
3950 handle = COALESCE(excluded.handle, beta_access.handle),
3951 invite_code_used = excluded.invite_code_used
3952 "#,
3953 )
3954 .bind(did)
3955 .bind(handle)
3956 // granted_by is the code's creator; look it up in-tx to keep provenance.
3957 .bind(
3958 sqlx::query("SELECT creator_did FROM invite_codes WHERE code = ?1")
3959 .bind(code)
3960 .fetch_one(&mut *tx)
3961 .await
3962 .context("redeem_code: creator lookup")?
3963 .get::<String, _>("creator_did"),
3964 )
3965 .bind(now)
3966 .bind(code)
3967 .execute(&mut *tx)
3968 .await
3969 .context("redeem_code: grant")?;
3970
3971 tx.commit().await.context("commit redeem_code tx")?;
3972 Ok(Ok(()))
3973}
3974
3975/// Sweep: flip every `active` code whose `expires_at` is in the past to
3976/// `expired`. Returns the number of codes expired. Called periodically by the
3977/// scheduler.
3978pub async fn expire_old_codes(pool: &SqlitePool) -> Result<u64> {
3979 let now = now_unix();
3980 let res = sqlx::query(
3981 "UPDATE invite_codes SET status = 'expired' WHERE status = 'active' AND expires_at < ?1",
3982 )
3983 .bind(now)
3984 .execute(pool)
3985 .await
3986 .context("expire_old_codes failed")?;
3987 Ok(res.rows_affected())
3988}
3989
3990/// Seed the admin-bootstrap DIDs: for each, insert a `beta_access` row
3991/// (`granted_by = 'admin'`) if one does not already exist. Idempotent — an
3992/// existing seat is left untouched. Returns how many new seats were created.
3993pub async fn ensure_seed(pool: &SqlitePool, dids: &[String]) -> Result<u64> {
3994 let mut tx = pool.begin().await.context("begin ensure_seed tx")?;
3995 let now = now_unix();
3996 let mut created = 0u64;
3997 for did in dids {
3998 let res = sqlx::query(
3999 r#"
4000 INSERT INTO beta_access (did, handle, granted_by, granted_at, invite_code_used)
4001 VALUES (?1, NULL, 'admin', ?2, NULL)
4002 ON CONFLICT (did) DO NOTHING
4003 "#,
4004 )
4005 .bind(did)
4006 .bind(now)
4007 .execute(&mut *tx)
4008 .await
4009 .with_context(|| format!("ensure_seed insert failed for {did}"))?;
4010 created += res.rows_affected();
4011 }
4012 tx.commit().await.context("commit ensure_seed tx")?;
4013 Ok(created)
4014}
4015
4016/// The row counts purged by [`purge_did_data`], for a confirmable success
4017/// message and for assertions in tests.
4018#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
4019pub struct PurgeCounts {
4020 /// `entry_state` rows removed (per-DID read/star flags).
4021 pub entry_state: u64,
4022 /// `read_cursor` rows removed (per-DID per-feed read cursors).
4023 pub read_cursor: u64,
4024 /// `sub_ref` rows removed (the DID's subscription projection).
4025 pub sub_ref: u64,
4026 /// `beta_access` rows removed (the DID's closed-beta seat: 0 or 1).
4027 pub beta_access: u64,
4028 /// `invite_codes` rows removed (codes this DID *created*).
4029 pub invite_codes: u64,
4030 /// `invite_codes` rows *scrubbed* (the code this DID *redeemed* to join —
4031 /// its `invitee_did` back-reference cleared to NULL, row kept).
4032 pub invitee_scrubbed: u64,
4033 /// `beta_access` rows *scrubbed* (seats this DID *granted* to others — the
4034 /// `granted_by` back-reference redacted to a sentinel, row kept).
4035 pub granted_by_scrubbed: u64,
4036}
4037
4038impl PurgeCounts {
4039 /// Total rows removed across every per-DID table. (Scrub counts are tracked
4040 /// separately — those rows belong to *other* DIDs and are redacted, not
4041 /// deleted — so they are excluded from the delete total.)
4042 pub fn total(&self) -> u64 {
4043 self.entry_state + self.read_cursor + self.sub_ref + self.beta_access + self.invite_codes
4044 }
4045}
4046
4047/// Sentinel written into `beta_access.granted_by` when the granting DID deletes
4048/// its data: the column is `NOT NULL`, so we redact rather than NULL it. Keeps
4049/// the grantee's seat valid while removing the departed DID's back-reference.
4050pub const REDACTED_DID: &str = "__redacted__";
4051
4052/// Delete **all** local rows owned by `did` in a single transaction: the
4053/// per-DID read/star state (`entry_state`), per-feed read cursors
4054/// (`read_cursor`), the subscription projection (`sub_ref`), the closed-beta
4055/// seat (`beta_access`), and any invite codes this DID *created*
4056/// (`invite_codes`). The shared `feeds`/`entries` cache is intentionally left
4057/// intact — it is deduped and not owned by any single DID.
4058///
4059/// This is the local half of "delete my data": the caller pairs it with a
4060/// sidecar `POST /internal/revoke` so the OAuth tokens + sidecar session rows
4061/// are dropped too. Idempotent — deleting a DID with no rows returns all-zero
4062/// counts.
4063pub async fn purge_did_data(pool: &SqlitePool, did: &str) -> Result<PurgeCounts> {
4064 let mut tx = pool.begin().await.context("begin purge_did_data tx")?;
4065
4066 let entry_state = sqlx::query("DELETE FROM entry_state WHERE did = ?1")
4067 .bind(did)
4068 .execute(&mut *tx)
4069 .await
4070 .with_context(|| format!("purge entry_state for {did}"))?
4071 .rows_affected();
4072
4073 let read_cursor = sqlx::query("DELETE FROM read_cursor WHERE did = ?1")
4074 .bind(did)
4075 .execute(&mut *tx)
4076 .await
4077 .with_context(|| format!("purge read_cursor for {did}"))?
4078 .rows_affected();
4079
4080 let sub_ref = sqlx::query("DELETE FROM sub_ref WHERE did = ?1")
4081 .bind(did)
4082 .execute(&mut *tx)
4083 .await
4084 .with_context(|| format!("purge sub_ref for {did}"))?
4085 .rows_affected();
4086
4087 let beta_access = sqlx::query("DELETE FROM beta_access WHERE did = ?1")
4088 .bind(did)
4089 .execute(&mut *tx)
4090 .await
4091 .with_context(|| format!("purge beta_access for {did}"))?
4092 .rows_affected();
4093
4094 let invite_codes = sqlx::query("DELETE FROM invite_codes WHERE creator_did = ?1")
4095 .bind(did)
4096 .execute(&mut *tx)
4097 .await
4098 .with_context(|| format!("purge invite_codes for {did}"))?
4099 .rows_affected();
4100
4101 // Scrub the DID's back-references from rows that belong to OTHER DIDs so no
4102 // per-DID residue survives the delete:
4103 // * the invite code this DID *redeemed* to join lives on the inviter's
4104 // row (`invitee_did`) — NULL it out (column is nullable).
4105 // * seats this DID *granted* to others carry `granted_by = <this did>` —
4106 // redact to a sentinel (column is NOT NULL) so the grantee keeps access
4107 // without retaining the departed DID.
4108 let invitee_scrubbed =
4109 sqlx::query("UPDATE invite_codes SET invitee_did = NULL WHERE invitee_did = ?1")
4110 .bind(did)
4111 .execute(&mut *tx)
4112 .await
4113 .with_context(|| format!("scrub invitee_did for {did}"))?
4114 .rows_affected();
4115
4116 // A departing DID may also be the TARGET of an outstanding bot claim
4117 // (`intended_did`, minted for them before they joined/left) — NULL it so no
4118 // per-DID residue survives. We ALSO expire the orphaned code in the same tx:
4119 // once `intended_did` is NULLed, an `active` row would otherwise keep counting
4120 // against the daily mint cap for its full 14-day TTL (and a re-follow would
4121 // double-count it), so `expired` it now. `redeemed`/already-`expired` rows are
4122 // untouched (the WHERE only matches `active`). (Cheap nit — purge orphan.)
4123 sqlx::query(
4124 "UPDATE invite_codes \
4125 SET intended_did = NULL, \
4126 status = CASE WHEN status = 'active' THEN 'expired' ELSE status END \
4127 WHERE intended_did = ?1",
4128 )
4129 .bind(did)
4130 .execute(&mut *tx)
4131 .await
4132 .with_context(|| format!("scrub intended_did for {did}"))?;
4133
4134 let granted_by_scrubbed =
4135 sqlx::query("UPDATE beta_access SET granted_by = ?2 WHERE granted_by = ?1")
4136 .bind(did)
4137 .bind(REDACTED_DID)
4138 .execute(&mut *tx)
4139 .await
4140 .with_context(|| format!("scrub granted_by for {did}"))?
4141 .rows_affected();
4142
4143 tx.commit().await.context("commit purge_did_data tx")?;
4144
4145 Ok(PurgeCounts {
4146 entry_state,
4147 read_cursor,
4148 sub_ref,
4149 beta_access,
4150 invite_codes,
4151 invitee_scrubbed,
4152 granted_by_scrubbed,
4153 })
4154}
4155
4156/// Aggregate poll health, for the public stats page.
4157///
4158/// **Deliberately aggregate-only.** No user counts, no error rates, no per-feed
4159/// detail: this is published to anyone, and a reader does not need to know how
4160/// many people use an instance or which feeds are failing. What it does answer
4161/// is the only question the page exists for — is the poller keeping up?
4162#[derive(Debug, Clone, PartialEq, Eq)]
4163pub struct PollHealth {
4164 /// Distinct feeds the poller is responsible for.
4165 pub feeds_tracked: i64,
4166 /// How many were polled within the last hour.
4167 pub polled_last_hour: i64,
4168 /// Feeds whose `next_poll` has passed — the backlog. A healthy instance
4169 /// clears this every tick; a growing number is the signal that the poller
4170 /// cannot keep up with the feed count.
4171 pub overdue: i64,
4172 /// Seconds since the most recent poll of any feed. `None` before the first.
4173 pub last_poll_secs_ago: Option<i64>,
4174 /// Seconds since the LEAST recently polled feed was polled — the worst
4175 /// staleness any reader is currently seeing.
4176 ///
4177 /// `None` when any feed has NEVER been polled, because that is a worse
4178 /// staleness than any finite age and reporting the finite one would make
4179 /// the page read healthiest exactly when it is least healthy.
4180 pub oldest_poll_secs_ago: Option<i64>,
4181 /// How many feeds have never been polled at all.
4182 pub never_polled: i64,
4183 /// Feeds currently in error backoff (`consecutive_errors > 0`).
4184 ///
4185 /// One of the two states that stop feeds updating, and previously visible
4186 /// nowhere: `consecutive_errors` was written by `bump_feed_errors` and read
4187 /// by nothing outside the backoff calculation — no page, no endpoint. Worse,
4188 /// a feed in backoff is NOT counted in `overdue`, because backoff is applied
4189 /// by pushing `next_poll` forward. So the one number a reader might have
4190 /// checked moved the wrong way: a feed failing every fetch made `overdue`
4191 /// look BETTER.
4192 pub in_backoff: i64,
4193 /// Of those, how many have reached `BADLY_BROKEN_ERRORS` consecutive
4194 /// failures — retried 2h40m apart rather than every 5 minutes.
4195 ///
4196 /// Not "will not recover on their own": the backoff ceiling is 24h at ten
4197 /// errors, and any of these recovers on its next successful poll. See
4198 /// `BADLY_BROKEN_ERRORS`.
4199 pub badly_broken: i64,
4200 /// Failing feeds grouped by **cause**, descending, as
4201 /// `(kind, count)` — `fetch`, `status`, `body`, `parse`.
4202 ///
4203 /// **Counts, never identities.** `/stats` is public and states that it
4204 /// reports machines rather than people: no per-feed detail, never which feed
4205 /// and never whose. A cause histogram keeps that promise and still answers
4206 /// the question `badly_broken` could not — whether sixty feeds are failing
4207 /// for sixty reasons or for one. Had this existed, #159 would have read
4208 /// `fetch: 60` on a page anyone could load, instead of costing a production
4209 /// investigation.
4210 pub failure_kinds: Vec<(String, i64)>,
4211}
4212
4213/// `consecutive_errors` at or above which a feed counts as `badly_broken`.
4214///
4215/// Chosen to mean "this is not a transient blip": `feed::backoff_for` climbs
4216/// exponentially, so by this many consecutive failures a feed is being retried
4217/// **2h40m apart** — `backoff_for(6)`.
4218///
4219/// **Not "at or near the ceiling", and not "effectively dead".** `BACKOFF_MAX`
4220/// is 24h and is first reached at *ten* errors, so a feed at this threshold is
4221/// still retried around nine times a day and recovers on its own the moment the
4222/// cause clears. Three doc comments claimed otherwise, and the claim was
4223/// load-bearing in the wrong direction.
4224///
4225/// **It says nothing about whose fault the failure is, and used to claim it
4226/// did.** This comment and the matching `/stats` copy read "almost certainly
4227/// gone rather than flaky" until 2026-09-20, when #159 found that 60-odd feeds
4228/// sat here because `guarded_get` was reading every `304 Not Modified` as a
4229/// malformed redirect. The publishers were live; the reader was broken. That
4230/// assertion is what stopped anyone looking, which is why `last_error_kind`
4231/// now exists — the row can answer the question the count never could.
4232const BADLY_BROKEN_ERRORS: i64 = 6;
4233
4234/// Compute [`PollHealth`] as of `now` (RFC3339, seconds precision — the same
4235/// format the scheduler writes, so the comparisons are lexicographic).
4236pub async fn poll_health(pool: &SqlitePool, now: &str, hour_ago: &str) -> Result<PollHealth> {
4237 // **Only what the poller sees.** `due_feeds` skips `at://` rows, so nothing
4238 // ever advances their `next_poll` or sets `last_polled`; counted here they
4239 // read as overdue and never-polled forever and force "oldest poll" to
4240 // `never` — unsupported shown as broken, on a public page, permanently.
4241 // The same predicate as the scheduler's, so the two cannot disagree.
4242 let aggregate = format!(
4243 r#"
4244 SELECT
4245 COUNT(*),
4246 COALESCE(SUM(CASE WHEN last_polled IS NOT NULL AND last_polled >= ?2 THEN 1 ELSE 0 END), 0),
4247 COALESCE(SUM(CASE WHEN next_poll IS NULL OR next_poll <= ?1 THEN 1 ELSE 0 END), 0),
4248 MAX(last_polled),
4249 -- NULL-AWARE. `MIN` skips NULLs, so an instance where most feeds
4250 -- had NEVER been polled reported the freshest of the few that had —
4251 -- the figure read healthiest in the most degraded state, which is
4252 -- the opposite of what a health page is for. A never-polled feed IS
4253 -- the worst staleness, so it wins outright.
4254 CASE WHEN SUM(CASE WHEN last_polled IS NULL THEN 1 ELSE 0 END) > 0
4255 THEN NULL ELSE MIN(last_polled) END,
4256 SUM(CASE WHEN last_polled IS NULL THEN 1 ELSE 0 END),
4257 COALESCE(SUM(CASE WHEN consecutive_errors > 0 THEN 1 ELSE 0 END), 0),
4258 COALESCE(SUM(CASE WHEN consecutive_errors >= ?3 THEN 1 ELSE 0 END), 0)
4259 FROM feeds
4260 WHERE kind IN ({POLLABLE_KINDS_SQL})
4261 "#
4262 );
4263 #[allow(clippy::type_complexity)]
4264 let row: (i64, i64, i64, Option<String>, Option<String>, i64, i64, i64) =
4265 sqlx::query_as(sqlx::AssertSqlSafe(aggregate))
4266 .bind(now)
4267 .bind(hour_ago)
4268 .bind(BADLY_BROKEN_ERRORS)
4269 .fetch_one(pool)
4270 .await
4271 .context("computing poll health")?;
4272
4273 // A second, tiny query rather than a join: the histogram groups rows the
4274 // aggregate above collapses, and one statement doing both would make the
4275 // counts above harder to read than the extra round trip is worth.
4276 //
4277 // **Every failing feed lands in a bucket, so this sums to `in_backoff`.**
4278 //
4279 // A row that predates the column is failing with no recorded cause, and it
4280 // must not be attributed to some other feed's reason — but it must not
4281 // vanish either. Filtering them out made the breakdown silently disagree
4282 // with the `Failing` figure beside it: on a migrated database that is EVERY
4283 // currently-failing feed, so the page would have read "70 failing" next to
4284 // "3 fetch" with 67 unexplained and no indication a remainder existed.
4285 //
4286 // `unknown` is a deliberate bucket rather than an omission. It cannot
4287 // collide with a real kind — `FailureKind::as_str` never returns it, and
4288 // `FailureKind::parse("unknown")` is `None`.
4289 let histogram = format!(
4290 r#"
4291 -- **`failure_kind`, not `kind`.** Aliasing this `kind` collided with
4292 -- the `feeds.kind` column added for the poller: SQLite resolved
4293 -- `GROUP BY kind` to the table column, so every failing feed collapsed
4294 -- into ONE bucket labelled from an arbitrary row — a public page
4295 -- reporting "10 fetch" for ten unrelated causes. Caught by
4296 -- `an_unrecognised_failure_kind_folds_into_unknown`.
4297 SELECT COALESCE(last_error_kind, 'unknown') AS failure_kind, COUNT(*) AS n
4298 FROM feeds
4299 WHERE consecutive_errors > 0 AND kind IN ({POLLABLE_KINDS_SQL})
4300 GROUP BY failure_kind
4301 ORDER BY n DESC, failure_kind ASC
4302 "#
4303 );
4304 let kinds: Vec<(String, i64)> = sqlx::query_as(sqlx::AssertSqlSafe(histogram))
4305 .fetch_all(pool)
4306 .await
4307 .context("computing the failure-cause histogram")?;
4308
4309 // **Close the vocabulary where it is READ.** `FailureKind::parse` promised
4310 // that a kind from a newer build would not be attributed to a cause this
4311 // one recognises — but nothing called it, so the raw column reached the
4312 // public template and an unrecognised string rendered as its own bucket.
4313 // Fold anything `parse` rejects into `unknown`, then re-aggregate and
4314 // re-order, so the histogram only ever shows the four kinds this build
4315 // knows plus the one honest bucket for what it does not.
4316 let mut folded: std::collections::BTreeMap<String, i64> = std::collections::BTreeMap::new();
4317 for (kind, n) in kinds {
4318 let key = if kind == "unknown" || crate::feed::FailureKind::parse(&kind).is_some() {
4319 kind
4320 } else {
4321 "unknown".to_string()
4322 };
4323 *folded.entry(key).or_insert(0) += n;
4324 }
4325 let mut kinds: Vec<(String, i64)> = folded.into_iter().collect();
4326 kinds.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(&b.0)));
4327
4328 Ok(PollHealth {
4329 feeds_tracked: row.0,
4330 polled_last_hour: row.1,
4331 overdue: row.2,
4332 last_poll_secs_ago: secs_between(row.3.as_deref(), now),
4333 oldest_poll_secs_ago: secs_between(row.4.as_deref(), now),
4334 never_polled: row.5,
4335 in_backoff: row.6,
4336 badly_broken: row.7,
4337 failure_kinds: kinds,
4338 })
4339}
4340
4341/// Whole seconds from `then` to `now`, or `None` if `then` is absent or
4342/// unparseable. Never negative: a clock skew that puts a poll in the future
4343/// reads as "just now" rather than as a negative age.
4344fn secs_between(then: Option<&str>, now: &str) -> Option<i64> {
4345 let then = chrono::DateTime::parse_from_rfc3339(then?).ok()?;
4346 let now = chrono::DateTime::parse_from_rfc3339(now).ok()?;
4347 Some((now - then).num_seconds().max(0))
4348}
4349
4350#[cfg(test)]
4351mod tests {
4352 use super::*;
4353
4354 const PUB_A: &str = "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3laa";
4355
4356 /// Step 2 of the 0.4.0 plan: a stored publication is pollable.
4357 #[tokio::test]
4358 async fn a_due_publication_is_handed_to_the_poller() -> Result<()> {
4359 let pool = init_url("sqlite::memory:").await?;
4360 upsert_feed(
4361 &pool,
4362 &NewFeed {
4363 url: PUB_A.into(),
4364 ..Default::default()
4365 },
4366 )
4367 .await?;
4368 let due = due_feeds(&pool, "2999-01-01T00:00:00Z", 50).await?;
4369 assert!(
4370 due.iter().any(|f| f.url == PUB_A),
4371 "a publication row is not handed to the poller"
4372 );
4373 Ok(())
4374 }
4375
4376 #[tokio::test]
4377 async fn admitting_publications_staggers_their_first_poll() -> Result<()> {
4378 let pool = init_url("sqlite::memory:").await?;
4379 for i in 0..19 {
4380 let url =
4381 format!("at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3l{i:02}");
4382 upsert_feed(
4383 &pool,
4384 &NewFeed {
4385 url,
4386 ..Default::default()
4387 },
4388 )
4389 .await?;
4390 }
4391 // Already polled, and an RSS row: neither is touched.
4392 upsert_feed(
4393 &pool,
4394 &NewFeed {
4395 url: "https://rss.example/feed.xml".into(),
4396 ..Default::default()
4397 },
4398 )
4399 .await?;
4400 let n = stagger_unscheduled(
4401 &pool,
4402 crate::feed::FeedKind::Publication,
4403 std::time::Duration::from_secs(3600),
4404 )
4405 .await?;
4406 assert_eq!(n, 19, "not every unscheduled publication was scheduled");
4407 let slots: Vec<String> = sqlx::query_scalar(
4408 "SELECT next_poll FROM feeds WHERE kind = 'publication' ORDER BY next_poll",
4409 )
4410 .fetch_all(&pool)
4411 .await?;
4412 let distinct: std::collections::BTreeSet<_> = slots.iter().collect();
4413 assert_eq!(distinct.len(), 19, "publications share slots: {slots:?}");
4414 let rss: Option<String> = sqlx::query_scalar(
4415 "SELECT next_poll FROM feeds WHERE url = 'https://rss.example/feed.xml'",
4416 )
4417 .fetch_one(&pool)
4418 .await?;
4419 assert_eq!(rss, None, "an RSS row was rescheduled");
4420 let again = stagger_unscheduled(
4421 &pool,
4422 crate::feed::FeedKind::Publication,
4423 std::time::Duration::from_secs(3600),
4424 )
4425 .await?;
4426 assert_eq!(
4427 again, 0,
4428 "a second boot re-staggered rows that already had a slot"
4429 );
4430 Ok(())
4431 }
4432
4433 /// The point of the stagger: an overdue RSS feed is not starved by a block
4434 /// of newly admitted rows.
4435 #[tokio::test]
4436 async fn admitted_rows_do_not_outrank_an_overdue_rss_feed() -> Result<()> {
4437 let pool = init_url("sqlite::memory:").await?;
4438 for i in 0..5 {
4439 let url =
4440 format!("at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3l{i:02}");
4441 upsert_feed(
4442 &pool,
4443 &NewFeed {
4444 url,
4445 ..Default::default()
4446 },
4447 )
4448 .await?;
4449 }
4450 upsert_feed(
4451 &pool,
4452 &NewFeed {
4453 url: "https://overdue.example/feed.xml".into(),
4454 next_poll: Some("2000-01-01T00:00:00Z".into()),
4455 ..Default::default()
4456 },
4457 )
4458 .await?;
4459 stagger_unscheduled(
4460 &pool,
4461 crate::feed::FeedKind::Publication,
4462 std::time::Duration::from_secs(3600),
4463 )
4464 .await?;
4465 let now = chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
4466 let first = due_feeds(&pool, &now, 1).await?;
4467 assert_eq!(first[0].url, "https://overdue.example/feed.xml");
4468 Ok(())
4469 }
4470
4471 /// **A re-poll refreshes `published`; it never refreshes `fetched_at`.**
4472 ///
4473 /// The asymmetry is the whole reason a date must be stable. `published`
4474 /// comes back from the publisher on every poll, so a value the mapper
4475 /// recomputes — "now", say — is rewritten every hour and the row can never
4476 /// age. `fetched_at` is written once, at first insert, so an entry stored
4477 /// with no date is effectively dated when we first saw it, and that date
4478 /// does hold still. Both the per-feed cap and the retention sweep order on
4479 /// `COALESCE(published, fetched_at)`, so which of the two a row lands in
4480 /// decides whether it can ever be evicted or swept.
4481 #[tokio::test]
4482 async fn a_repoll_refreshes_published_but_never_fetched_at() -> Result<()> {
4483 let pool = init_url("sqlite::memory:").await?;
4484 let feed_id = upsert_feed(
4485 &pool,
4486 &NewFeed {
4487 url: "https://example.com/f.xml".to_string(),
4488 ..Default::default()
4489 },
4490 )
4491 .await?;
4492 let seen = |at: &str| {
4493 vec![NewEntry {
4494 guid: "g".to_string(),
4495 published: Some(at.to_string()),
4496 fetched_at: Some(at.to_string()),
4497 ..Default::default()
4498 }]
4499 };
4500 insert_entries(&pool, feed_id, &seen("2026-01-01T00:00:00Z"), 0).await?;
4501 insert_entries(&pool, feed_id, &seen("2026-09-20T00:00:00Z"), 0).await?;
4502
4503 let (published, fetched_at): (Option<String>, String) =
4504 sqlx::query_as("SELECT published, fetched_at FROM entries WHERE guid = 'g'")
4505 .fetch_one(&pool)
4506 .await?;
4507 assert_eq!(
4508 published.as_deref(),
4509 Some("2026-09-20T00:00:00Z"),
4510 "the second poll's date did not replace the first"
4511 );
4512 assert_eq!(
4513 fetched_at, "2026-01-01T00:00:00Z",
4514 "fetched_at moved, so an undated entry would never age either"
4515 );
4516 Ok(())
4517 }
4518
4519 /// A partial upsert must not erase the conditional-GET validators.
4520 ///
4521 /// `set_next_poll` supplies only `url` + `next_poll` and runs after EVERY
4522 /// poll of EVERY feed. While `upsert_feed` assigned etag/last_modified
4523 /// unconditionally, that call wrote both back to NULL, so `If-None-Match`
4524 /// was never sent, `304` was unreachable, and every feed was re-downloaded
4525 /// and re-parsed in full on every cycle. Nothing failed; it was invisible.
4526 #[tokio::test]
4527 async fn validators_survive_a_partial_upsert() -> Result<()> {
4528 let pool = init_url("sqlite::memory:").await?;
4529 let url = "https://example.com/feed.xml";
4530
4531 upsert_feed(
4532 &pool,
4533 &NewFeed {
4534 url: url.to_string(),
4535 etag: Some("\"abc123\"".to_string()),
4536 last_modified: Some("Wed, 01 Jan 2026 00:00:00 GMT".to_string()),
4537 ..Default::default()
4538 },
4539 )
4540 .await?;
4541
4542 // Exactly what `scheduler::set_next_poll` sends.
4543 upsert_feed(
4544 &pool,
4545 &NewFeed {
4546 url: url.to_string(),
4547 next_poll: Some("2026-07-12T00:00:00Z".to_string()),
4548 ..Default::default()
4549 },
4550 )
4551 .await?;
4552
4553 let feed = get_feed_by_url(&pool, url).await?.expect("feed");
4554 assert_eq!(
4555 feed.etag.as_deref(),
4556 Some("\"abc123\""),
4557 "a partial upsert erased the ETag, disabling conditional GET"
4558 );
4559 assert_eq!(
4560 feed.last_modified.as_deref(),
4561 Some("Wed, 01 Jan 2026 00:00:00 GMT"),
4562 "a partial upsert erased Last-Modified"
4563 );
4564 assert_eq!(feed.next_poll.as_deref(), Some("2026-07-12T00:00:00Z"));
4565 Ok(())
4566 }
4567
4568 /// A hard ceiling that is not strictly older than the window is IGNORED.
4569 ///
4570 /// `hard_days.max(days)` made `0` — the obvious "off" value, and the
4571 /// documented disable value for `RETENTION_DAYS` — collapse the ceiling onto
4572 /// the soft window, where the delete spares nothing. The starred and unread
4573 /// rows the window exists to protect were purged at `retention_days`.
4574 #[tokio::test]
4575 async fn a_ceiling_inside_the_window_is_ignored_not_applied() -> Result<()> {
4576 for hard in [0_i64, 1, 7, 14] {
4577 let pool = init_url("sqlite::memory:").await?;
4578 let feed_id = upsert_feed(
4579 &pool,
4580 &NewFeed {
4581 url: "https://example.com/f.xml".to_string(),
4582 ..Default::default()
4583 },
4584 )
4585 .await?;
4586 let old = (chrono::Utc::now() - chrono::Duration::days(30))
4587 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
4588 insert_entries(
4589 &pool,
4590 feed_id,
4591 &[
4592 NewEntry {
4593 guid: "starred-30d".to_string(),
4594 published: Some(old.clone()),
4595 ..Default::default()
4596 },
4597 NewEntry {
4598 guid: "unread-30d".to_string(),
4599 published: Some(old.clone()),
4600 ..Default::default()
4601 },
4602 ],
4603 0,
4604 )
4605 .await?;
4606 // Both need an explicit `entry_state` row: sparing keys off a
4607 // DELIBERATE mark, and an entry with no row at all is unclaimed
4608 // cache that the window is supposed to evict.
4609 sqlx::query(
4610 "INSERT INTO entry_state (did, entry_id, read, starred, updated_at)
4611 SELECT 'did:plc:x', id, 1, 1, '2026-01-01T00:00:00Z'
4612 FROM entries WHERE guid = 'starred-30d'",
4613 )
4614 .execute(&pool)
4615 .await?;
4616 sqlx::query(
4617 "INSERT INTO entry_state (did, entry_id, read, starred, updated_at)
4618 SELECT 'did:plc:x', id, 0, 0, '2026-01-01T00:00:00Z'
4619 FROM entries WHERE guid = 'unread-30d'",
4620 )
4621 .execute(&pool)
4622 .await?;
4623
4624 prune_old_entries(&pool, 14, hard, 0).await?;
4625
4626 let left: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
4627 .fetch_one(&pool)
4628 .await?;
4629 assert_eq!(
4630 left, 2,
4631 "hard_days={hard} destroyed starred/unread rows at the soft window"
4632 );
4633 }
4634 Ok(())
4635 }
4636
4637 /// Turning the rolling window off must NOT also turn the ceiling off.
4638 ///
4639 /// `prune_old_entries` used to return on `days <= 0` before the ceiling was
4640 /// even computed, so `RETENTION_DAYS=0` — advertised as "disables eviction" —
4641 /// meant no window AND no ceiling. That is the one configuration with no
4642 /// bound on the shared cache at all, and it stopped being survivable when the
4643 /// per-feed trim started sparing starred entries: nothing was left to catch
4644 /// them. The two knobs are independent now.
4645 #[tokio::test]
4646 async fn a_disabled_window_does_not_disable_the_ceiling() -> Result<()> {
4647 let pool = init_url("sqlite::memory:").await?;
4648 let feed_id = upsert_feed(
4649 &pool,
4650 &NewFeed {
4651 url: "https://example.com/f.xml".to_string(),
4652 ..Default::default()
4653 },
4654 )
4655 .await?;
4656 let age = |d: i64| {
4657 (chrono::Utc::now() - chrono::Duration::days(d))
4658 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true)
4659 };
4660 insert_entries(
4661 &pool,
4662 feed_id,
4663 &[
4664 NewEntry {
4665 guid: "starred-400d".to_string(),
4666 published: Some(age(400)),
4667 ..Default::default()
4668 },
4669 NewEntry {
4670 guid: "starred-30d".to_string(),
4671 published: Some(age(30)),
4672 ..Default::default()
4673 },
4674 ],
4675 0,
4676 )
4677 .await?;
4678 // Star both, so only the ceiling can remove either one — the soft
4679 // window's exception would spare them both even if it did run.
4680 sqlx::query(
4681 "INSERT INTO entry_state (did, entry_id, read, starred, updated_at)
4682 SELECT 'did:plc:x', id, 1, 1, '2026-01-01T00:00:00Z' FROM entries",
4683 )
4684 .execute(&pool)
4685 .await?;
4686
4687 // No rolling window; a 180-day ceiling.
4688 let deleted = prune_old_entries(&pool, 0, 180, 0).await?;
4689
4690 assert_eq!(
4691 deleted, 1,
4692 "retention_days=0 skipped the hard ceiling, leaving the cache unbounded"
4693 );
4694 let left: Vec<String> = sqlx::query_scalar("SELECT guid FROM entries ORDER BY guid")
4695 .fetch_all(&pool)
4696 .await?;
4697 assert_eq!(
4698 left,
4699 vec!["starred-30d".to_string()],
4700 "the ceiling removed the wrong rows with the window disabled"
4701 );
4702 Ok(())
4703 }
4704
4705 /// With BOTH knobs off, nothing is deleted — that is the documented
4706 /// "no eviction at all" configuration, and it must stay a true no-op rather
4707 /// than falling through to one of the two deletes with a degenerate cutoff.
4708 #[tokio::test]
4709 async fn both_knobs_off_deletes_nothing() -> Result<()> {
4710 let pool = init_url("sqlite::memory:").await?;
4711 let feed_id = upsert_feed(
4712 &pool,
4713 &NewFeed {
4714 url: "https://example.com/f.xml".to_string(),
4715 ..Default::default()
4716 },
4717 )
4718 .await?;
4719 insert_entries(
4720 &pool,
4721 feed_id,
4722 &[NewEntry {
4723 guid: "ancient".to_string(),
4724 published: Some(
4725 (chrono::Utc::now() - chrono::Duration::days(9999))
4726 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true),
4727 ),
4728 ..Default::default()
4729 }],
4730 0,
4731 )
4732 .await?;
4733
4734 assert_eq!(prune_old_entries(&pool, 0, 0, 0).await?, 0);
4735 let left: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
4736 .fetch_one(&pool)
4737 .await?;
4738 assert_eq!(left, 1);
4739 Ok(())
4740 }
4741
4742 /// Starred sparing must not remove the per-feed cap.
4743 ///
4744 /// The first version spared every starred row without limit: at cap=5 with
4745 /// 50 starred entries, 55 survived — 11x the cap, i.e. no cap at all.
4746 #[tokio::test]
4747 async fn per_feed_trim_stays_bounded_when_everything_is_starred() -> Result<()> {
4748 let pool = init_url("sqlite::memory:").await?;
4749 let feed_id = upsert_feed(
4750 &pool,
4751 &NewFeed {
4752 url: "https://example.com/f.xml".to_string(),
4753 ..Default::default()
4754 },
4755 )
4756 .await?;
4757 let entries: Vec<NewEntry> = (0..100)
4758 .map(|i| NewEntry {
4759 guid: format!("g-{i}"),
4760 published: Some(format!("2026-01-{:02}T00:00:00Z", (i % 28) + 1)),
4761 ..Default::default()
4762 })
4763 .collect();
4764 insert_entries(&pool, feed_id, &entries, 0).await?;
4765 sqlx::query(
4766 "INSERT INTO entry_state (did, entry_id, read, starred, updated_at)
4767 SELECT 'did:plc:x', id, 0, 1, '2026-01-01T00:00:00Z'
4768 FROM entries LIMIT 50",
4769 )
4770 .execute(&pool)
4771 .await?;
4772
4773 // Re-run the trim with cap = 5.
4774 insert_entries(&pool, feed_id, &[], 5).await?;
4775
4776 let left: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
4777 .fetch_one(&pool)
4778 .await?;
4779 assert!(
4780 left <= 10,
4781 "per-feed trim kept {left} rows for a cap of 5; sparing removed the bound"
4782 );
4783 Ok(())
4784 }
4785
4786 /// Init an in-memory SQLite, insert a feed + entries, read them back.
4787 #[tokio::test]
4788 async fn init_insert_readback() -> Result<()> {
4789 let pool = init_url("sqlite::memory:").await?;
4790
4791 // Insert a feed.
4792 let feed_id = upsert_feed(
4793 &pool,
4794 &NewFeed {
4795 url: "https://example.com/feed.xml".to_string(),
4796 title: Some("Example".to_string()),
4797 site_url: Some("https://example.com".to_string()),
4798 next_poll: Some("2026-07-12T00:00:00Z".to_string()),
4799 ..Default::default()
4800 },
4801 )
4802 .await?;
4803 assert!(feed_id > 0);
4804
4805 // Read the feed back by URL.
4806 let feed = get_feed_by_url(&pool, "https://example.com/feed.xml")
4807 .await?
4808 .expect("feed should exist");
4809 assert_eq!(feed.id, feed_id);
4810 assert_eq!(feed.title.as_deref(), Some("Example"));
4811 assert_eq!(feed.site_url.as_deref(), Some("https://example.com"));
4812
4813 // Upsert on the same URL updates rather than duplicating.
4814 let feed_id2 = upsert_feed(
4815 &pool,
4816 &NewFeed {
4817 url: "https://example.com/feed.xml".to_string(),
4818 title: Some("Example (renamed)".to_string()),
4819 ..Default::default()
4820 },
4821 )
4822 .await?;
4823 assert_eq!(feed_id, feed_id2, "same URL must reuse the same row");
4824
4825 // Insert two entries.
4826 let n = insert_entries(
4827 &pool,
4828 feed_id,
4829 &[
4830 NewEntry {
4831 guid: "guid-1".to_string(),
4832 url: Some("https://example.com/a".to_string()),
4833 title: Some("First".to_string()),
4834 published: Some("2026-07-10T08:00:00Z".to_string()),
4835 content_html: Some("<p>hello</p>".to_string()),
4836 ..Default::default()
4837 },
4838 NewEntry {
4839 guid: "guid-2".to_string(),
4840 url: Some("https://example.com/b".to_string()),
4841 title: Some("Second".to_string()),
4842 published: Some("2026-07-11T08:00:00Z".to_string()),
4843 ..Default::default()
4844 },
4845 ],
4846 0, // per-feed trim disabled for this test
4847 )
4848 .await?;
4849 assert_eq!(n, 2);
4850
4851 // The reader must subscribe to the feed for the scoped reads to return
4852 // its entries (per-DID isolation projection).
4853 let did = "did:plc:abc123";
4854 replace_sub_refs(&pool, did, &[feed_id]).await?;
4855
4856 // Read entries back (newest-published first).
4857 let entries = entries_for_feed(&pool, did, feed_id).await?;
4858 assert_eq!(entries.len(), 2);
4859 assert_eq!(entries[0].guid, "guid-2");
4860 assert_eq!(entries[1].guid, "guid-1");
4861 // The body is stored, but it is NOT in the list projection — that is the
4862 // point of `EntryListRow`. Read it the way the single-entry reader does.
4863 let body: Option<String> =
4864 sqlx::query_scalar("SELECT content_html FROM entries WHERE guid = 'guid-1'")
4865 .fetch_one(&pool)
4866 .await?;
4867 assert_eq!(body.as_deref(), Some("<p>hello</p>"));
4868
4869 // Re-inserting the same GUID dedups (updates in place, no new row).
4870 let n2 = insert_entries(
4871 &pool,
4872 feed_id,
4873 &[NewEntry {
4874 guid: "guid-1".to_string(),
4875 title: Some("First (edited)".to_string()),
4876 ..Default::default()
4877 }],
4878 0,
4879 )
4880 .await?;
4881 assert_eq!(n2, 1);
4882 assert_eq!(entries_for_feed(&pool, did, feed_id).await?.len(), 2);
4883
4884 // --- per-DID read state ---
4885 let e1 = entries.iter().find(|e| e.guid == "guid-1").unwrap().id;
4886
4887 // Both entries start unread.
4888 assert_eq!(get_unread_for_did(&pool, did).await?.len(), 2);
4889
4890 // Mark one read; unread count drops to 1.
4891 mark_read(&pool, did, e1, true).await?;
4892 let unread = get_unread_for_did(&pool, did).await?;
4893 assert_eq!(unread.len(), 1);
4894 assert_eq!(unread[0].guid, "guid-2");
4895
4896 // Star it; it shows in the starred list.
4897 mark_starred(&pool, did, e1, true).await?;
4898 let starred = get_starred_for_did(&pool, did).await?;
4899 assert_eq!(starred.len(), 1);
4900 assert_eq!(starred[0].id, e1);
4901
4902 // Mark-all-read clears the remaining unread.
4903 mark_feed_read(&pool, did, feed_id, true).await?;
4904 assert_eq!(get_unread_for_did(&pool, did).await?.len(), 0);
4905
4906 // --- read cursor (batched-sync bookkeeping) ---
4907 let cursor = ReadCursor {
4908 did: did.to_string(),
4909 feed_url: "https://example.com/feed.xml".to_string(),
4910 read_through: Some("2026-07-11T08:00:00Z".to_string()),
4911 read_ids: "[]".to_string(),
4912 unread_ids: "[]".to_string(),
4913 dirty: true,
4914 pds_created: false,
4915 updated_at: now_rfc3339(),
4916 };
4917 upsert_cursor(&pool, &cursor).await?;
4918
4919 let fetched = get_cursor(&pool, did, "https://example.com/feed.xml")
4920 .await?
4921 .expect("cursor should exist");
4922 assert_eq!(
4923 fetched.read_through.as_deref(),
4924 Some("2026-07-11T08:00:00Z")
4925 );
4926 assert!(fetched.dirty);
4927
4928 // The flusher sees exactly one dirty cursor.
4929 let dirty = dirty_cursors(&pool, did).await?;
4930 assert_eq!(dirty.len(), 1);
4931 let flushed_at = dirty[0].updated_at.clone();
4932
4933 // After a flush, clearing dirty (with the flushed snapshot's updated_at)
4934 // removes it from the flusher's view.
4935 clear_cursor_dirty(&pool, did, "https://example.com/feed.xml", &flushed_at).await?;
4936 assert_eq!(dirty_cursors(&pool, did).await?.len(), 0);
4937
4938 Ok(())
4939 }
4940
4941 // -----------------------------------------------------------------------
4942 // The bounded, body-free list projection.
4943 //
4944 // The three queries these replaced were `SELECT e.*` with no `LIMIT`. Both
4945 // halves of that are load-bearing on a 512 MB box: the projection dragged
4946 // an ~11.9 KB article body per row that no list surface reads, and the
4947 // missing bound let one reader's backlog decide how much a handler
4948 // allocates.
4949 // -----------------------------------------------------------------------
4950
4951 /// Seed `count` entries in one feed, each with a large body, subscribed by
4952 /// `did`. Returns the feed id.
4953 async fn seed_big_entries(pool: &SqlitePool, did: &str, count: usize) -> Result<i64> {
4954 let feed_id = upsert_feed(
4955 pool,
4956 &NewFeed {
4957 url: "https://example.com/big.xml".to_string(),
4958 ..Default::default()
4959 },
4960 )
4961 .await?;
4962 let body = "x".repeat(20_000);
4963 let entries: Vec<NewEntry> = (0..count)
4964 .map(|i| NewEntry {
4965 guid: format!("guid-{i:04}"),
4966 url: Some(format!("https://example.com/a/{i}")),
4967 title: Some(format!("Article {i}")),
4968 // Descending guid order matches descending published order, so
4969 // assertions can name the rows they expect.
4970 published: Some(format!("2026-01-{:02}T00:00:00Z", (i % 28) + 1)),
4971 content_html: Some(body.clone()),
4972 ..Default::default()
4973 })
4974 .collect();
4975 insert_entries(pool, feed_id, &entries, 0).await?;
4976 replace_sub_refs(pool, did, &[feed_id]).await?;
4977 Ok(feed_id)
4978 }
4979
4980 /// Review of #213: the ceiling only stops NEW future dates. A row stored
4981 /// with one before it, whose item has since left its feed, is never polled
4982 /// again to be corrected — so it stayed first in the list and survived the
4983 /// per-feed cap forever. Startup re-dates it.
4984 #[tokio::test]
4985 async fn a_stored_future_date_is_cleared_at_startup() -> Result<()> {
4986 let pool = init_url("sqlite::memory:").await?;
4987 let feed_id = upsert_feed(
4988 &pool,
4989 &NewFeed {
4990 url: "https://clock.example/f.xml".into(),
4991 ..Default::default()
4992 },
4993 )
4994 .await?;
4995 let tomorrow = (chrono::Utc::now() + chrono::Duration::days(1))
4996 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
4997 for (guid, published) in [
4998 ("bogus", "2999-01-01T00:00:00Z"),
4999 ("soon", tomorrow.as_str()),
5000 ] {
5001 sqlx::query(
5002 "INSERT INTO entries (feed_id, guid, published, fetched_at) \
5003 VALUES (?1, ?2, ?3, '2026-07-11T00:00:00Z')",
5004 )
5005 .bind(feed_id)
5006 .bind(guid)
5007 .bind(published)
5008 .execute(&pool)
5009 .await?;
5010 }
5011 apply_migrations(&pool).await?;
5012 let dated: Vec<(String, Option<String>)> =
5013 sqlx::query_as("SELECT guid, published FROM entries ORDER BY guid")
5014 .fetch_all(&pool)
5015 .await?;
5016 assert_eq!(
5017 dated[0],
5018 ("bogus".to_string(), None),
5019 "a 2999 date survived startup"
5020 );
5021 assert_eq!(
5022 dated[1].1.as_deref(),
5023 Some(tomorrow.as_str()),
5024 "a near-future date was cleared"
5025 );
5026 Ok(())
5027 }
5028
5029 /// **The cap, the reading list and prev/next must agree about what an undated
5030 /// entry's date IS.** They did not, and the disagreement had a direction.
5031 ///
5032 /// The per-feed keep-set and both retention sweeps order on
5033 /// `COALESCE(published, fetched_at)` — correctly, because a feed of undated
5034 /// items would otherwise trim its own freshest rows. The reading list
5035 /// ordered on bare `e.published DESC`, and in SQLite `NULL` sorts LAST under
5036 /// `DESC`. So one undated entry was simultaneously the NEWEST row in the
5037 /// feed as far as eviction was concerned, and the OLDEST row in every list
5038 /// view — parked below years of read articles where no reader would see it,
5039 /// while the cap declined to drop it to make room for something they would.
5040 ///
5041 /// `site.standard.document` makes `publishedAt` optional, so publication
5042 /// feeds reach this far more readily than RSS ever did.
5043 ///
5044 /// Both directions here: the undated row must come first, AND the two dated
5045 /// rows must stay in their own order, or "order by nothing" would pass.
5046 ///
5047 /// **On the index worry, measured on the query the app actually sends.**
5048 /// #187 flagged that a `COALESCE` in `ORDER BY` cannot use
5049 /// `idx_entries_feed_published` for ordering. The real list and prev/next
5050 /// queries (LEFT JOIN `entry_state`, EXISTS `sub_ref`) did not use it for
5051 /// ordering before this change either, and timing them at 40 feeds x 1,000
5052 /// entries showed the new ordering costs nothing on the existing index. A
5053 /// `(feed_id, published, fetched_at)` index meant to keep them covering was
5054 /// never chosen on the default prev/next query and made it ~3.8x slower,
5055 /// so it was not kept (review of #213).
5056 #[tokio::test]
5057 async fn an_undated_entry_leads_the_reading_list_as_it_leads_the_cap() -> Result<()> {
5058 let pool = init_url("sqlite::memory:").await?;
5059 let did = "did:plc:undated";
5060 let feed_id = upsert_feed(
5061 &pool,
5062 &NewFeed {
5063 url: "https://undated.example/f.xml".to_string(),
5064 ..Default::default()
5065 },
5066 )
5067 .await?;
5068 insert_entries(
5069 &pool,
5070 feed_id,
5071 &[
5072 NewEntry {
5073 guid: "dated-old".to_string(),
5074 title: Some("Old".to_string()),
5075 published: Some("2024-01-01T00:00:00Z".to_string()),
5076 ..Default::default()
5077 },
5078 NewEntry {
5079 guid: "dated-new".to_string(),
5080 title: Some("Newer".to_string()),
5081 published: Some("2025-01-01T00:00:00Z".to_string()),
5082 ..Default::default()
5083 },
5084 // No `published` at all — dated by `fetched_at`, which is now,
5085 // so it is the freshest row in the feed.
5086 NewEntry {
5087 guid: "undated".to_string(),
5088 title: Some("Undated".to_string()),
5089 ..Default::default()
5090 },
5091 ],
5092 0,
5093 )
5094 .await?;
5095 replace_sub_refs(&pool, did, &[feed_id]).await?;
5096
5097 let rows = list_entries(&pool, did, ListView::All, None, 100, 0).await?;
5098 let order: Vec<&str> = rows.iter().map(|r| r.guid.as_str()).collect();
5099 assert_eq!(
5100 order,
5101 vec!["undated", "dated-new", "dated-old"],
5102 "the list disagrees with the cap about an undated entry's date",
5103 );
5104
5105 // `list_entry_ids` is the sequence PREV/NEXT walks — its only non-test
5106 // caller is `web::neighbors_in_scope`. Ordered differently from the
5107 // list, "next entry" would take the reader somewhere that is not the
5108 // next row on screen. (`mark_read` and `mark_all_read` are id-based and
5109 // never use this ordering; an earlier version of this comment said they
5110 // did, naming a failure that cannot happen and omitting the one that
5111 // can.)
5112 let ids = list_entry_ids(&pool, did, ListView::All, None, 100).await?;
5113 let by_guid: std::collections::HashMap<i64, &str> =
5114 rows.iter().map(|r| (r.id, r.guid.as_str())).collect();
5115 let id_order: Vec<&str> = ids.iter().filter_map(|i| by_guid.get(i).copied()).collect();
5116 assert_eq!(
5117 id_order,
5118 vec!["undated", "dated-new", "dated-old"],
5119 "the id projection orders differently from the list it projects",
5120 );
5121 Ok(())
5122 }
5123
5124 /// `limit` is honoured, and `offset` walks the same ordering without gaps or
5125 /// repeats. Against the unbounded originals the first assertion returned all
5126 /// 250 rows.
5127 #[tokio::test]
5128 async fn list_entries_is_bounded_and_pages_without_overlap() -> Result<()> {
5129 let pool = init_url("sqlite::memory:").await?;
5130 let did = "did:plc:pager";
5131 seed_big_entries(&pool, did, 250).await?;
5132
5133 let page1 = list_entries(&pool, did, ListView::All, None, 100, 0).await?;
5134 assert_eq!(page1.len(), 100, "limit was not applied");
5135 let page2 = list_entries(&pool, did, ListView::All, None, 100, 100).await?;
5136 let page3 = list_entries(&pool, did, ListView::All, None, 100, 200).await?;
5137 assert_eq!(page3.len(), 50, "the last page should be the remainder");
5138
5139 let walked: Vec<i64> = page1
5140 .iter()
5141 .chain(&page2)
5142 .chain(&page3)
5143 .map(|e| e.id)
5144 .collect();
5145 let unique: std::collections::HashSet<i64> = walked.iter().copied().collect();
5146 assert_eq!(unique.len(), 250, "paging repeated or skipped rows");
5147
5148 // And the walk is the same order an unpaged read would produce.
5149 let whole = list_entries(&pool, did, ListView::All, None, 1_000, 0).await?;
5150 assert_eq!(
5151 walked,
5152 whole.iter().map(|e| e.id).collect::<Vec<_>>(),
5153 "paging changed the ordering"
5154 );
5155
5156 // **The tie-break is pinned, not left to the engine.** The seed gives
5157 // 250 rows only 28 distinct dates, so the order is mostly ties; with
5158 // the `id DESC` tie-break deleted, SQLite happened to return ties in a
5159 // stable order and both assertions above still held. The expected
5160 // order is computed from the seed pattern here — newest date first,
5161 // then newest id — and must match exactly.
5162 let mut expected: Vec<(i64, i64)> = whole
5163 .iter()
5164 .map(|e| {
5165 let day: i64 = e.published.as_deref().unwrap()[8..10].parse().unwrap();
5166 (day, e.id)
5167 })
5168 .collect();
5169 expected.sort_by(|a, b| b.cmp(a));
5170 assert_eq!(
5171 walked,
5172 expected.iter().map(|(_, id)| *id).collect::<Vec<_>>(),
5173 "ties are not broken by newest id"
5174 );
5175
5176 assert_eq!(
5177 count_entries_for_view(&pool, did, ListView::All, None).await?,
5178 250,
5179 "the unpaged count must survive paging"
5180 );
5181 Ok(())
5182 }
5183
5184 /// The list projection must not read `content_html`.
5185 ///
5186 /// A type-level fact — `EntryListRow` has no body field — so the test proves
5187 /// it the only way that survives a refactor: by asking SQLite what the query
5188 /// it runs actually names. `SELECT e.*` would list every column.
5189 #[tokio::test]
5190 async fn the_list_projection_does_not_name_the_body_column() -> Result<()> {
5191 let pool = init_url("sqlite::memory:").await?;
5192 let did = "did:plc:projection";
5193 seed_big_entries(&pool, did, 3).await?;
5194
5195 // **The projection the query actually runs**, not a copy re-typed here.
5196 // The earlier version passed its own literal to `list_query_sql` and
5197 // asserted on that, so adding `e.content_html` to `list_entries` left
5198 // this green.
5199 let (sql, _) = list_entries_sql(ListView::All, None);
5200 assert!(
5201 !sql.contains("content_html") && !sql.contains("e.*"),
5202 "the list query reads the article body: {sql}"
5203 );
5204
5205 // And the rows really do come back without it, which is what bounds the
5206 // per-request allocation.
5207 let rows = list_entries(&pool, did, ListView::All, None, 10, 0).await?;
5208 assert_eq!(rows.len(), 3);
5209 let widest = rows
5210 .iter()
5211 .map(|r| {
5212 r.guid.len()
5213 + r.url.as_deref().map_or(0, str::len)
5214 + r.title.as_deref().map_or(0, str::len)
5215 })
5216 .max()
5217 .unwrap_or(0);
5218 assert!(
5219 widest < 1_000,
5220 "a list row carries {widest} bytes of text; the 20,000-byte body leaked in"
5221 );
5222 Ok(())
5223 }
5224
5225 /// **A large scope must not become a large SQL statement.**
5226 ///
5227 /// The scope filter used to emit one placeholder per feed id, so the SQL
5228 /// string and the bind list both grew with a reader's subscription count —
5229 /// which comes from the PDS and is bounded only by a 20,000-record list
5230 /// ceiling. The first attempt at fixing that truncated the subscription
5231 /// list, which silently removed the reader's access to the dropped feeds
5232 /// (`sub_ref` is written from the same list). `json_each` takes the whole
5233 /// set as ONE bind, so neither trade-off is needed.
5234 #[tokio::test]
5235 async fn a_large_scope_is_one_bind_and_still_filters() -> Result<()> {
5236 let pool = init_url("sqlite::memory:").await?;
5237 let did = "did:plc:widescope";
5238
5239 // 300 feeds, one entry each; the scope names 200 of them.
5240 let mut all_ids = Vec::new();
5241 for i in 0..300 {
5242 let feed_id = upsert_feed(
5243 &pool,
5244 &NewFeed {
5245 url: format!("https://wide{i}.example/f.xml"),
5246 ..Default::default()
5247 },
5248 )
5249 .await?;
5250 insert_entries(
5251 &pool,
5252 feed_id,
5253 &[NewEntry {
5254 guid: format!("w-{i}"),
5255 ..Default::default()
5256 }],
5257 0,
5258 )
5259 .await?;
5260 all_ids.push(feed_id);
5261 }
5262 replace_sub_refs(&pool, did, &all_ids).await?;
5263
5264 let scope: Vec<i64> = all_ids.iter().copied().take(200).collect();
5265 let rows = list_entries(&pool, did, ListView::All, Some(&scope), 1_000, 0).await?;
5266 assert_eq!(rows.len(), 200, "the scope filter did not narrow correctly");
5267 let in_scope: std::collections::HashSet<i64> = scope.iter().copied().collect();
5268 assert!(
5269 rows.iter().all(|r| in_scope.contains(&r.feed_id)),
5270 "a feed outside the scope came back"
5271 );
5272 assert_eq!(
5273 count_entries_for_view(&pool, did, ListView::All, Some(&scope)).await?,
5274 200
5275 );
5276
5277 // The statement itself carries no per-id placeholders — that is the
5278 // property, and it is what stops the SQL growing with the reader.
5279 let (sql, n) = list_query_sql(Projection::Ids, ListView::All, Some(&scope));
5280 assert_eq!(n, 1, "the scope must contribute exactly one placeholder");
5281 assert!(
5282 sql.contains("json_each(?2)") && !sql.contains("?3"),
5283 "the scope is still expanded into per-id placeholders: {sql}"
5284 );
5285 Ok(())
5286 }
5287
5288 /// Scope is applied INSIDE the query, so a page is a page of rows the reader
5289 /// will see. Filtering after the `LIMIT` (what the handler used to do) made
5290 /// pages arbitrarily short for any narrowed scope.
5291 #[tokio::test]
5292 async fn a_feed_scope_narrows_the_query_not_the_page() -> Result<()> {
5293 let pool = init_url("sqlite::memory:").await?;
5294 let did = "did:plc:scope";
5295 let wanted = seed_big_entries(&pool, did, 10).await?;
5296
5297 let other = upsert_feed(
5298 &pool,
5299 &NewFeed {
5300 url: "https://other.example/f.xml".to_string(),
5301 ..Default::default()
5302 },
5303 )
5304 .await?;
5305 let noise: Vec<NewEntry> = (0..40)
5306 .map(|i| NewEntry {
5307 guid: format!("noise-{i}"),
5308 // Newer than everything in `wanted`, so an unscoped query would
5309 // fill the whole page with these.
5310 published: Some("2027-01-01T00:00:00Z".to_string()),
5311 ..Default::default()
5312 })
5313 .collect();
5314 insert_entries(&pool, other, &noise, 0).await?;
5315 replace_sub_refs(&pool, did, &[wanted, other]).await?;
5316
5317 let scoped = list_entries(&pool, did, ListView::All, Some(&[wanted]), 10, 0).await?;
5318 assert_eq!(
5319 scoped.len(),
5320 10,
5321 "the scoped page came back short — the filter ran after the LIMIT"
5322 );
5323 assert!(scoped.iter().all(|e| e.feed_id == wanted));
5324
5325 // An EMPTY scope means "no feeds in scope", not "every feed".
5326 assert!(list_entries(&pool, did, ListView::All, Some(&[]), 10, 0)
5327 .await?
5328 .is_empty());
5329 assert_eq!(
5330 count_entries_for_view(&pool, did, ListView::All, Some(&[])).await?,
5331 0
5332 );
5333 Ok(())
5334 }
5335
5336 /// The per-row `read` / `starred` bits come off the row's own join, matching
5337 /// what the separate full-set queries used to compute — including the
5338 /// "no `entry_state` row means unread" rule the views depend on.
5339 #[tokio::test]
5340 async fn list_rows_carry_their_own_read_and_star_bits() -> Result<()> {
5341 let pool = init_url("sqlite::memory:").await?;
5342 let did = "did:plc:bits";
5343 seed_big_entries(&pool, did, 3).await?;
5344 let ids: Vec<i64> = list_entries(&pool, did, ListView::All, None, 10, 0)
5345 .await?
5346 .iter()
5347 .map(|e| e.id)
5348 .collect();
5349
5350 mark_read(&pool, did, ids[0], true).await?;
5351 mark_starred(&pool, did, ids[1], true).await?;
5352
5353 let all = list_entries(&pool, did, ListView::All, None, 10, 0).await?;
5354 let by_id = |id: i64| all.iter().find(|e| e.id == id).expect("row present");
5355 assert!(by_id(ids[0]).read && !by_id(ids[0]).starred);
5356 assert!(!by_id(ids[1]).read && by_id(ids[1]).starred);
5357 // Never touched: no state row at all, which must read as unread.
5358 assert!(!by_id(ids[2]).read && !by_id(ids[2]).starred);
5359
5360 // And the view predicates agree with the bits.
5361 let unread = list_entries(&pool, did, ListView::Unread, None, 10, 0).await?;
5362 assert_eq!(unread.len(), 2);
5363 assert!(unread.iter().all(|e| !e.read));
5364 let starred = list_entries(&pool, did, ListView::Starred, None, 10, 0).await?;
5365 assert_eq!(starred.len(), 1);
5366 assert_eq!(starred[0].id, ids[1]);
5367 Ok(())
5368 }
5369
5370 /// The sidebar's per-feed unread badges, counted in SQL rather than by
5371 /// materializing every unread entry and filtering in Rust.
5372 #[tokio::test]
5373 async fn unread_counts_are_per_feed_and_exclude_read_rows() -> Result<()> {
5374 let pool = init_url("sqlite::memory:").await?;
5375 let did = "did:plc:counts";
5376 let a = seed_big_entries(&pool, did, 5).await?;
5377 let b = upsert_feed(
5378 &pool,
5379 &NewFeed {
5380 url: "https://b.example/f.xml".to_string(),
5381 ..Default::default()
5382 },
5383 )
5384 .await?;
5385 insert_entries(
5386 &pool,
5387 b,
5388 &[
5389 NewEntry {
5390 guid: "b-1".to_string(),
5391 ..Default::default()
5392 },
5393 NewEntry {
5394 guid: "b-2".to_string(),
5395 ..Default::default()
5396 },
5397 ],
5398 0,
5399 )
5400 .await?;
5401 replace_sub_refs(&pool, did, &[a, b]).await?;
5402
5403 let first_a = list_entries(&pool, did, ListView::All, Some(&[a]), 1, 0).await?[0].id;
5404 mark_read(&pool, did, first_a, true).await?;
5405
5406 let counts = unread_counts_by_feed(&pool, did).await?;
5407 assert_eq!(counts.get(&a).copied(), Some(4));
5408 assert_eq!(counts.get(&b).copied(), Some(2));
5409
5410 // A feed the DID does not subscribe to contributes nothing.
5411 replace_sub_refs(&pool, did, &[b]).await?;
5412 let counts = unread_counts_by_feed(&pool, did).await?;
5413 assert_eq!(counts.get(&a), None);
5414 assert_eq!(counts.get(&b).copied(), Some(2));
5415 Ok(())
5416 }
5417
5418 /// **Read-state compaction: the water-mark must absorb the id set.**
5419 ///
5420 /// `read_through` was never computed, so `read_ids` was the only mechanism
5421 /// and grew one id per article read against a 2000-entry per-feed ceiling —
5422 /// while the flusher truncates the record at 1000, keeping the tail. Past
5423 /// 1000 read articles in a feed, the oldest read-state stopped syncing and
5424 /// those articles came back UNREAD in every other atproto reader.
5425 #[tokio::test]
5426 async fn compaction_folds_read_ids_into_the_water_mark() -> Result<()> {
5427 let pool = init_url("sqlite::memory:").await?;
5428 let did = "did:plc:compact";
5429 let feed_url = "https://compact.example/f.xml";
5430 let feed_id = upsert_feed(
5431 &pool,
5432 &NewFeed {
5433 url: feed_url.to_string(),
5434 ..Default::default()
5435 },
5436 )
5437 .await?;
5438 // 40 entries, oldest first by published date.
5439 let entries: Vec<NewEntry> = (0..40)
5440 .map(|i| NewEntry {
5441 guid: format!("c-{i:03}"),
5442 published: Some(format!("2026-01-{:02}T00:00:00Z", i + 1)),
5443 ..Default::default()
5444 })
5445 .collect();
5446 insert_entries(&pool, feed_id, &entries, 0).await?;
5447 replace_sub_refs(&pool, did, &[feed_id]).await?;
5448
5449 let all = list_entries(&pool, did, ListView::All, None, 100, 0).await?;
5450 // Oldest first, so the read prefix is contiguous from the start.
5451 let mut oldest_first = all.clone();
5452 oldest_first.reverse();
5453 for row in oldest_first.iter().take(30) {
5454 mark_read(&pool, did, row.id, true).await?;
5455 }
5456
5457 let before = get_cursor(&pool, did, feed_url).await?.expect("cursor");
5458 assert!(before.read_through.is_none(), "read_through starts unset");
5459 let before_ids: Vec<String> = serde_json::from_str(&before.read_ids)?;
5460 assert_eq!(before_ids.len(), 30, "every read is its own exception");
5461
5462 let watermark = compact_cursor(&pool, did, feed_url)
5463 .await?
5464 .expect("the water-mark must advance");
5465
5466 let after = get_cursor(&pool, did, feed_url).await?.expect("cursor");
5467 assert_eq!(after.read_through.as_deref(), Some(watermark.as_str()));
5468 let after_ids: Vec<String> = serde_json::from_str(&after.read_ids)?;
5469 assert!(
5470 after_ids.is_empty(),
5471 "a contiguous read prefix must fold entirely into the water-mark, left {after_ids:?}"
5472 );
5473 // The 30th entry is read and the 31st is not, so the mark sits on the
5474 // 30th — STRICTLY below the oldest unread, never equal to it.
5475 assert_eq!(watermark, "2026-01-30T00:00:00Z");
5476 assert!(after.dirty, "a rewritten cursor must be re-flushed");
5477 Ok(())
5478 }
5479
5480 /// The water-mark may never cover an unread entry, and may never move
5481 /// backwards. Both would re-assert articles as read that are not.
5482 #[tokio::test]
5483 async fn compaction_stops_below_the_oldest_unread_entry() -> Result<()> {
5484 let pool = init_url("sqlite::memory:").await?;
5485 let did = "did:plc:gap";
5486 let feed_url = "https://gap.example/f.xml";
5487 let feed_id = upsert_feed(
5488 &pool,
5489 &NewFeed {
5490 url: feed_url.to_string(),
5491 ..Default::default()
5492 },
5493 )
5494 .await?;
5495 let entries: Vec<NewEntry> = (0..10)
5496 .map(|i| NewEntry {
5497 guid: format!("g-{i:02}"),
5498 published: Some(format!("2026-02-{:02}T00:00:00Z", i + 1)),
5499 ..Default::default()
5500 })
5501 .collect();
5502 insert_entries(&pool, feed_id, &entries, 0).await?;
5503 replace_sub_refs(&pool, did, &[feed_id]).await?;
5504
5505 let mut oldest_first = list_entries(&pool, did, ListView::All, None, 100, 0).await?;
5506 oldest_first.reverse();
5507 // Read everything EXCEPT the third-oldest: a hole at 2026-02-03.
5508 for (i, row) in oldest_first.iter().enumerate() {
5509 if i != 2 {
5510 mark_read(&pool, did, row.id, true).await?;
5511 }
5512 }
5513
5514 let watermark = compact_cursor(&pool, did, feed_url)
5515 .await?
5516 .expect("advances");
5517 assert_eq!(
5518 watermark, "2026-02-02T00:00:00Z",
5519 "the water-mark jumped the unread hole"
5520 );
5521 let after = get_cursor(&pool, did, feed_url).await?.expect("cursor");
5522 let kept: Vec<String> = serde_json::from_str(&after.read_ids)?;
5523 assert_eq!(
5524 kept.len(),
5525 7,
5526 "the 7 reads ABOVE the hole must stay as explicit exceptions"
5527 );
5528 // The unread hole is above the water-mark, so it needs no unread
5529 // exception — everything above the mark is unread by default.
5530 let unread: Vec<String> = serde_json::from_str(&after.unread_ids)?;
5531 assert!(
5532 unread.is_empty(),
5533 "redundant unread exceptions survived: {unread:?}"
5534 );
5535
5536 // Idempotent, and never backwards: re-running changes nothing.
5537 assert_eq!(
5538 compact_cursor(&pool, did, feed_url).await?,
5539 None,
5540 "a second compaction moved a water-mark that was already correct"
5541 );
5542 Ok(())
5543 }
5544
5545 /// Nothing read yet, or nothing in the feed: compaction must be a no-op
5546 /// rather than inventing a water-mark that asserts the backlog is read.
5547 #[tokio::test]
5548 async fn compaction_never_invents_a_water_mark() -> Result<()> {
5549 let pool = init_url("sqlite::memory:").await?;
5550 let did = "did:plc:none";
5551 let feed_url = "https://none.example/f.xml";
5552 let feed_id = upsert_feed(
5553 &pool,
5554 &NewFeed {
5555 url: feed_url.to_string(),
5556 ..Default::default()
5557 },
5558 )
5559 .await?;
5560 replace_sub_refs(&pool, did, &[feed_id]).await?;
5561
5562 // Empty feed: no entries at all.
5563 assert_eq!(compact_cursor(&pool, did, feed_url).await?, None);
5564
5565 insert_entries(
5566 &pool,
5567 feed_id,
5568 &[
5569 NewEntry {
5570 guid: "n-1".to_string(),
5571 published: Some("2026-03-01T00:00:00Z".to_string()),
5572 ..Default::default()
5573 },
5574 NewEntry {
5575 guid: "n-2".to_string(),
5576 published: Some("2026-03-02T00:00:00Z".to_string()),
5577 ..Default::default()
5578 },
5579 ],
5580 0,
5581 )
5582 .await?;
5583
5584 // Nothing read: the OLDEST entry is unread, so there is no timestamp
5585 // strictly below it and the mark cannot move at all.
5586 assert_eq!(
5587 compact_cursor(&pool, did, feed_url).await?,
5588 None,
5589 "a water-mark appeared with nothing read — that asserts the backlog is read"
5590 );
5591 Ok(())
5592 }
5593
5594 /// **The unsave desync: clearing a star must work for an UNSUBSCRIBED feed.**
5595 ///
5596 /// That is the whole case. Every other starred path is `sub_ref`-scoped, so
5597 /// an entry that is cached AND starred in a feed the reader has since
5598 /// unsubscribed from is invisible to all of them — including the starred
5599 /// list itself. Its PDS record therefore renders as "not cached", and the
5600 /// button on that row deletes the record. If clearing the local star were
5601 /// `sub_ref`-scoped too, it would silently do nothing, and the star would
5602 /// reappear with no record behind it the moment the reader resubscribed.
5603 #[tokio::test]
5604 async fn a_star_can_be_cleared_after_unsubscribing_from_its_feed() -> Result<()> {
5605 let pool = init_url("sqlite::memory:").await?;
5606 let did = "did:plc:unsub";
5607 let feed_id = seed_big_entries(&pool, did, 3).await?;
5608 let rows = list_entries(&pool, did, ListView::All, None, 10, 0).await?;
5609 let target = rows[0].clone();
5610 mark_starred(&pool, did, target.id, true).await?;
5611 assert_eq!(get_starred_for_did(&pool, did).await?.len(), 1);
5612
5613 // Unsubscribe. The entry stays cached and stays starred, but every
5614 // sub_ref-scoped read now skips it.
5615 replace_sub_refs(&pool, did, &[]).await?;
5616 assert!(
5617 get_starred_for_did(&pool, did).await?.is_empty(),
5618 "fixture precondition: the star must be invisible to the scoped read"
5619 );
5620 assert!(
5621 matches!(
5622 starred_identities(&pool, did, 1_000).await?,
5623 StarredIdentities::All(ref v) if v.is_empty()
5624 ),
5625 "fixture precondition: the identity lookup must miss it too"
5626 );
5627 let still_starred: i64 =
5628 sqlx::query_scalar("SELECT COUNT(*) FROM entry_state WHERE did = ?1 AND starred = 1")
5629 .bind(did)
5630 .fetch_one(&pool)
5631 .await?;
5632 assert_eq!(
5633 still_starred, 1,
5634 "the star is still there, just unreachable"
5635 );
5636
5637 // The removal path must reach it anyway.
5638 let cleared =
5639 clear_star_by_identity(&pool, did, target.url.as_deref(), Some(&target.guid)).await?;
5640 assert_eq!(cleared, 1, "the star survived the unsave");
5641 let after: i64 =
5642 sqlx::query_scalar("SELECT COUNT(*) FROM entry_state WHERE did = ?1 AND starred = 1")
5643 .bind(did)
5644 .fetch_one(&pool)
5645 .await?;
5646 assert_eq!(after, 0);
5647
5648 // Resubscribing must NOT bring it back.
5649 replace_sub_refs(&pool, did, &[feed_id]).await?;
5650 assert!(
5651 get_starred_for_did(&pool, did).await?.is_empty(),
5652 "the star came back after resubscribing — the desync is still there"
5653 );
5654 Ok(())
5655 }
5656
5657 /// It clears only the CALLER's star, and only for the matching article.
5658 ///
5659 /// Omitting `sub_ref` is safe precisely because `did` is not optional; this
5660 /// pins that, and that a non-matching identity is a no-op rather than a
5661 /// wildcard.
5662 #[tokio::test]
5663 async fn clearing_a_star_touches_only_that_did_and_that_article() -> Result<()> {
5664 let pool = init_url("sqlite::memory:").await?;
5665 let mine = "did:plc:mine";
5666 let theirs = "did:plc:theirs";
5667 let feed_id = seed_big_entries(&pool, mine, 3).await?;
5668 replace_sub_refs(&pool, theirs, &[feed_id]).await?;
5669 let rows = list_entries(&pool, mine, ListView::All, None, 10, 0).await?;
5670
5671 for r in &rows {
5672 mark_starred(&pool, mine, r.id, true).await?;
5673 mark_starred(&pool, theirs, r.id, true).await?;
5674 }
5675
5676 let target = &rows[1];
5677 assert_eq!(
5678 clear_star_by_identity(&pool, mine, target.url.as_deref(), Some(&target.guid)).await?,
5679 1
5680 );
5681
5682 let count = |did: &'static str| {
5683 let pool = pool.clone();
5684 async move {
5685 sqlx::query_scalar::<_, i64>(
5686 "SELECT COUNT(*) FROM entry_state WHERE did = ?1 AND starred = 1",
5687 )
5688 .bind(did)
5689 .fetch_one(&pool)
5690 .await
5691 .unwrap()
5692 }
5693 };
5694 assert_eq!(count(mine).await, 2, "it cleared more than the one article");
5695 assert_eq!(count(theirs).await, 3, "it cleared another DID's stars");
5696
5697 // An identity that matches nothing is a no-op, not a wildcard.
5698 assert_eq!(
5699 clear_star_by_identity(&pool, mine, Some("https://nope.example/x"), Some("nope"))
5700 .await?,
5701 0
5702 );
5703 assert_eq!(count(mine).await, 2);
5704 // **Clearing an already-cleared star is a no-op**, reported as one:
5705 // `web::unsave` branches on `Ok(0)` vs `Ok(n)` to decide whether a
5706 // local star was actually cleared. This used to be untested — every
5707 // article here was starred first — so `starred = 1` in the WHERE clause
5708 // could be widened to `IN (0, 1)` with the suite green, rewriting
5709 // `updated_at` on rows that changed nothing and logging clears that
5710 // never happened.
5711 let before: String = sqlx::query_scalar(
5712 "SELECT updated_at FROM entry_state WHERE did = ?1 AND entry_id = ?2",
5713 )
5714 .bind(mine)
5715 .bind(target.id)
5716 .fetch_one(&pool)
5717 .await?;
5718 assert_eq!(
5719 clear_star_by_identity(&pool, mine, target.url.as_deref(), Some(&target.guid)).await?,
5720 0,
5721 "a second clear reported rows it did not change"
5722 );
5723 let after: String = sqlx::query_scalar(
5724 "SELECT updated_at FROM entry_state WHERE did = ?1 AND entry_id = ?2",
5725 )
5726 .bind(mine)
5727 .bind(target.id)
5728 .fetch_one(&pool)
5729 .await?;
5730 assert_eq!(before, after, "a no-op clear rewrote updated_at");
5731
5732 // And neither identifier present does nothing at all.
5733 assert_eq!(clear_star_by_identity(&pool, mine, None, None).await?, 0);
5734 assert_eq!(
5735 clear_star_by_identity(&pool, mine, Some(""), Some("")).await?,
5736 0
5737 );
5738 assert_eq!(count(mine).await, 2);
5739 Ok(())
5740 }
5741
5742 /// `starred_identities` must span the WHOLE starred set, not a page.
5743 ///
5744 /// The starred view matches PDS saved records against it; a cached article
5745 /// missing from the set renders as "not cached", and that row's button
5746 /// deletes the PDS RECORD instead of un-starring the entry. Narrowing this
5747 /// set changes what a click destroys.
5748 #[tokio::test]
5749 async fn starred_identities_span_the_whole_set() -> Result<()> {
5750 let pool = init_url("sqlite::memory:").await?;
5751 let did = "did:plc:ident";
5752 seed_big_entries(&pool, did, 150).await?;
5753 for row in list_entries(&pool, did, ListView::All, None, 1_000, 0).await? {
5754 mark_starred(&pool, did, row.id, true).await?;
5755 }
5756
5757 let identities = match starred_identities(&pool, did, 20_000).await? {
5758 StarredIdentities::All(v) => v,
5759 StarredIdentities::Truncated => panic!("150 rows must not read as truncated"),
5760 };
5761 assert_eq!(
5762 identities.len(),
5763 150,
5764 "the identity set was truncated to a page"
5765 );
5766 assert!(identities
5767 .iter()
5768 .all(|(url, guid)| url.is_some() && !guid.is_empty()));
5769
5770 // **Hitting the cap must be REPORTED, not absorbed.** It used to return
5771 // an arbitrary subset with no way to tell, and every starred article
5772 // outside that subset then rendered an un-save button that deletes the
5773 // PDS record rather than un-starring the entry.
5774 assert!(
5775 matches!(
5776 starred_identities(&pool, did, 10).await?,
5777 StarredIdentities::Truncated
5778 ),
5779 "a truncated identity set reported itself as complete"
5780 );
5781 // Landing EXACTLY on the cap is complete, not truncated — the query asks
5782 // for one extra row precisely so the two are distinguishable.
5783 assert!(
5784 matches!(
5785 starred_identities(&pool, did, 150).await?,
5786 StarredIdentities::All(ref v) if v.len() == 150
5787 ),
5788 "a set exactly at the cap was misreported as truncated"
5789 );
5790 Ok(())
5791 }
5792
5793 /// Prev/next ids are bounded too, and keep the list's ordering.
5794 #[tokio::test]
5795 async fn entry_ids_are_ordered_and_capped() -> Result<()> {
5796 let pool = init_url("sqlite::memory:").await?;
5797 let did = "did:plc:ids";
5798 seed_big_entries(&pool, did, 60).await?;
5799
5800 let capped = list_entry_ids(&pool, did, ListView::All, None, 25).await?;
5801 assert_eq!(capped.len(), 25);
5802
5803 let rows = list_entries(&pool, did, ListView::All, None, 25, 0).await?;
5804 assert_eq!(
5805 capped,
5806 rows.iter().map(|e| e.id).collect::<Vec<_>>(),
5807 "the id list and the row list disagree on ordering"
5808 );
5809 Ok(())
5810 }
5811
5812 // -----------------------------------------------------------------------
5813 // Read-state PDS sync wiring: marking read/unread must project into the
5814 // per-feed `read_cursor` and mark it dirty so the batched flusher pushes it.
5815 // Before this wiring `mark_read` touched only `entry_state`; nothing dirtied
5816 // a cursor, so the flusher never synced read-state to the PDS.
5817 // -----------------------------------------------------------------------
5818
5819 #[tokio::test]
5820 async fn mark_read_dirties_the_feed_cursor() -> Result<()> {
5821 let pool = init_url("sqlite::memory:").await?;
5822 let feed_url = "https://example.com/feed.xml";
5823 let feed_id = upsert_feed(
5824 &pool,
5825 &NewFeed {
5826 url: feed_url.to_string(),
5827 title: Some("Example".to_string()),
5828 ..Default::default()
5829 },
5830 )
5831 .await?;
5832 insert_entries(
5833 &pool,
5834 feed_id,
5835 &[
5836 NewEntry {
5837 guid: "g1".to_string(),
5838 published: Some("2026-07-10T00:00:00Z".to_string()),
5839 ..Default::default()
5840 },
5841 NewEntry {
5842 guid: "g2".to_string(),
5843 published: Some("2026-07-11T00:00:00Z".to_string()),
5844 ..Default::default()
5845 },
5846 ],
5847 0,
5848 )
5849 .await?;
5850 let did = "did:plc:reader";
5851 replace_sub_refs(&pool, did, &[feed_id]).await?;
5852
5853 // No cursor exists yet.
5854 assert!(get_cursor(&pool, did, feed_url).await?.is_none());
5855 assert_eq!(dirty_cursors(&pool, did).await?.len(), 0);
5856
5857 // Mark one entry read → the feed's read_cursor row now exists, dirty=1,
5858 // and dirty_cursors returns it (the exact assertion the fix requires).
5859 let e1 = entries_for_feed(&pool, did, feed_id).await?[0].id;
5860 assert!(mark_read(&pool, did, e1, true).await?);
5861
5862 let cursor = get_cursor(&pool, did, feed_url)
5863 .await?
5864 .expect("mark_read must create the feed's read_cursor");
5865 assert!(cursor.dirty, "cursor must be dirty after mark_read");
5866 assert!(
5867 cursor.read_ids.contains(&e1.to_string()),
5868 "the read entry id must be in read_ids: {}",
5869 cursor.read_ids
5870 );
5871 let dirty = dirty_cursors(&pool, did).await?;
5872 assert_eq!(dirty.len(), 1, "flusher must see the newly dirty cursor");
5873 assert_eq!(dirty[0].feed_url, feed_url);
5874
5875 // Marking it unread again moves the id to unread_ids and keeps it dirty.
5876 assert!(mark_read(&pool, did, e1, false).await?);
5877 let cursor = get_cursor(&pool, did, feed_url).await?.unwrap();
5878 assert!(cursor.dirty);
5879 assert!(
5880 cursor.unread_ids.contains(&e1.to_string()),
5881 "unread id must be in unread_ids: {}",
5882 cursor.unread_ids
5883 );
5884 assert!(
5885 !cursor.read_ids.contains(&e1.to_string()),
5886 "id must have left read_ids: {}",
5887 cursor.read_ids
5888 );
5889
5890 // mark_feed_read dirties the one per-feed cursor too (batched, not
5891 // per-article).
5892 assert!(mark_feed_read(&pool, did, feed_id, true).await? > 0);
5893 let cursor = get_cursor(&pool, did, feed_url).await?.unwrap();
5894 assert!(cursor.dirty);
5895 assert_eq!(dirty_cursors(&pool, did).await?.len(), 1);
5896
5897 // A non-subscriber's mark_read is a no-op and dirties NO cursor.
5898 let outsider = "did:plc:outsider";
5899 assert!(!mark_read(&pool, outsider, e1, true).await?);
5900 assert_eq!(dirty_cursors(&pool, outsider).await?.len(), 0);
5901
5902 // The conditional clear only clears when updated_at matches the snapshot.
5903 let snap = dirty_cursors(&pool, did).await?[0].clone();
5904 // A stale updated_at must NOT clear (models a concurrent re-dirty).
5905 clear_cursor_dirty(&pool, did, feed_url, "1999-01-01T00:00:00Z").await?;
5906 assert_eq!(
5907 dirty_cursors(&pool, did).await?.len(),
5908 1,
5909 "stale-snapshot clear must be a no-op"
5910 );
5911 // The matching updated_at clears it.
5912 clear_cursor_dirty(&pool, did, feed_url, &snap.updated_at).await?;
5913 assert_eq!(dirty_cursors(&pool, did).await?.len(), 0);
5914
5915 Ok(())
5916 }
5917
5918 #[test]
5919 fn json_id_set_toggle_is_set_like() {
5920 // Add is idempotent, remove drops, output is a JSON string array.
5921 let s = json_id_set_toggle("[]", 5, true);
5922 assert_eq!(s, r#"["5"]"#);
5923 assert_eq!(json_id_set_toggle(&s, 5, true), r#"["5"]"#); // no dup
5924 let s = json_id_set_toggle(&s, 7, true);
5925 assert_eq!(s, r#"["5","7"]"#);
5926 let s = json_id_set_toggle(&s, 5, false);
5927 assert_eq!(s, r#"["7"]"#);
5928 // Tolerates numeric-array input and malformed input.
5929 assert_eq!(json_id_set_toggle("[1,2]", 3, true), r#"["1","2","3"]"#);
5930 assert_eq!(json_id_set_toggle("garbage", 1, true), r#"["1"]"#);
5931 }
5932
5933 // -----------------------------------------------------------------------
5934 // Per-DID isolation: the shared cache is one row per URL, but the READ
5935 // SURFACE (entries/unread/starred) and the read/star MUTATIONS are scoped
5936 // to the caller's own subscriptions (`sub_ref`). User A must never see or
5937 // mutate user B's entries.
5938 // -----------------------------------------------------------------------
5939
5940 #[tokio::test]
5941 async fn per_did_isolation_scopes_reads_and_mutations() -> Result<()> {
5942 let pool = init_url("sqlite::memory:").await?;
5943
5944 // Two feeds in the SHARED cache; A subscribes to feed_a, B to feed_b.
5945 let feed_a = upsert_feed(
5946 &pool,
5947 &NewFeed {
5948 url: "https://a.example/feed.xml".to_string(),
5949 title: Some("A".to_string()),
5950 ..Default::default()
5951 },
5952 )
5953 .await?;
5954 let feed_b = upsert_feed(
5955 &pool,
5956 &NewFeed {
5957 url: "https://b.example/feed.xml".to_string(),
5958 title: Some("B".to_string()),
5959 ..Default::default()
5960 },
5961 )
5962 .await?;
5963
5964 insert_entries(
5965 &pool,
5966 feed_a,
5967 &[NewEntry {
5968 guid: "a-1".to_string(),
5969 url: Some("https://a.example/1".to_string()),
5970 title: Some("A one".to_string()),
5971 published: Some("2026-07-10T00:00:00Z".to_string()),
5972 content_html: Some("<p>secret A body</p>".to_string()),
5973 ..Default::default()
5974 }],
5975 0,
5976 )
5977 .await?;
5978 insert_entries(
5979 &pool,
5980 feed_b,
5981 &[NewEntry {
5982 guid: "b-1".to_string(),
5983 url: Some("https://b.example/1".to_string()),
5984 title: Some("B one".to_string()),
5985 published: Some("2026-07-11T00:00:00Z".to_string()),
5986 content_html: Some("<p>secret B body</p>".to_string()),
5987 ..Default::default()
5988 }],
5989 0,
5990 )
5991 .await?;
5992
5993 let did_a = "did:plc:aaaa";
5994 let did_b = "did:plc:bbbb";
5995 replace_sub_refs(&pool, did_a, &[feed_a]).await?;
5996 replace_sub_refs(&pool, did_b, &[feed_b]).await?;
5997
5998 // The id of B's only entry (the one A must not be able to touch).
5999 let b_entry_id = entries_for_feed(&pool, did_b, feed_b).await?[0].id;
6000
6001 // --- entries_for_feed is scoped: A sees A's feed, not B's ------------
6002 assert_eq!(entries_for_feed(&pool, did_a, feed_a).await?.len(), 1);
6003 assert!(
6004 entries_for_feed(&pool, did_a, feed_b).await?.is_empty(),
6005 "A must not read entries of a feed it does not subscribe to"
6006 );
6007
6008 // --- unread list is scoped -------------------------------------------
6009 let unread_a = get_unread_for_did(&pool, did_a).await?;
6010 assert_eq!(unread_a.len(), 1);
6011 assert_eq!(unread_a[0].guid, "a-1");
6012 let unread_b = get_unread_for_did(&pool, did_b).await?;
6013 assert_eq!(unread_b.len(), 1);
6014 assert_eq!(unread_b[0].guid, "b-1");
6015
6016 // --- did_subscribes_to_entry authorizes correctly --------------------
6017 assert!(did_subscribes_to_entry(&pool, did_b, b_entry_id).await?);
6018 assert!(
6019 !did_subscribes_to_entry(&pool, did_a, b_entry_id).await?,
6020 "A does not subscribe to B's feed"
6021 );
6022
6023 // --- mark_read is authorized: A CANNOT mark B's entry ----------------
6024 assert!(
6025 !mark_read(&pool, did_a, b_entry_id, true).await?,
6026 "non-subscriber mark_read must be a no-op (→ 404), never a mutation"
6027 );
6028 // B's unread list is untouched by A's attempt.
6029 assert_eq!(get_unread_for_did(&pool, did_b).await?.len(), 1);
6030 // A subscriber CAN mark it.
6031 assert!(mark_read(&pool, did_b, b_entry_id, true).await?);
6032 assert_eq!(get_unread_for_did(&pool, did_b).await?.len(), 0);
6033
6034 // --- toggle_star is authorized the same way --------------------------
6035 assert!(
6036 !mark_starred(&pool, did_a, b_entry_id, true).await?,
6037 "non-subscriber mark_starred must be a no-op (→ 404)"
6038 );
6039 assert!(
6040 get_starred_for_did(&pool, did_a).await?.is_empty(),
6041 "A's starred list stays empty after the rejected attempt"
6042 );
6043 assert!(mark_starred(&pool, did_b, b_entry_id, true).await?);
6044 assert_eq!(get_starred_for_did(&pool, did_b).await?.len(), 1);
6045 // B's star never leaks into A's starred list.
6046 assert!(get_starred_for_did(&pool, did_a).await?.is_empty());
6047
6048 // --- feeds_for_did is scoped to the DID's OWN sub_ref ----------------
6049 // This is the PDS-unreachable fallback's projection: it must NEVER
6050 // widen a DID's surface to feeds it does not subscribe to. A sees only
6051 // feed_a; B (still subscribed to feed_b here) sees only feed_b.
6052 let a_feeds = feeds_for_did(&pool, did_a).await?;
6053 assert_eq!(a_feeds.len(), 1);
6054 assert_eq!(a_feeds[0].id, feed_a);
6055 let b_feeds = feeds_for_did(&pool, did_b).await?;
6056 assert_eq!(b_feeds.len(), 1);
6057 assert_eq!(b_feeds[0].id, feed_b);
6058
6059 // --- resync drops a feed from the surface when the sub goes away ------
6060 replace_sub_refs(&pool, did_b, &[]).await?;
6061 assert!(get_unread_for_did(&pool, did_b).await?.is_empty());
6062 assert!(get_starred_for_did(&pool, did_b).await?.is_empty());
6063 assert!(entries_for_feed(&pool, did_b, feed_b).await?.is_empty());
6064 // And the fallback projection is empty too — fail CLOSED, not open.
6065 assert!(feeds_for_did(&pool, did_b).await?.is_empty());
6066
6067 Ok(())
6068 }
6069
6070 // -----------------------------------------------------------------------
6071 // PDS-outage authorization (fail CLOSED). REGRESSION GUARD for the past
6072 // FAIL-OPEN bug (fixed in 2e53e0e): `resolve_subscriptions`' PDS/sidecar-
6073 // unreachable fallback used to synthesize a DID's `sub_ref` from EVERY
6074 // cached feed (`due_feeds(.., i64::MAX)`), granting cross-tenant read +
6075 // mutate during any outage. The fix serves the DID's OWN last-known
6076 // `sub_ref` via `feeds_for_did(did)` and NEVER widens it.
6077 //
6078 // This test replays that fixed fallback at the store layer — the seam the
6079 // web handler drives when `list_subscriptions_sorted(did) -> Err`. The
6080 // key adversarial shape is an ORPHAN cached feed (in the shared cache but
6081 // subscribed by NO ONE): the old fail-open code would have folded it into
6082 // the caller's surface. If the fail-open is reintroduced, `feeds_for_did`
6083 // would include that orphan and every assertion below flips — so this is a
6084 // real guard, not a tautology.
6085 // -----------------------------------------------------------------------
6086
6087 #[tokio::test]
6088 async fn pds_outage_fallback_fails_closed_not_open() -> Result<()> {
6089 let pool = init_url("sqlite::memory:").await?;
6090
6091 let did_a = "did:plc:aaaa";
6092
6093 // feed_a: A's own subscription (its last-known `sub_ref`; the fallback
6094 // may serve this stale but must not widen past it).
6095 let feed_a = upsert_feed(
6096 &pool,
6097 &NewFeed {
6098 url: "https://a.example/feed.xml".to_string(),
6099 title: Some("A".to_string()),
6100 ..Default::default()
6101 },
6102 )
6103 .await?;
6104 // feed_orphan: present in the SHARED cache but subscribed by NO DID.
6105 // This is exactly what the fail-open path would have leaked to A.
6106 let feed_orphan = upsert_feed(
6107 &pool,
6108 &NewFeed {
6109 url: "https://orphan.example/feed.xml".to_string(),
6110 title: Some("Orphan".to_string()),
6111 ..Default::default()
6112 },
6113 )
6114 .await?;
6115
6116 insert_entries(
6117 &pool,
6118 feed_a,
6119 &[NewEntry {
6120 guid: "a-1".to_string(),
6121 url: Some("https://a.example/1".to_string()),
6122 title: Some("A one".to_string()),
6123 published: Some("2026-07-10T00:00:00Z".to_string()),
6124 content_html: Some("<p>A body</p>".to_string()),
6125 ..Default::default()
6126 }],
6127 0,
6128 )
6129 .await?;
6130 insert_entries(
6131 &pool,
6132 feed_orphan,
6133 &[NewEntry {
6134 guid: "orphan-1".to_string(),
6135 url: Some("https://orphan.example/1".to_string()),
6136 title: Some("Orphan one".to_string()),
6137 published: Some("2026-07-11T00:00:00Z".to_string()),
6138 content_html: Some("<p>secret orphan body</p>".to_string()),
6139 ..Default::default()
6140 }],
6141 0,
6142 )
6143 .await?;
6144
6145 // A's last-known subscription set is feed_a ONLY. No `sub_ref` row ever
6146 // points any DID at feed_orphan.
6147 replace_sub_refs(&pool, did_a, &[feed_a]).await?;
6148
6149 // Grab the orphan entry id via a transient sub so we can address it,
6150 // then drop the sub — nobody subscribes to feed_orphan afterwards.
6151 replace_sub_refs(&pool, "did:plc:seed", &[feed_orphan]).await?;
6152 let orphan_entry_id = entries_for_feed(&pool, "did:plc:seed", feed_orphan).await?[0].id;
6153 replace_sub_refs(&pool, "did:plc:seed", &[]).await?;
6154
6155 // --- Replay the FIXED fallback projection ----------------------------
6156 // This is what `resolve_subscriptions` serves on the Err (outage) path:
6157 // the caller's OWN feeds, never widened. It must contain feed_a and
6158 // NEVER the orphan. (The old fail-open synthesized from every cached
6159 // feed → this vec would have held feed_orphan too.)
6160 let fallback = feeds_for_did(&pool, did_a).await?;
6161 let fallback_ids: Vec<i64> = fallback.iter().map(|f| f.id).collect();
6162 assert_eq!(
6163 fallback_ids,
6164 vec![feed_a],
6165 "outage fallback must serve ONLY A's own last-known sub_ref, \
6166 never widen to the orphan cached feed"
6167 );
6168 assert!(
6169 !fallback_ids.contains(&feed_orphan),
6170 "FAIL-OPEN regression: outage fallback leaked an unsubscribed \
6171 cached feed into A's surface"
6172 );
6173
6174 // --- With that projection in place, EVERY scoped read denies A -------
6175 assert!(
6176 !did_subscribes_to_entry(&pool, did_a, orphan_entry_id).await?,
6177 "A must not be authorized for an orphan feed's entry during an outage"
6178 );
6179 assert!(
6180 entries_for_feed(&pool, did_a, feed_orphan)
6181 .await?
6182 .is_empty(),
6183 "entries_for_feed must not expose the orphan feed to A during an outage"
6184 );
6185 // Neither the unread nor the starred list may surface the orphan entry.
6186 let unread_guids: Vec<String> = get_unread_for_did(&pool, did_a)
6187 .await?
6188 .into_iter()
6189 .map(|e| e.guid)
6190 .collect();
6191 assert!(
6192 !unread_guids.iter().any(|g| g == "orphan-1"),
6193 "orphan entry leaked into A's unread list during an outage"
6194 );
6195 assert!(
6196 get_starred_for_did(&pool, did_a).await?.is_empty(),
6197 "A has no starred entries; the orphan must not appear"
6198 );
6199
6200 // --- And EVERY scoped mutation is a no-op (→ 404 at the web layer) ---
6201 assert!(
6202 !mark_read(&pool, did_a, orphan_entry_id, true).await?,
6203 "A must not mark an orphan feed's entry read during an outage"
6204 );
6205 assert!(
6206 !mark_starred(&pool, did_a, orphan_entry_id, true).await?,
6207 "A must not star an orphan feed's entry during an outage"
6208 );
6209 assert_eq!(
6210 mark_feed_read(&pool, did_a, feed_orphan, true).await?,
6211 0,
6212 "A must not mark-all-read the orphan feed during an outage"
6213 );
6214
6215 // Nothing was written for A against the orphan entry.
6216 let es_count: i64 =
6217 sqlx::query_scalar("SELECT COUNT(*) FROM entry_state WHERE did = ?1 AND entry_id = ?2")
6218 .bind(did_a)
6219 .bind(orphan_entry_id)
6220 .fetch_one(&pool)
6221 .await?;
6222 assert_eq!(es_count, 0, "no cross-tenant mutation during the outage");
6223
6224 Ok(())
6225 }
6226
6227 // -----------------------------------------------------------------------
6228 // Closed-beta invite gate
6229 // -----------------------------------------------------------------------
6230
6231 #[test]
6232 fn code_gen_shape_and_alphabet() {
6233 for _ in 0..200 {
6234 let code = generate_invite_code().unwrap();
6235 assert!(code.starts_with("FEATHER-"), "bad prefix: {code}");
6236 let body = &code["FEATHER-".len()..];
6237 assert_eq!(body.len(), CODE_BODY_LEN, "bad body length: {code}");
6238 // Every body char must be from the ambiguity-free alphabet — in
6239 // particular NEVER I/O/0/1.
6240 for c in body.chars() {
6241 assert!(
6242 CODE_ALPHABET.contains(&(c as u8)),
6243 "char {c:?} not in alphabet ({code})"
6244 );
6245 assert!(
6246 !matches!(c, 'I' | 'O' | '0' | '1'),
6247 "ambiguous char {c:?} leaked into {code}"
6248 );
6249 }
6250 }
6251 // Two codes in a row must differ (unguessable / random).
6252 assert_ne!(
6253 generate_invite_code().unwrap(),
6254 generate_invite_code().unwrap()
6255 );
6256 }
6257
6258 #[tokio::test]
6259 async fn busy_timeout_is_applied() -> Result<()> {
6260 // Opening an on-disk DB and reading back the PRAGMA proves the pool
6261 // carries busy_timeout = 5000 ms.
6262 let dir = std::env::temp_dir().join(format!("fr-busy-{}", std::process::id()));
6263 std::fs::create_dir_all(&dir).ok();
6264 let path = dir.join("busy.db");
6265 let url = format!("sqlite://{}", path.display());
6266 let pool = init_url(&url).await?;
6267 let row = sqlx::query("PRAGMA busy_timeout").fetch_one(&pool).await?;
6268 let timeout: i64 = row.get(0);
6269 assert_eq!(timeout, 5000, "busy_timeout should be 5000 ms");
6270 pool.close().await;
6271 std::fs::remove_dir_all(&dir).ok();
6272 Ok(())
6273 }
6274
6275 #[tokio::test]
6276 async fn redeem_valid_grants_seat() -> Result<()> {
6277 let pool = init_url("sqlite::memory:").await?;
6278 let code = mint_code(&pool, "did:plc:creator", 3600).await?;
6279 assert!(!has_beta_access(&pool, "did:plc:new").await?);
6280
6281 let out = redeem_code(&pool, &code, "did:plc:new", Some("new.bsky"), 100).await?;
6282 assert_eq!(out, Ok(()));
6283 assert!(has_beta_access(&pool, "did:plc:new").await?);
6284 assert_eq!(count_beta_access(&pool).await?, 1);
6285
6286 // The code is now spent — a second redeem is AlreadyRedeemed.
6287 let again = redeem_code(&pool, &code, "did:plc:other", None, 100).await?;
6288 assert_eq!(again, Err(RedeemError::AlreadyRedeemed));
6289 Ok(())
6290 }
6291
6292 #[tokio::test]
6293 async fn redeem_not_found() -> Result<()> {
6294 let pool = init_url("sqlite::memory:").await?;
6295 let out = redeem_code(&pool, "FEATHER-NOPENOPE", "did:plc:x", None, 100).await?;
6296 assert_eq!(out, Err(RedeemError::NotFound));
6297 Ok(())
6298 }
6299
6300 /// Insert an already-expired `active` code directly (mint_code clamps a
6301 /// negative ttl to 0, so the past-expiry case is set up by hand).
6302 async fn insert_expired_code(pool: &SqlitePool, code: &str, creator: &str) -> Result<()> {
6303 let now = now_unix();
6304 sqlx::query(
6305 r#"INSERT INTO invite_codes
6306 (code, creator_did, status, invitee_did, created_at, expires_at, redeemed_at)
6307 VALUES (?1, ?2, 'active', NULL, ?3, ?4, NULL)"#,
6308 )
6309 .bind(code)
6310 .bind(creator)
6311 .bind(now - 100)
6312 .bind(now - 10) // expires_at in the past
6313 .execute(pool)
6314 .await?;
6315 Ok(())
6316 }
6317
6318 #[tokio::test]
6319 async fn redeem_expired() -> Result<()> {
6320 let pool = init_url("sqlite::memory:").await?;
6321 insert_expired_code(&pool, "FEATHER-EXPIRED0", "did:plc:creator").await?;
6322 let out = redeem_code(&pool, "FEATHER-EXPIRED0", "did:plc:new", None, 100).await?;
6323 assert_eq!(out, Err(RedeemError::Expired));
6324 // No seat granted.
6325 assert_eq!(count_beta_access(&pool).await?, 0);
6326 Ok(())
6327 }
6328
6329 #[tokio::test]
6330 async fn redeem_capacity_full() -> Result<()> {
6331 let pool = init_url("sqlite::memory:").await?;
6332 // Cap of 1, one seat already taken by an admin seed.
6333 ensure_seed(&pool, &["did:plc:admin".to_string()]).await?;
6334 assert_eq!(count_beta_access(&pool).await?, 1);
6335
6336 let code = mint_code(&pool, "did:plc:admin", 3600).await?;
6337 let out = redeem_code(&pool, &code, "did:plc:new", None, 1).await?;
6338 assert_eq!(out, Err(RedeemError::CapacityFull));
6339 // Seat NOT granted and the code NOT consumed (tx rolled back).
6340 assert!(!has_beta_access(&pool, "did:plc:new").await?);
6341 // Raising the cap lets the same code redeem.
6342 let ok = redeem_code(&pool, &code, "did:plc:new", None, 2).await?;
6343 assert_eq!(ok, Ok(()));
6344 Ok(())
6345 }
6346
6347 #[tokio::test]
6348 async fn count_active_codes_excludes_expired_and_redeemed() -> Result<()> {
6349 let pool = init_url("sqlite::memory:").await?;
6350 assert_eq!(count_active_codes(&pool).await?, 0);
6351
6352 // Two live codes.
6353 let a = mint_code(&pool, "did:plc:bot", 3600).await?;
6354 let _b = mint_code(&pool, "did:plc:bot", 3600).await?;
6355 assert_eq!(count_active_codes(&pool).await?, 2);
6356
6357 // An expired code doesn't count.
6358 insert_expired_code(&pool, "FEATHER-EXPIRED0", "did:plc:bot").await?;
6359 assert_eq!(count_active_codes(&pool).await?, 2);
6360
6361 // Redeeming one drops the active count.
6362 let out = redeem_code(&pool, &a, "did:plc:new", None, 100).await?;
6363 assert_eq!(out, Ok(()));
6364 assert_eq!(count_active_codes(&pool).await?, 1);
6365 Ok(())
6366 }
6367
6368 #[tokio::test]
6369 async fn expire_and_seed() -> Result<()> {
6370 let pool = init_url("sqlite::memory:").await?;
6371 // An already-expired code is swept to `expired`.
6372 insert_expired_code(&pool, "FEATHER-EXPIRED1", "did:plc:creator").await?;
6373 let live = mint_code(&pool, "did:plc:creator", 3600).await?;
6374 let n = expire_old_codes(&pool).await?;
6375 assert_eq!(n, 1, "exactly the past-expiry code should flip");
6376 // The live code still redeems.
6377 assert_eq!(
6378 redeem_code(&pool, &live, "did:plc:new", None, 100).await?,
6379 Ok(())
6380 );
6381
6382 // ensure_seed is idempotent.
6383 let created = ensure_seed(
6384 &pool,
6385 &["did:plc:seed1".to_string(), "did:plc:seed2".to_string()],
6386 )
6387 .await?;
6388 assert_eq!(created, 2);
6389 let created2 = ensure_seed(&pool, &["did:plc:seed1".to_string()]).await?;
6390 assert_eq!(created2, 0, "re-seeding an existing DID is a no-op");
6391 assert!(has_beta_access(&pool, "did:plc:seed1").await?);
6392 Ok(())
6393 }
6394
6395 /// **The sweep spares a REDEEMED code that is past its TTL.** The
6396 /// existing sweep test seeds one active past-expiry code and one live
6397 /// one, so the `status = 'active'` guard never excludes anything — with
6398 /// it deleted the suite stayed green. Without it the hourly sweep rewrites
6399 /// redeemed codes to `expired`, destroying the redemption the invite audit
6400 /// trail depends on and inflating the logged sweep count.
6401 #[tokio::test]
6402 async fn the_expiry_sweep_spares_redeemed_codes() -> Result<()> {
6403 let pool = init_url("sqlite::memory:").await?;
6404 let code = mint_code(&pool, "did:plc:creator", 3600).await?;
6405 assert!(redeem_code(&pool, &code, "did:plc:new", None, 100)
6406 .await?
6407 .is_ok());
6408 // Time passes: the redeemed code is now past its TTL.
6409 sqlx::query("UPDATE invite_codes SET expires_at = ?1 WHERE code = ?2")
6410 .bind(now_unix() - 10)
6411 .bind(&code)
6412 .execute(&pool)
6413 .await?;
6414 insert_expired_code(&pool, "FEATHER-EXPIRED2", "did:plc:creator").await?;
6415
6416 let n = expire_old_codes(&pool).await?;
6417 assert_eq!(n, 1, "the sweep counted the redeemed code");
6418 let status: String = sqlx::query_scalar("SELECT status FROM invite_codes WHERE code = ?1")
6419 .bind(&code)
6420 .fetch_one(&pool)
6421 .await?;
6422 assert_eq!(status, "redeemed", "the sweep rewrote a redemption");
6423 Ok(())
6424 }
6425
6426 // -----------------------------------------------------------------------
6427 // Hardening caps: per-DID sub count, global feed count, per-feed entry trim.
6428 // -----------------------------------------------------------------------
6429
6430 #[tokio::test]
6431 async fn count_helpers_track_feeds_and_subs() -> Result<()> {
6432 let pool = init_url("sqlite::memory:").await?;
6433 assert_eq!(count_feeds(&pool).await?, 0);
6434
6435 let mut ids = Vec::new();
6436 for i in 0..3 {
6437 let id = upsert_feed(
6438 &pool,
6439 &NewFeed {
6440 url: format!("https://f{i}.example/feed.xml"),
6441 ..Default::default()
6442 },
6443 )
6444 .await?;
6445 ids.push(id);
6446 }
6447 assert_eq!(count_feeds(&pool).await?, 3);
6448
6449 let did = "did:plc:capcheck";
6450 assert_eq!(count_subscriptions_for_did(&pool, did).await?, 0);
6451 replace_sub_refs(&pool, did, &ids).await?;
6452 assert_eq!(count_subscriptions_for_did(&pool, did).await?, 3);
6453 Ok(())
6454 }
6455
6456 #[tokio::test]
6457 async fn insert_entries_trims_over_cap_keeping_newest() -> Result<()> {
6458 let pool = init_url("sqlite::memory:").await?;
6459 let feed_id = upsert_feed(
6460 &pool,
6461 &NewFeed {
6462 url: "https://firehose.example/feed.xml".to_string(),
6463 ..Default::default()
6464 },
6465 )
6466 .await?;
6467
6468 // Insert 5 entries with ascending published dates, cap retained to 2.
6469 let batch: Vec<NewEntry> = (0..5)
6470 .map(|i| NewEntry {
6471 guid: format!("g-{i}"),
6472 title: Some(format!("E{i}")),
6473 published: Some(format!("2026-07-0{}T00:00:00Z", i + 1)),
6474 ..Default::default()
6475 })
6476 .collect();
6477 insert_entries(&pool, feed_id, &batch, 2).await?;
6478
6479 let did = "did:plc:trim";
6480 replace_sub_refs(&pool, did, &[feed_id]).await?;
6481 let kept = entries_for_feed(&pool, did, feed_id).await?;
6482 assert_eq!(
6483 kept.len(),
6484 2,
6485 "over-cap feed trimmed to the newest 2 entries"
6486 );
6487 // Newest first: g-4 (2026-07-05), g-3 (2026-07-04).
6488 assert_eq!(kept[0].guid, "g-4");
6489 assert_eq!(kept[1].guid, "g-3");
6490 Ok(())
6491 }
6492
6493 /// Regression: an UNDATED entry (NULL `published`) that was fetched most
6494 /// recently must NOT be evicted in favour of an older *dated* entry. The
6495 /// trim orders by `COALESCE(published, fetched_at) DESC`; under the old
6496 /// `ORDER BY published DESC` a NULL-published row sorts LAST and is dropped
6497 /// first even when it is the freshest thing in the feed.
6498 #[tokio::test]
6499 async fn insert_entries_trims_keeps_fresh_undated_over_stale_dated() -> Result<()> {
6500 let pool = init_url("sqlite::memory:").await?;
6501 let feed_id = upsert_feed(
6502 &pool,
6503 &NewFeed {
6504 url: "https://undated.example/feed.xml".to_string(),
6505 ..Default::default()
6506 },
6507 )
6508 .await?;
6509
6510 // Two OLD dated entries (fetched long ago), plus one UNDATED entry
6511 // fetched most recently. Cap = 2, so exactly one row must be evicted.
6512 let batch = vec![
6513 NewEntry {
6514 guid: "old-dated-1".to_string(),
6515 title: Some("Old A".to_string()),
6516 published: Some("2026-07-01T00:00:00Z".to_string()),
6517 fetched_at: Some("2026-07-01T00:00:00Z".to_string()),
6518 ..Default::default()
6519 },
6520 NewEntry {
6521 guid: "old-dated-2".to_string(),
6522 title: Some("Old B".to_string()),
6523 published: Some("2026-07-02T00:00:00Z".to_string()),
6524 fetched_at: Some("2026-07-02T00:00:00Z".to_string()),
6525 ..Default::default()
6526 },
6527 NewEntry {
6528 guid: "fresh-undated".to_string(),
6529 title: Some("Fresh undated".to_string()),
6530 published: None,
6531 fetched_at: Some("2026-07-11T00:00:00Z".to_string()),
6532 ..Default::default()
6533 },
6534 ];
6535 insert_entries(&pool, feed_id, &batch, 2).await?;
6536
6537 let did = "did:plc:undated";
6538 replace_sub_refs(&pool, did, &[feed_id]).await?;
6539 let kept = entries_for_feed(&pool, did, feed_id).await?;
6540 assert_eq!(kept.len(), 2, "over-cap feed trimmed to 2 entries");
6541 let guids: Vec<&str> = kept.iter().map(|e| e.guid.as_str()).collect();
6542 assert!(
6543 guids.contains(&"fresh-undated"),
6544 "the freshly-fetched undated entry must survive the trim, kept: {guids:?}"
6545 );
6546 assert!(
6547 guids.contains(&"old-dated-2"),
6548 "the newer dated entry survives; the OLDEST dated entry is the one evicted, kept: {guids:?}"
6549 );
6550 assert!(
6551 !guids.contains(&"old-dated-1"),
6552 "the oldest dated entry is the one that should be evicted, kept: {guids:?}"
6553 );
6554 Ok(())
6555 }
6556
6557 /// **It must actually GROW — the name used to be a lie.**
6558 ///
6559 /// The earlier body was three lines asserting only `before > 0`. There was
6560 /// no second measurement, so `db_size_bytes` returning a constant `1` passed.
6561 /// That matters because this number is the poller's disk watermark: a size
6562 /// that never moves means the pause never trips and the volume fills
6563 /// instead.
6564 #[tokio::test]
6565 async fn db_size_is_positive_and_grows() -> Result<()> {
6566 let pool = init_url("sqlite::memory:").await?;
6567 let before = db_size_bytes(&pool).await?;
6568 assert!(before > 0, "a schema-initialised DB has a non-zero size");
6569
6570 // Enough rows that the file must gain pages, not just fill slack.
6571 seed_big_entries(&pool, "did:plc:growth", 400).await?;
6572
6573 let after = db_size_bytes(&pool).await?;
6574 assert!(
6575 after > before,
6576 "the database grew by {} bytes after 400 seeded entries; the size is \
6577 not tracking the data, so the disk watermark can never trip",
6578 after.saturating_sub(before),
6579 );
6580 Ok(())
6581 }
6582
6583 /// `purge_did_data` removes every per-DID row the caller owns (read/star
6584 /// state, cursors, sub_ref projection, beta seat, created invite codes) —
6585 /// and touches no other DID's rows nor the shared feeds/entries cache.
6586 #[tokio::test]
6587 async fn purge_did_data_removes_only_the_callers_rows() -> Result<()> {
6588 let pool = init_url("sqlite::memory:").await?;
6589
6590 // A shared feed + entry both DIDs can subscribe to.
6591 let feed_id = upsert_feed(
6592 &pool,
6593 &NewFeed {
6594 url: "https://example.com/feed.xml".to_string(),
6595 title: Some("Example".to_string()),
6596 ..Default::default()
6597 },
6598 )
6599 .await?;
6600 insert_entries(
6601 &pool,
6602 feed_id,
6603 &[NewEntry {
6604 guid: "g-1".to_string(),
6605 url: Some("https://example.com/a".to_string()),
6606 title: Some("First".to_string()),
6607 published: Some("2026-07-10T08:00:00Z".to_string()),
6608 ..Default::default()
6609 }],
6610 0,
6611 )
6612 .await?;
6613 let entry_id: i64 = sqlx::query_scalar("SELECT id FROM entries WHERE guid = 'g-1'")
6614 .fetch_one(&pool)
6615 .await?;
6616
6617 let victim = "did:plc:victim";
6618 let bystander = "did:plc:bystander";
6619
6620 // Seed BOTH DIDs with a full spread of per-DID rows.
6621 for did in [victim, bystander] {
6622 replace_sub_refs(&pool, did, &[feed_id]).await?;
6623 assert!(mark_read(&pool, did, entry_id, true).await?);
6624 assert!(mark_starred(&pool, did, entry_id, true).await?);
6625 upsert_cursor(
6626 &pool,
6627 &ReadCursor {
6628 did: did.to_string(),
6629 feed_url: "https://example.com/feed.xml".to_string(),
6630 read_through: Some("2026-07-10T08:00:00Z".to_string()),
6631 read_ids: "[]".to_string(),
6632 unread_ids: "[]".to_string(),
6633 dirty: false,
6634 pds_created: false,
6635 updated_at: now_rfc3339(),
6636 },
6637 )
6638 .await?;
6639 grant_access(&pool, did, Some("h.example"), "admin", None).await?;
6640 mint_code(&pool, did, 3600).await?;
6641 }
6642
6643 // Purge only the victim.
6644 let counts = purge_did_data(&pool, victim).await?;
6645 assert_eq!(
6646 counts.entry_state, 1,
6647 "one entry_state row (read+star merge)"
6648 );
6649 assert_eq!(counts.read_cursor, 1);
6650 assert_eq!(counts.sub_ref, 1);
6651 assert_eq!(counts.beta_access, 1);
6652 assert_eq!(counts.invite_codes, 1);
6653 assert_eq!(counts.total(), 5);
6654
6655 // The victim has zero rows left in every per-DID table.
6656 let es: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entry_state WHERE did = ?1")
6657 .bind(victim)
6658 .fetch_one(&pool)
6659 .await?;
6660 assert_eq!(es, 0, "victim still had entry_state rows");
6661 let rc: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM read_cursor WHERE did = ?1")
6662 .bind(victim)
6663 .fetch_one(&pool)
6664 .await?;
6665 assert_eq!(rc, 0, "victim still had read_cursor rows");
6666 let sr: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM sub_ref WHERE did = ?1")
6667 .bind(victim)
6668 .fetch_one(&pool)
6669 .await?;
6670 assert_eq!(sr, 0, "victim still had sub_ref rows");
6671 assert!(
6672 !has_beta_access(&pool, victim).await?,
6673 "victim still had a beta seat"
6674 );
6675 let victim_codes: i64 =
6676 sqlx::query_scalar("SELECT COUNT(*) FROM invite_codes WHERE creator_did = ?1")
6677 .bind(victim)
6678 .fetch_one(&pool)
6679 .await?;
6680 assert_eq!(victim_codes, 0);
6681
6682 // The bystander is untouched.
6683 assert!(has_beta_access(&pool, bystander).await?);
6684 let bystander_subs = count_subscriptions_for_did(&pool, bystander).await?;
6685 assert_eq!(bystander_subs, 1, "bystander's sub_ref survived");
6686 let bystander_codes: i64 =
6687 sqlx::query_scalar("SELECT COUNT(*) FROM invite_codes WHERE creator_did = ?1")
6688 .bind(bystander)
6689 .fetch_one(&pool)
6690 .await?;
6691 assert_eq!(bystander_codes, 1);
6692
6693 // The shared cache is intact.
6694 assert_eq!(count_feeds(&pool).await?, 1);
6695
6696 // Idempotent: purging again removes nothing.
6697 let again = purge_did_data(&pool, victim).await?;
6698 assert_eq!(again.total(), 0);
6699
6700 Ok(())
6701 }
6702
6703 /// A departing DID leaves back-references on rows that belong to OTHER DIDs:
6704 /// * the invite code it *redeemed* to join (inviter's row: `invitee_did`);
6705 /// * seats it *granted* to others (`beta_access.granted_by`).
6706 /// `purge_did_data` must scrub both so no per-DID residue survives, while
6707 /// leaving those other DIDs' rows otherwise intact (their access is kept).
6708 #[tokio::test]
6709 async fn purge_did_data_scrubs_cross_did_back_references() -> Result<()> {
6710 let pool = init_url("sqlite::memory:").await?;
6711
6712 let inviter = "did:plc:inviter";
6713 let leaver = "did:plc:leaver";
6714 let friend = "did:plc:friend";
6715
6716 // inviter mints a code; leaver redeems it to join (stamps invitee_did).
6717 let inviter_code = mint_code(&pool, inviter, 3600).await?;
6718 grant_access(&pool, inviter, None, "admin", None).await?;
6719 assert_eq!(
6720 redeem_code(&pool, &inviter_code, leaver, Some("leaver.bsky"), 100).await?,
6721 Ok(())
6722 );
6723
6724 // leaver mints a code; friend redeems it (stamps friend's granted_by).
6725 let leaver_code = mint_code(&pool, leaver, 3600).await?;
6726 assert_eq!(
6727 redeem_code(&pool, &leaver_code, friend, Some("friend.bsky"), 100).await?,
6728 Ok(())
6729 );
6730
6731 // Precondition: the leaver DID is present in both back-reference columns.
6732 let invitee_before: i64 =
6733 sqlx::query_scalar("SELECT COUNT(*) FROM invite_codes WHERE invitee_did = ?1")
6734 .bind(leaver)
6735 .fetch_one(&pool)
6736 .await?;
6737 assert_eq!(
6738 invitee_before, 1,
6739 "leaver should be an invitee before purge"
6740 );
6741 let granted_before: i64 =
6742 sqlx::query_scalar("SELECT COUNT(*) FROM beta_access WHERE granted_by = ?1")
6743 .bind(leaver)
6744 .fetch_one(&pool)
6745 .await?;
6746 assert_eq!(granted_before, 1, "leaver should be a granter before purge");
6747
6748 // Purge the leaver.
6749 let counts = purge_did_data(&pool, leaver).await?;
6750 assert_eq!(
6751 counts.invitee_scrubbed, 1,
6752 "the redeemed code's invitee_did"
6753 );
6754 assert_eq!(counts.granted_by_scrubbed, 1, "the seat leaver granted");
6755
6756 // No residue: the leaver DID appears in NEITHER back-reference column.
6757 let invitee_after: i64 =
6758 sqlx::query_scalar("SELECT COUNT(*) FROM invite_codes WHERE invitee_did = ?1")
6759 .bind(leaver)
6760 .fetch_one(&pool)
6761 .await?;
6762 assert_eq!(invitee_after, 0, "leaver survived in invitee_did");
6763 let granted_after: i64 =
6764 sqlx::query_scalar("SELECT COUNT(*) FROM beta_access WHERE granted_by = ?1")
6765 .bind(leaver)
6766 .fetch_one(&pool)
6767 .await?;
6768 assert_eq!(granted_after, 0, "leaver survived in granted_by");
6769
6770 // The other DIDs' rows are kept: the friend still has a seat (redacted
6771 // granter), and the inviter's code row still exists (invitee NULLed).
6772 assert!(
6773 has_beta_access(&pool, friend).await?,
6774 "friend's seat must survive the leaver's scrub"
6775 );
6776 let friend_granted_by: String =
6777 sqlx::query_scalar("SELECT granted_by FROM beta_access WHERE did = ?1")
6778 .bind(friend)
6779 .fetch_one(&pool)
6780 .await?;
6781 assert_eq!(friend_granted_by, REDACTED_DID);
6782 let inviter_code_rows: i64 =
6783 sqlx::query_scalar("SELECT COUNT(*) FROM invite_codes WHERE creator_did = ?1")
6784 .bind(inviter)
6785 .fetch_one(&pool)
6786 .await?;
6787 assert_eq!(inviter_code_rows, 1, "inviter's code row must survive");
6788
6789 Ok(())
6790 }
6791
6792 // -- F2: consecutive-error count drives the poll backoff -----------------
6793
6794 /// **Rows that failed only because we could not poll them are cleared.**
6795 ///
6796 /// Excluding `at://` from `due_feeds` stops NEW failures; it does nothing
6797 /// about the ones already recorded. This instance carries 19 such rows at
6798 /// 35+ consecutive errors each — accumulated entirely by our own refusal to
6799 /// fetch a scheme we had not implemented. Left alone they keep counting
6800 /// toward `in_backoff` and `badly_broken`, so a public page would report
6801 /// unsupported feeds as broken publishers forever, with no poll that could
6802 /// ever clear them since they are no longer selected.
6803 ///
6804 /// Safe to re-run because of WHAT it clears, not because the count cannot
6805 /// grow: only rows never polled successfully (`last_polled IS NULL`) — see
6806 /// `the_at_uri_error_clearing_spares_a_row_that_has_been_polled`.
6807 #[tokio::test]
6808 async fn the_migration_clears_error_counts_on_unpollable_at_uri_rows() -> Result<()> {
6809 let pool = init_url("sqlite::memory:").await?;
6810 for url in [
6811 "https://real.example/feed.xml",
6812 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/app.bsky.feed.post/3lab",
6813 ] {
6814 upsert_feed(
6815 &pool,
6816 &NewFeed {
6817 url: url.to_string(),
6818 ..Default::default()
6819 },
6820 )
6821 .await?;
6822 sqlx::query(
6823 "UPDATE feeds SET consecutive_errors = 35, last_error_kind = 'fetch', \
6824 last_error = 'unsupported scheme' WHERE url = ?1",
6825 )
6826 .bind(url)
6827 .execute(&pool)
6828 .await?;
6829 }
6830
6831 apply_migrations(&pool).await?;
6832
6833 let (at_errors, at_kind, at_detail): (i64, Option<String>, Option<String>) =
6834 sqlx::query_as(sqlx::AssertSqlSafe(format!(
6835 "SELECT consecutive_errors, last_error_kind, last_error FROM feeds \
6836 WHERE kind = '{}'",
6837 crate::feed::FeedKind::Unsupported.as_str()
6838 )))
6839 .fetch_one(&pool)
6840 .await?;
6841 assert_eq!(at_errors, 0, "an unpollable row kept its failure count");
6842 // A row with no errors carries no reason — the invariant
6843 // `reset_feed_errors` upholds, and the migration must too.
6844 assert_eq!(at_kind, None, "an unpollable row kept its failure kind");
6845 assert_eq!(at_detail, None, "an unpollable row kept its failure detail");
6846
6847 // A real feed's failure history is NOT touched — it is still meaningful.
6848 let http_errors: i64 = sqlx::query_scalar(
6849 "SELECT consecutive_errors FROM feeds WHERE url = 'https://real.example/feed.xml'",
6850 )
6851 .fetch_one(&pool)
6852 .await?;
6853 assert_eq!(http_errors, 35, "a real feed's history was discarded");
6854 Ok(())
6855 }
6856
6857 /// **An `at://` feed is never selected for polling.**
6858 ///
6859 /// (Since 0.4.0 these are rows of kind `unsupported`: an at-URI that is not a
6860 /// well-formed publication, which no reader can fetch.) Selecting them does not
6861 /// leave the feature dormant — it manufactures a permanent failure per row,
6862 /// which since the cause histogram is *published* as an unreachable
6863 /// publisher. This instance already carries 19 such rows, subscribed before
6864 /// the scheme was refused.
6865 ///
6866 /// They are skipped rather than failed: unsupported is not broken, and the
6867 /// difference is the whole point of recording a cause at all.
6868 #[tokio::test]
6869 async fn an_at_uri_feed_is_never_due_for_polling() -> Result<()> {
6870 let pool = init_url("sqlite::memory:").await?;
6871 for url in [
6872 "https://example.com/feed.xml",
6873 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/app.bsky.feed.post/3lab",
6874 "at://alice.example.com/site.standard.publication/3lab",
6875 ] {
6876 upsert_feed(
6877 &pool,
6878 &NewFeed {
6879 url: url.to_string(),
6880 ..Default::default()
6881 },
6882 )
6883 .await?;
6884 }
6885 // All three have a NULL next_poll, which sorts FIRST — so if at:// were
6886 // selectable at all it would be selected before the http feed.
6887 let due = due_feeds(&pool, "2026-09-20T00:00:00Z", 50).await?;
6888 let urls: Vec<&str> = due.iter().map(|f| f.url.as_str()).collect();
6889 assert_eq!(
6890 urls,
6891 ["https://example.com/feed.xml"],
6892 "an at:// feed was handed to the poller"
6893 );
6894 Ok(())
6895 }
6896
6897 #[tokio::test]
6898 async fn feed_error_count_bumps_and_resets() -> Result<()> {
6899 let pool = init_url("sqlite::memory:").await?;
6900 let url = "https://broken.example/feed.xml";
6901 upsert_feed(
6902 &pool,
6903 &NewFeed {
6904 url: url.to_string(),
6905 ..Default::default()
6906 },
6907 )
6908 .await?;
6909
6910 // A fresh feed starts at 0 errors.
6911 let feed = get_feed_by_url(&pool, url).await?.expect("feed exists");
6912 assert_eq!(feed.consecutive_errors, 0);
6913
6914 // N consecutive failures grow the count 1,2,3, and — fed through
6915 // `backoff_for` — the backoff grows with it (never latched at the floor).
6916 let mut last = std::time::Duration::ZERO;
6917 for expected in 1..=3 {
6918 let count = bump_feed_errors(
6919 &pool,
6920 url,
6921 crate::feed::FailureKind::Fetch,
6922 "connection refused",
6923 )
6924 .await?;
6925 assert_eq!(count, expected, "bump returns the new count");
6926 let backoff = crate::feed::backoff_for(count as u32);
6927 assert!(
6928 backoff >= last,
6929 "backoff must not shrink as errors accumulate"
6930 );
6931 last = backoff;
6932 }
6933 // Growth actually happened (2 errors backs off longer than 1).
6934 assert!(crate::feed::backoff_for(2) > crate::feed::backoff_for(1));
6935 assert_eq!(
6936 get_feed_by_url(&pool, url)
6937 .await?
6938 .unwrap()
6939 .consecutive_errors,
6940 3
6941 );
6942
6943 // A success resets the streak to 0 (back to the normal cadence).
6944 reset_feed_errors(&pool, url).await?;
6945 assert_eq!(
6946 get_feed_by_url(&pool, url)
6947 .await?
6948 .unwrap()
6949 .consecutive_errors,
6950 0
6951 );
6952 Ok(())
6953 }
6954
6955 /// **A recovered feed keeps no reason for having failed.**
6956 ///
6957 /// Added because a mutation found this untested: deleting the
6958 /// `last_error_kind = NULL, last_error = NULL` half of `reset_feed_errors`
6959 /// left the entire suite green. The histogram filters on
6960 /// `consecutive_errors > 0`, so a stale row would not inflate the public
6961 /// count — but anything reading the row directly would be handed a cause
6962 /// that stopped applying, which is the exact failure this column was added
6963 /// to end. A guarantee nothing checks is a comment.
6964 #[tokio::test]
6965 async fn a_successful_poll_clears_the_recorded_failure_reason() -> Result<()> {
6966 let pool = init_url("sqlite::memory:").await?;
6967 let url = "https://recovers.example/feed.xml";
6968 upsert_feed(
6969 &pool,
6970 &NewFeed {
6971 url: url.to_string(),
6972 ..Default::default()
6973 },
6974 )
6975 .await?;
6976
6977 bump_feed_errors(&pool, url, crate::feed::FailureKind::Fetch, "SENTINEL_WHY").await?;
6978 let failing: (Option<String>, Option<String>) =
6979 sqlx::query_as("SELECT last_error_kind, last_error FROM feeds WHERE url = ?1")
6980 .bind(url)
6981 .fetch_one(&pool)
6982 .await?;
6983 assert_eq!(
6984 failing.0.as_deref(),
6985 Some("fetch"),
6986 "the kind was not stored"
6987 );
6988 assert_eq!(
6989 failing.1.as_deref(),
6990 Some("SENTINEL_WHY"),
6991 "the detail was not stored"
6992 );
6993
6994 reset_feed_errors(&pool, url).await?;
6995 let recovered: (Option<String>, Option<String>) =
6996 sqlx::query_as("SELECT last_error_kind, last_error FROM feeds WHERE url = ?1")
6997 .bind(url)
6998 .fetch_one(&pool)
6999 .await?;
7000 assert_eq!(
7001 recovered.0, None,
7002 "a healthy feed still names a failure kind"
7003 );
7004 assert_eq!(
7005 recovered.1, None,
7006 "a healthy feed still carries error detail"
7007 );
7008 Ok(())
7009 }
7010
7011 /// **The closed vocabulary is closed where it is READ, not only written.**
7012 ///
7013 /// `FailureKind::parse` promises that a kind string from a newer build is
7014 /// not "silently attributed to a cause this one recognises" — and the
7015 /// histogram's comment leaned on it. But review found `parse` had zero
7016 /// production callers: `poll_health` handed the raw column to the public
7017 /// template, so an unrecognised string got its own bucket, rendered
7018 /// verbatim. The protection existed only as a doc comment.
7019 ///
7020 /// A row written by a future build must land in `unknown`.
7021 #[tokio::test]
7022 async fn an_unrecognised_failure_kind_folds_into_unknown() -> Result<()> {
7023 let pool = init_url("sqlite::memory:").await?;
7024 for (url, kind) in [
7025 ("https://a.example/f.xml", Some("fetch")),
7026 ("https://b.example/f.xml", Some("quota")), // a newer build's kind
7027 ("https://c.example/f.xml", None), // a legacy row
7028 ] {
7029 upsert_feed(
7030 &pool,
7031 &NewFeed {
7032 url: url.to_string(),
7033 ..Default::default()
7034 },
7035 )
7036 .await?;
7037 sqlx::query(
7038 "UPDATE feeds SET consecutive_errors = 1, last_error_kind = ?2 WHERE url = ?1",
7039 )
7040 .bind(url)
7041 .bind(kind)
7042 .execute(&pool)
7043 .await?;
7044 }
7045 let now = chrono::Utc::now();
7046 let health = poll_health(
7047 &pool,
7048 &now.to_rfc3339_opts(chrono::SecondsFormat::Secs, true),
7049 &(now - chrono::Duration::hours(1)).to_rfc3339_opts(chrono::SecondsFormat::Secs, true),
7050 )
7051 .await?;
7052 let mut kinds = health.failure_kinds.clone();
7053 kinds.sort();
7054 assert_eq!(
7055 kinds,
7056 vec![("fetch".to_string(), 1), ("unknown".to_string(), 2)],
7057 "an unrecognised kind reached the public histogram as its own bucket: {:?}",
7058 health.failure_kinds
7059 );
7060 Ok(())
7061 }
7062
7063 /// **The migration is exercised against a table that predates the columns.**
7064 ///
7065 /// Every other test here builds a fresh database, where `CREATE TABLE`
7066 /// already contains `last_error_kind` / `last_error` — so `ensure_column`,
7067 /// the code path that actually runs against the production volume, was
7068 /// never executed by any of them. A bad `ALTER` would have been found at
7069 /// boot, on the one machine, by crash-looping: `apply_migrations` runs
7070 /// inside `init`, and the entrypoint takes the container down when a child
7071 /// dies.
7072 ///
7073 /// Builds the OLD table shape by hand, puts a failing row in it, migrates,
7074 /// and asserts both that the columns arrive and that the pre-existing row
7075 /// survives with NULLs rather than being rewritten or dropped.
7076 #[tokio::test]
7077 async fn the_last_error_columns_migrate_onto_a_table_that_predates_them() -> Result<()> {
7078 let pool = init_url("sqlite::memory:").await?;
7079
7080 // Drop the current shape and rebuild the pre-migration one.
7081 sqlx::query("DROP TABLE feeds").execute(&pool).await?;
7082 sqlx::query(
7083 "CREATE TABLE feeds (
7084 id INTEGER PRIMARY KEY AUTOINCREMENT,
7085 url TEXT NOT NULL UNIQUE,
7086 title TEXT,
7087 site_url TEXT,
7088 etag TEXT,
7089 last_modified TEXT,
7090 last_polled TEXT,
7091 next_poll TEXT,
7092 consecutive_errors INTEGER NOT NULL DEFAULT 0
7093 )",
7094 )
7095 .execute(&pool)
7096 .await?;
7097 sqlx::query("INSERT INTO feeds (url, consecutive_errors) VALUES (?1, 7)")
7098 .bind("https://legacy.example/feed.xml")
7099 .execute(&pool)
7100 .await?;
7101
7102 apply_migrations(&pool).await?;
7103
7104 // The columns exist...
7105 let cols: Vec<String> = sqlx::query("PRAGMA table_info(feeds)")
7106 .fetch_all(&pool)
7107 .await?
7108 .iter()
7109 .map(|r| r.get::<String, _>("name"))
7110 .collect();
7111 assert!(cols.iter().any(|c| c == "last_error_kind"), "{cols:?}");
7112 assert!(cols.iter().any(|c| c == "last_error"), "{cols:?}");
7113
7114 // ...and the pre-existing row is intact, with no invented cause.
7115 let row: (i64, Option<String>, Option<String>) = sqlx::query_as(
7116 "SELECT consecutive_errors, last_error_kind, last_error FROM feeds WHERE url = ?1",
7117 )
7118 .bind("https://legacy.example/feed.xml")
7119 .fetch_one(&pool)
7120 .await?;
7121 assert_eq!(row.0, 7, "the migration disturbed an existing error count");
7122 assert_eq!(row.1, None, "a legacy row was given a cause it never had");
7123 assert_eq!(row.2, None);
7124
7125 // And it is idempotent — `init` runs this on every boot.
7126 apply_migrations(&pool).await?;
7127 Ok(())
7128 }
7129
7130 /// The stored detail is bounded — it is a remote server's text on an
7131 /// unattended path.
7132 #[tokio::test]
7133 async fn the_stored_error_detail_is_truncated() -> Result<()> {
7134 let pool = init_url("sqlite::memory:").await?;
7135 let url = "https://verbose.example/feed.xml";
7136 upsert_feed(
7137 &pool,
7138 &NewFeed {
7139 url: url.to_string(),
7140 ..Default::default()
7141 },
7142 )
7143 .await?;
7144 bump_feed_errors(
7145 &pool,
7146 url,
7147 crate::feed::FailureKind::Body,
7148 &"x".repeat(10_000),
7149 )
7150 .await?;
7151 let stored: (Option<String>,) =
7152 sqlx::query_as("SELECT last_error FROM feeds WHERE url = ?1")
7153 .bind(url)
7154 .fetch_one(&pool)
7155 .await?;
7156 assert_eq!(stored.0.unwrap().chars().count(), MAX_ERROR_DETAIL_CHARS);
7157 Ok(())
7158 }
7159
7160 // -- F3: db_size_bytes ignores freed pages and drops after reclaim -------
7161
7162 /// A new on-disk database must be created in INCREMENTAL mode.
7163 ///
7164 /// This is the whole fix for new instances: `auto_vacuum` was read by
7165 /// `reclaim` and set nowhere, so every database ran in NONE and `reclaim`
7166 /// always took its full-`VACUUM` branch — the one that cannot complete on a
7167 /// volume under the pressure that triggered the sweep. The pragma only binds
7168 /// on a database with no tables yet, so "at creation" is the load-bearing
7169 /// part, not "somewhere in init".
7170 #[tokio::test]
7171 async fn a_new_database_is_created_in_incremental_vacuum_mode() -> Result<()> {
7172 let dir = std::env::temp_dir();
7173 let path = dir.join(format!("fr-autovac-{}.db", std::process::id()));
7174 for p in [
7175 path.display().to_string(),
7176 format!("{}-wal", path.display()),
7177 format!("{}-shm", path.display()),
7178 ] {
7179 std::fs::remove_file(&p).ok();
7180 }
7181 let pool = init_url(&format!("sqlite://{}", path.display())).await?;
7182
7183 assert_eq!(
7184 auto_vacuum_mode(&pool).await?,
7185 AutoVacuum::Incremental,
7186 "a fresh database is still in the mode where reclaim needs a full VACUUM"
7187 );
7188 // And the WAL is bounded rather than growing to its high-water mark
7189 // forever.
7190 let limit: i64 = sqlx::query_scalar("PRAGMA journal_size_limit")
7191 .fetch_one(&pool)
7192 .await?;
7193 assert_eq!(
7194 limit, WAL_SIZE_LIMIT_BYTES,
7195 "journal_size_limit not applied"
7196 );
7197
7198 // Being INCREMENTAL, the migration is a no-op — which is what makes the
7199 // flag safe for an operator to run without checking first.
7200 assert_eq!(
7201 migrate_to_incremental_vacuum(&pool, None).await?,
7202 VacuumMigration::NotNeeded(AutoVacuum::Incremental)
7203 );
7204
7205 pool.close().await;
7206 for p in [
7207 path.display().to_string(),
7208 format!("{}-wal", path.display()),
7209 format!("{}-shm", path.display()),
7210 ] {
7211 std::fs::remove_file(&p).ok();
7212 }
7213 Ok(())
7214 }
7215
7216 /// The migration refuses itself when the volume cannot hold the rebuild.
7217 ///
7218 /// A full `VACUUM` writes a complete second copy, so attempting one without
7219 /// headroom burns I/O on a box that has none and finishes nothing. Refusing
7220 /// is the entire reason this is an operator step rather than something
7221 /// `reclaim` does on its own.
7222 #[tokio::test]
7223 async fn the_vacuum_migration_refuses_without_headroom() -> Result<()> {
7224 let dir = std::env::temp_dir();
7225 let path = dir.join(format!("fr-autovac-none-{}.db", std::process::id()));
7226 for p in [
7227 path.display().to_string(),
7228 format!("{}-wal", path.display()),
7229 format!("{}-shm", path.display()),
7230 ] {
7231 std::fs::remove_file(&p).ok();
7232 }
7233 // Build a database the way one that predates this change looks: create
7234 // the file in NONE mode explicitly, then populate it.
7235 let url = format!("sqlite://{}", path.display());
7236 let opts = SqliteConnectOptions::from_str(&url)?
7237 .create_if_missing(true)
7238 .foreign_keys(true)
7239 .journal_mode(sqlx::sqlite::SqliteJournalMode::Wal)
7240 .auto_vacuum(sqlx::sqlite::SqliteAutoVacuum::None);
7241 let pool = SqlitePoolOptions::new()
7242 .min_connections(1)
7243 .max_connections(1)
7244 .connect_with(opts)
7245 .await?;
7246 init_schema(&pool).await?;
7247 assert_eq!(auto_vacuum_mode(&pool).await?, AutoVacuum::None);
7248
7249 // Zero free space: refused, and the mode is untouched.
7250 let refused = migrate_to_incremental_vacuum(&pool, Some(0)).await?;
7251 assert!(
7252 matches!(refused, VacuumMigration::RefusedNoHeadroom { .. }),
7253 "expected a refusal, got {refused:?}"
7254 );
7255 assert_eq!(
7256 auto_vacuum_mode(&pool).await?,
7257 AutoVacuum::None,
7258 "a refused migration must not have changed the mode"
7259 );
7260
7261 // With headroom it runs, and the database ends up INCREMENTAL — which is
7262 // what makes `reclaim` cheap from then on.
7263 let done = migrate_to_incremental_vacuum(&pool, Some(u64::MAX)).await?;
7264 let VacuumMigration::Migrated {
7265 bytes_after,
7266 file_after,
7267 ..
7268 } = done
7269 else {
7270 panic!("expected a migration, got {done:?}");
7271 };
7272 assert_eq!(auto_vacuum_mode(&pool).await?, AutoVacuum::Incremental);
7273 // The reported size must not include the WAL the VACUUM just filled. In
7274 // WAL mode a VACUUM writes the whole rebuilt database through the WAL,
7275 // so without the truncating checkpoint this reads as roughly double —
7276 // "the migration doubled my database", from the one line the command
7277 // prints.
7278 let file_after = file_after.expect("an on-disk database has a file size") as i64;
7279 assert!(
7280 bytes_after <= file_after * 2,
7281 "bytes_after ({bytes_after}) is inflated by an untruncated WAL against a \
7282 {file_after}-byte file"
7283 );
7284
7285 pool.close().await;
7286 for p in [
7287 path.display().to_string(),
7288 format!("{}-wal", path.display()),
7289 format!("{}-shm", path.display()),
7290 ] {
7291 std::fs::remove_file(&p).ok();
7292 }
7293 Ok(())
7294 }
7295
7296 /// **R6 benchmark: what the retention sweep actually costs, and what fixes it.**
7297 ///
7298 /// `#[ignore]` — builds a ~1M-row database once per shape per scale (ten
7299 /// times), so it is a measurement tool rather than a test. Run with:
7300 ///
7301 /// ```text
7302 /// cargo test --lib -- --ignored --nocapture r6_measure_retention_sweep
7303 /// ```
7304 ///
7305 /// It exists because R6 was "every delete batch re-scans `entry_state`" and
7306 /// the honest answer was "measure before changing an index". Kept so the next
7307 /// candidate index can be tried against the same fixture rather than a new
7308 /// one. Findings are recorded in `design/REVIEW-ROUND-2.md`.
7309 #[tokio::test]
7310 #[ignore]
7311 async fn r6_measure_retention_sweep() -> Result<()> {
7312 const FEEDS: i64 = 500;
7313 const PER_FEED: i64 = 2_000; // matches `max_entries_per_feed`
7314 const PINNED: i64 = 50_000; // entry_state rows a reader has touched
7315
7316 /// Build the fixture, apply `extra_indexes`, then plan and time a sweep.
7317 async fn run(
7318 label: &str,
7319 extra_indexes: &[&str],
7320 pinned: i64,
7321 old_list_form: bool,
7322 ) -> Result<()> {
7323 let dir = std::env::temp_dir();
7324 let path = dir.join(format!("fr-r6-{}-{label}.db", std::process::id()));
7325 // RAII, because every `?` between here and the end used to leak a
7326 // 1M-row fixture plus its -wal/-shm into the temp dir — six per run.
7327 struct Fixture(std::path::PathBuf);
7328 impl Fixture {
7329 fn wipe(&self) {
7330 for p in [
7331 self.0.display().to_string(),
7332 format!("{}-wal", self.0.display()),
7333 format!("{}-shm", self.0.display()),
7334 ] {
7335 std::fs::remove_file(&p).ok();
7336 }
7337 }
7338 }
7339 impl Drop for Fixture {
7340 fn drop(&mut self) {
7341 self.wipe();
7342 }
7343 }
7344 let fixture = Fixture(path.clone());
7345 fixture.wipe();
7346 let pool = init_url(&format!("sqlite://{}", path.display())).await?;
7347
7348 // Bulk-build with SQL: a million round trips would measure the
7349 // fixture, not the sweep. Recursive CTE because `generate_series` is
7350 // not compiled into the bundled SQLite.
7351 sqlx::query(
7352 "WITH RECURSIVE n(value) AS ( \
7353 SELECT 1 UNION ALL SELECT value + 1 FROM n WHERE value < ?1 \
7354 ) \
7355 INSERT INTO feeds (url) \
7356 SELECT 'https://f' || value || '.example/x.xml' FROM n",
7357 )
7358 .bind(FEEDS)
7359 .execute(&pool)
7360 .await
7361 .context("seeding feeds")?;
7362
7363 // Half the entries older than the window, half inside it.
7364 sqlx::query(
7365 "WITH RECURSIVE n(value) AS ( \
7366 SELECT 1 UNION ALL SELECT value + 1 FROM n WHERE value < ?1 \
7367 ) \
7368 INSERT INTO entries (feed_id, guid, title, published, fetched_at) \
7369 SELECT f.id, \
7370 'g' || f.id || '-' || s.value, \
7371 'Entry ' || s.value, \
7372 CASE WHEN s.value % 2 = 0 THEN '2020-01-01T00:00:00Z' \
7373 ELSE '2099-01-01T00:00:00Z' END, \
7374 '2026-01-01T00:00:00Z' \
7375 FROM feeds f, n s",
7376 )
7377 .bind(PER_FEED)
7378 .execute(&pool)
7379 .await?;
7380
7381 // **A REALISTIC pin distribution, which the first version did not
7382 // have.** It made every row `read=0,starred=0` or `read=1,starred=1`,
7383 // so 100% of `entry_state` matched `starred = 1 OR read = 0` — there
7384 // were no "read and not starred" rows at all, which is the commonest
7385 // state a reader leaves behind. That mattered: a PARTIAL index on the
7386 // pinned predicate then covers the whole table and cannot be
7387 // selective, so measuring one against that fixture measures nothing.
7388 //
7389 // 90% read-and-unstarred (evictable), 10% pinned, split between
7390 // starred and unread.
7391 sqlx::query(
7392 "INSERT INTO entry_state (did, entry_id, read, starred, updated_at) \
7393 SELECT 'did:plc:reader', id, \
7394 CASE WHEN id % 10 <> 0 THEN 1 \
7395 WHEN id % 20 = 0 THEN 1 ELSE 0 END, \
7396 CASE WHEN id % 10 <> 0 THEN 0 \
7397 WHEN id % 20 = 0 THEN 1 ELSE 0 END, \
7398 '2026-01-01T00:00:00Z' \
7399 FROM entries LIMIT ?1",
7400 )
7401 .bind(pinned)
7402 .execute(&pool)
7403 .await?;
7404
7405 // Space is the other half of the trade: this is a 1 GB volume with a
7406 // 768 MiB watermark, so an index that buys time and costs disk can be
7407 // a net loss.
7408 let pages_before: i64 = sqlx::query_scalar("PRAGMA page_count")
7409 .fetch_one(&pool)
7410 .await?;
7411 let page_size: i64 = sqlx::query_scalar("PRAGMA page_size")
7412 .fetch_one(&pool)
7413 .await?;
7414 for idx in extra_indexes {
7415 sqlx::query(sqlx::AssertSqlSafe((*idx).to_string()))
7416 .execute(&pool)
7417 .await
7418 .with_context(|| format!("creating {idx}"))?;
7419 }
7420 let pages_after: i64 = sqlx::query_scalar("PRAGMA page_count")
7421 .fetch_one(&pool)
7422 .await?;
7423 let index_bytes = (pages_after - pages_before) * page_size;
7424
7425 // What the index costs on the WRITE path — the poller inserts
7426 // constantly, the sweep runs once a day.
7427 let t_ins = std::time::Instant::now();
7428 sqlx::query(
7429 "WITH RECURSIVE n(value) AS ( \
7430 SELECT 1 UNION ALL SELECT value + 1 FROM n WHERE value < 10000 \
7431 ) \
7432 INSERT INTO entries (feed_id, guid, published, fetched_at) \
7433 SELECT 1, 'ins-' || value, '2099-06-01T00:00:00Z', '2026-01-01T00:00:00Z' \
7434 FROM n",
7435 )
7436 .execute(&pool)
7437 .await?;
7438 let insert_10k = t_ins.elapsed();
7439
7440 // Give the planner statistics, as a long-lived instance would have.
7441 sqlx::query("ANALYZE").execute(&pool).await?;
7442
7443 // The plan must describe the query this run actually TIMES. It used
7444 // to be hardcoded to the `NOT IN` form regardless, so four of six
7445 // runs printed a plan for a different query than the one measured —
7446 // in the artifact kept precisely to be the evidence.
7447 let planned = if old_list_form {
7448 "EXPLAIN QUERY PLAN SELECT id FROM entries \
7449 WHERE COALESCE(published, fetched_at) < '2026-06-01T00:00:00Z' \
7450 AND id NOT IN (SELECT entry_id FROM entry_state \
7451 WHERE starred = 1 OR read = 0) \
7452 LIMIT 1000"
7453 } else {
7454 "EXPLAIN QUERY PLAN SELECT e.id FROM entries e \
7455 WHERE COALESCE(e.published, e.fetched_at) < '2026-06-01T00:00:00Z' \
7456 AND NOT EXISTS (SELECT 1 FROM entry_state s \
7457 WHERE s.entry_id = e.id \
7458 AND (s.starred = 1 OR s.read = 0)) \
7459 LIMIT 1000"
7460 };
7461 let plan: Vec<String> = sqlx::query(sqlx::AssertSqlSafe(planned))
7462 .fetch_all(&pool)
7463 .await?
7464 .into_iter()
7465 .map(|r| r.get::<String, _>("detail"))
7466 .collect();
7467
7468 // ONE variant per fixture — running both against the same database
7469 // measured the second against an already-emptied table, which
7470 // reported a 0-row "win" the first time this was written.
7471 let cutoff = (chrono::Utc::now() - chrono::Duration::days(30))
7472 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
7473 let t = std::time::Instant::now();
7474 let deleted = if old_list_form {
7475 // The shape `prune_old_entries` used to have: the pinned set as
7476 // an `IN` list, re-materialised on every batch.
7477 let mut n = 0u64;
7478 loop {
7479 let got = sqlx::query(
7480 "DELETE FROM entries WHERE id IN ( \
7481 SELECT id FROM entries \
7482 WHERE COALESCE(published, fetched_at) < ?1 \
7483 AND id NOT IN ( \
7484 SELECT entry_id FROM entry_state \
7485 WHERE starred = 1 OR read = 0 \
7486 ) \
7487 LIMIT 1000)",
7488 )
7489 .bind(&cutoff)
7490 .execute(&pool)
7491 .await?
7492 .rows_affected();
7493 n += got;
7494 if got == 0 {
7495 break;
7496 }
7497 tokio::time::sleep(std::time::Duration::from_millis(10)).await;
7498 }
7499 n
7500 } else {
7501 // NOTE the arms are not identical work: this one goes through the
7502 // real `prune_old_entries`, which also runs the hard-ceiling pass
7503 // and the cursor scrub. The bias therefore runs AGAINST the
7504 // shipped form, so a win measured here is a lower bound — but the
7505 // two numbers are not a like-for-like microbenchmark.
7506 prune_old_entries(&pool, 30, 3650, 0).await?
7507 };
7508 let elapsed = t.elapsed();
7509
7510 // State the fixture's shape, so a future reader cannot mistake a
7511 // degenerate distribution for a representative one again.
7512 let matching: i64 = sqlx::query_scalar(
7513 "SELECT COUNT(*) FROM entry_state WHERE starred = 1 OR read = 0",
7514 )
7515 .fetch_one(&pool)
7516 .await?;
7517 println!("\n=== {label} (entry_state = {pinned}, pinned = {matching}) ===");
7518 println!(
7519 " index cost: {:.1} MiB on disk, 10k inserts in {insert_10k:?}",
7520 index_bytes as f64 / 1024.0 / 1024.0
7521 );
7522 for l in &plan {
7523 println!(" plan: {l}");
7524 }
7525 println!(
7526 " deleted {deleted} in {elapsed:?} ({:?}/batch)",
7527 elapsed / (deleted as u32 / PRUNE_BATCH as u32).max(1)
7528 );
7529
7530 pool.close().await;
7531 drop(fixture);
7532 Ok(())
7533 }
7534
7535 // R6's own hypothesis was that the per-batch `entry_state` scan is the
7536 // cost. Both scales are measured because that scan grows with TOTAL
7537 // users, not with the feed being swept — 50k is one active reader,
7538 // 600k is the figure the schema comment cites as realistic.
7539 const AGE_IDX: &str =
7540 "CREATE INDEX idx_entries_age ON entries(COALESCE(published, fetched_at))";
7541 // The index R6 actually asked for. Its row is the one the rejection
7542 // turns on — "changes the plan, changes the time by nothing" — and an
7543 // earlier version of this benchmark dropped it, leaving that claim
7544 // resting on prose while the artifact kept to prove it could not.
7545 const PINNED_IDX: &str = "CREATE INDEX idx_es_pinned ON entry_state(entry_id) \
7546 WHERE starred = 1 OR read = 0";
7547 for pinned in [PINNED, 600_000] {
7548 // `false` = the shipped `prune_old_entries`, whatever shape it
7549 // currently uses; `true` = the raw `NOT IN` list form it replaced,
7550 // kept so the regression stays measurable rather than remembered.
7551 run("as shipped (NOT EXISTS)", &[], pinned, false).await?;
7552 run("old NOT IN list form", &[], pinned, true).await?;
7553 run("old NOT IN + pinned index", &[PINNED_IDX], pinned, true).await?;
7554 // The row that was never measured: the pinned index against the
7555 // query that SHIPPED, rather than against the one being deleted.
7556 // Rejecting it on the strength of the latter was the error.
7557 run("as shipped + pinned index", &[PINNED_IDX], pinned, false).await?;
7558 run("as shipped + age index", &[AGE_IDX], pinned, false).await?;
7559 }
7560 Ok(())
7561 }
7562
7563 /// **The migration must not ask the pool for anything while holding a
7564 /// connection.** A single-connection pool is always saturated, so any such
7565 /// call stalls for the full acquire timeout.
7566 ///
7567 /// This has now been introduced twice — once by acquiring a connection for
7568 /// the pragma pair, and once by resolving the temp directory inside that
7569 /// block. The second was worse than a stall: `main_db_path` swallows errors
7570 /// into `None`, so it waited 30 s and then silently skipped the pragma it
7571 /// existed to set. A wall-clock assertion is crude, but it is the only thing
7572 /// that distinguishes "works" from "works after a 30-second timeout".
7573 #[tokio::test]
7574 async fn the_vacuum_migration_never_waits_on_its_own_pool() -> Result<()> {
7575 let dir = std::env::temp_dir();
7576 let path = dir.join(format!("fr-nodeadlock-{}.db", std::process::id()));
7577 for p in [
7578 path.display().to_string(),
7579 format!("{}-wal", path.display()),
7580 format!("{}-shm", path.display()),
7581 ] {
7582 std::fs::remove_file(&p).ok();
7583 }
7584 let url = format!("sqlite://{}", path.display());
7585 let opts = SqliteConnectOptions::from_str(&url)?
7586 .create_if_missing(true)
7587 .foreign_keys(true)
7588 .journal_mode(sqlx::sqlite::SqliteJournalMode::Wal)
7589 .auto_vacuum(sqlx::sqlite::SqliteAutoVacuum::None);
7590 // ONE connection: any pool call made while the migration holds it will
7591 // block until the acquire timeout rather than deadlocking forever.
7592 let pool = SqlitePoolOptions::new()
7593 .min_connections(1)
7594 .max_connections(1)
7595 .connect_with(opts)
7596 .await?;
7597 init_schema(&pool).await?;
7598
7599 let t0 = std::time::Instant::now();
7600 let outcome = migrate_to_incremental_vacuum(&pool, Some(u64::MAX)).await?;
7601 let elapsed = t0.elapsed();
7602
7603 assert!(
7604 matches!(outcome, VacuumMigration::Migrated { .. }),
7605 "expected a migration, got {outcome:?}"
7606 );
7607 assert!(
7608 elapsed < std::time::Duration::from_secs(5),
7609 "the migration took {elapsed:?} on an empty database — it is waiting on \
7610 its own pool while holding a connection"
7611 );
7612
7613 pool.close().await;
7614 for p in [
7615 path.display().to_string(),
7616 format!("{}-wal", path.display()),
7617 format!("{}-shm", path.display()),
7618 ] {
7619 std::fs::remove_file(&p).ok();
7620 }
7621 Ok(())
7622 }
7623
7624 /// `reclaim` must NOT run a full VACUUM in NONE mode — the branch that used
7625 /// to be the only one that ever executed, and the one that cannot finish on
7626 /// a volume under the pressure that triggers a sweep.
7627 ///
7628 /// Observable without timing a VACUUM: a full VACUUM returns freed pages to
7629 /// the OS, so `page_count` falls. Skipping it leaves the allocation in
7630 /// place — while `db_size_bytes`, which subtracts the freelist, still drops.
7631 /// That pairing is the actual claim: the watermark does not latch even
7632 /// though the file does not shrink.
7633 #[tokio::test]
7634 async fn reclaim_does_not_full_vacuum_in_none_mode() -> Result<()> {
7635 let dir = std::env::temp_dir();
7636 let path = dir.join(format!("fr-noneclaim-{}.db", std::process::id()));
7637 for p in [
7638 path.display().to_string(),
7639 format!("{}-wal", path.display()),
7640 format!("{}-shm", path.display()),
7641 ] {
7642 std::fs::remove_file(&p).ok();
7643 }
7644 let url = format!("sqlite://{}", path.display());
7645 let opts = SqliteConnectOptions::from_str(&url)?
7646 .create_if_missing(true)
7647 .foreign_keys(true)
7648 .journal_mode(sqlx::sqlite::SqliteJournalMode::Wal)
7649 .auto_vacuum(sqlx::sqlite::SqliteAutoVacuum::None);
7650 let pool = SqlitePoolOptions::new()
7651 .min_connections(1)
7652 .max_connections(1)
7653 .connect_with(opts)
7654 .await?;
7655 init_schema(&pool).await?;
7656
7657 let feed_id = upsert_feed(
7658 &pool,
7659 &NewFeed {
7660 url: "https://none.example/f.xml".to_string(),
7661 ..Default::default()
7662 },
7663 )
7664 .await?;
7665 let entries: Vec<NewEntry> = (0..1500)
7666 .map(|i| NewEntry {
7667 guid: format!("n-{i}"),
7668 content_html: Some("x".repeat(800)),
7669 ..Default::default()
7670 })
7671 .collect();
7672 insert_entries(&pool, feed_id, &entries, 0).await?;
7673 // Fold the WAL in so the "full" baseline is file pages, not WAL churn.
7674 sqlx::query("PRAGMA wal_checkpoint(TRUNCATE)")
7675 .execute(&pool)
7676 .await?;
7677 let used_full = db_size_bytes(&pool).await?;
7678
7679 sqlx::query("DELETE FROM entries").execute(&pool).await?;
7680 let pages_before: i64 = sqlx::query_scalar("PRAGMA page_count")
7681 .fetch_one(&pool)
7682 .await?;
7683
7684 reclaim(&pool).await?;
7685
7686 let pages_after: i64 = sqlx::query_scalar("PRAGMA page_count")
7687 .fetch_one(&pool)
7688 .await?;
7689 assert_eq!(
7690 pages_after, pages_before,
7691 "reclaim shrank the file in NONE mode, so it ran the full VACUUM this \
7692 branch exists to avoid"
7693 );
7694 // …and the watermark still falls, which is what makes skipping safe.
7695 // `db_size_bytes` subtracts the freelist, so the delete alone lowers it
7696 // even though the file kept every page it had allocated.
7697 let used_after = db_size_bytes(&pool).await?;
7698 assert!(
7699 used_after < used_full,
7700 "used size did not fall after the delete ({used_after} !< {used_full}); \
7701 without a VACUUM the DB-size watermark would latch the poller off"
7702 );
7703
7704 pool.close().await;
7705 for p in [
7706 path.display().to_string(),
7707 format!("{}-wal", path.display()),
7708 format!("{}-shm", path.display()),
7709 ] {
7710 std::fs::remove_file(&p).ok();
7711 }
7712 Ok(())
7713 }
7714
7715 #[tokio::test]
7716 async fn db_size_drops_after_prune_and_reclaim() -> Result<()> {
7717 // On-disk DB so VACUUM has a file to shrink (in-memory has no freelist to
7718 // speak of the same way). Temp path, cleaned up at the end.
7719 let dir = std::env::temp_dir();
7720 let path = dir.join(format!("fr-reclaim-{}.db", std::process::id()));
7721 let url = format!("sqlite://{}", path.display());
7722 let pool = init_url(&url).await?;
7723
7724 let feed_id = upsert_feed(
7725 &pool,
7726 &NewFeed {
7727 url: "https://bulk.example/feed.xml".to_string(),
7728 ..Default::default()
7729 },
7730 )
7731 .await?;
7732
7733 // Insert a large batch so the file allocates real pages.
7734 let entries: Vec<NewEntry> = (0..2000)
7735 .map(|i| NewEntry {
7736 guid: format!("guid-{i}"),
7737 title: Some(format!("Entry number {i} with some padding text")),
7738 content_html: Some("<p>".to_string() + &"x".repeat(400) + "</p>"),
7739 published: Some("2026-01-01T00:00:00Z".to_string()),
7740 ..Default::default()
7741 })
7742 .collect();
7743 insert_entries(&pool, feed_id, &entries, 0).await?;
7744 let full = db_size_bytes(&pool).await?;
7745 assert!(full > 0);
7746
7747 // Prune: delete every entry (the retention sweep's effect). This frees
7748 // pages onto the freelist but does NOT shrink the file yet.
7749 sqlx::query("DELETE FROM entries WHERE feed_id = ?1")
7750 .bind(feed_id)
7751 .execute(&pool)
7752 .await?;
7753
7754 // Because db_size_bytes subtracts freelist pages, the USED size already
7755 // reflects the delete even before the file shrinks.
7756 let after_delete = db_size_bytes(&pool).await?;
7757 assert!(
7758 after_delete < full,
7759 "used size must drop once rows are deleted (freed pages excluded): \
7760 {after_delete} !< {full}"
7761 );
7762
7763 // Reclaim returns the freed pages to the OS; used size stays low (and the
7764 // file itself shrinks). The key property F3 needs: the watermark can now
7765 // fall back below its threshold instead of latching polling off.
7766 reclaim(&pool).await?;
7767 let after_reclaim = db_size_bytes(&pool).await?;
7768 assert!(
7769 after_reclaim <= after_delete,
7770 "reclaim must not grow used size: {after_reclaim} !<= {after_delete}"
7771 );
7772 assert!(
7773 after_reclaim < full,
7774 "after prune+reclaim the DB is smaller than when full: \
7775 {after_reclaim} !< {full}"
7776 );
7777
7778 drop(pool);
7779 let _ = std::fs::remove_file(&path);
7780 let _ = std::fs::remove_file(format!("{}-wal", path.display()));
7781 let _ = std::fs::remove_file(format!("{}-shm", path.display()));
7782 Ok(())
7783 }
7784
7785 // -- F4 support: pds_created flag round-trips + flips ---------------------
7786
7787 #[tokio::test]
7788 async fn cursor_pds_created_defaults_false_and_flips() -> Result<()> {
7789 let pool = init_url("sqlite::memory:").await?;
7790 let did = "did:plc:f4";
7791 let feed_url = "https://example.com/feed.xml";
7792 upsert_cursor(
7793 &pool,
7794 &ReadCursor {
7795 did: did.to_string(),
7796 feed_url: feed_url.to_string(),
7797 read_through: None,
7798 read_ids: r#"["1"]"#.to_string(),
7799 unread_ids: "[]".to_string(),
7800 dirty: true,
7801 pds_created: false,
7802 updated_at: now_rfc3339(),
7803 },
7804 )
7805 .await?;
7806
7807 // A brand-new cursor's PDS record does NOT yet exist.
7808 let c = get_cursor(&pool, did, feed_url).await?.unwrap();
7809 assert!(!c.pds_created, "first flush must emit a create, not update");
7810
7811 // Two bystanders: the same DID on another feed, another DID on the same
7812 // feed. **The UPDATE must be scoped to exactly one row.** With its WHERE
7813 // clause deleted this test still passed — it seeded one cursor, so
7814 // "every row" and "this row" were the same row. Unscoped, every DID's
7815 // every cursor is flagged as created, their readState records are never
7816 // created, and every later flush emits `update` against nothing.
7817 for (d, f) in [
7818 (did, "https://other.example/feed.xml"),
7819 ("did:plc:other", feed_url),
7820 ] {
7821 upsert_cursor(
7822 &pool,
7823 &ReadCursor {
7824 did: d.to_string(),
7825 feed_url: f.to_string(),
7826 read_through: None,
7827 read_ids: "[]".to_string(),
7828 unread_ids: "[]".to_string(),
7829 dirty: false,
7830 pds_created: false,
7831 updated_at: now_rfc3339(),
7832 },
7833 )
7834 .await?;
7835 }
7836
7837 // After the create-flush lands, the flag flips so future flushes update.
7838 mark_cursor_pds_created(&pool, did, feed_url).await?;
7839 let c = get_cursor(&pool, did, feed_url).await?.unwrap();
7840 assert!(c.pds_created);
7841 for (d, f) in [
7842 (did, "https://other.example/feed.xml"),
7843 ("did:plc:other", feed_url),
7844 ] {
7845 let bystander = get_cursor(&pool, d, f).await?.unwrap();
7846 assert!(
7847 !bystander.pds_created,
7848 "marking ({did}, {feed_url}) also flagged ({d}, {f})"
7849 );
7850 }
7851 Ok(())
7852 }
7853
7854 // -- STORAGE HYGIENE: retention prune + orphan-id scrub -------------------
7855
7856 /// Count entries currently in the cache.
7857 async fn count_entries(pool: &SqlitePool) -> Result<i64> {
7858 Ok(sqlx::query_scalar::<_, i64>("SELECT COUNT(*) FROM entries")
7859 .fetch_one(pool)
7860 .await?)
7861 }
7862
7863 /// **The rolling window and the hard ceiling do not touch a publication, and
7864 /// this is the test that says the feature works at all.**
7865 ///
7866 /// Measured on 2026-09-27 against three real publications: the newest
7867 /// document Standard.site offered was 131 days old, Annotated's 109, minus
7868 /// listens' 241. Under the 14-day window every one of them stored **zero**
7869 /// rows — a successful poll and an empty feed. So age is not the policy here;
7870 /// COUNT is (`max_entries_per_feed`), and the ceiling below is only the
7871 /// not-immortal backstop.
7872 ///
7873 /// Both directions in one test on purpose: the RSS twin must still be
7874 /// deleted, or "nothing is ever swept" would pass.
7875 #[tokio::test]
7876 async fn the_window_and_the_ceiling_spare_a_publication_but_not_an_rss_entry() -> Result<()> {
7877 let pool = init_url("sqlite::memory:").await?;
7878 let rss = upsert_feed(
7879 &pool,
7880 &NewFeed {
7881 url: "https://aged.example/feed.xml".to_string(),
7882 ..Default::default()
7883 },
7884 )
7885 .await?;
7886 let publication = upsert_feed(
7887 &pool,
7888 &NewFeed {
7889 url: "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab"
7890 .to_string(),
7891 ..Default::default()
7892 },
7893 )
7894 .await?;
7895 // The kind column is what the sweep filters on, so assert the fixture
7896 // really produced two different kinds rather than trusting `FeedKind::of`.
7897 let kinds: Vec<String> = sqlx::query_scalar("SELECT kind FROM feeds ORDER BY id")
7898 .fetch_all(&pool)
7899 .await?;
7900 assert_eq!(kinds, vec!["rss".to_string(), "publication".to_string()]);
7901
7902 // A year old, and READ by somebody — so the window's own sparing rule
7903 // ("starred or unread survives") cannot be what keeps either row.
7904 let ancient = (chrono::Utc::now() - chrono::Duration::days(365))
7905 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
7906 for feed_id in [rss, publication] {
7907 insert_entries(
7908 &pool,
7909 feed_id,
7910 &[NewEntry {
7911 guid: format!("ancient-{feed_id}"),
7912 published: Some(ancient.clone()),
7913 fetched_at: Some(ancient.clone()),
7914 ..Default::default()
7915 }],
7916 0,
7917 )
7918 .await?;
7919 }
7920 replace_sub_refs(&pool, "did:plc:reader", &[rss, publication]).await?;
7921 for id in sqlx::query_scalar::<_, i64>("SELECT id FROM entries ORDER BY id")
7922 .fetch_all(&pool)
7923 .await?
7924 {
7925 mark_read(&pool, "did:plc:reader", id, true).await?;
7926 }
7927 assert_eq!(count_entries(&pool).await?, 2);
7928
7929 // **The shipped configuration, all three knobs at their defaults.** An
7930 // earlier version of this test passed `0` for the archive ceiling, so the
7931 // combination under test was not the one any instance runs; at 3650 the
7932 // publication's year-old document is inside the ceiling and must still
7933 // survive.
7934 let deleted = prune_old_entries(&pool, 14, 180, 3_650).await?;
7935 assert_eq!(deleted, 1, "exactly one of the two should have gone");
7936 let surviving: Vec<i64> = sqlx::query_scalar("SELECT feed_id FROM entries")
7937 .fetch_all(&pool)
7938 .await?;
7939 assert_eq!(
7940 surviving,
7941 vec![publication],
7942 "the publication's year-old document was swept — under the 14-day \
7943 window that is every document a real publication has, so the feed a \
7944 reader subscribed to would be permanently empty",
7945 );
7946 Ok(())
7947 }
7948
7949 /// **"Not aged out" must not mean "immortal".**
7950 ///
7951 /// The per-feed trim is what bounds a publication, and it only runs when a
7952 /// poll stores something — so entries of a feed nobody polls any more have
7953 /// nothing else to reap them. This ceiling is that backstop, and it spares
7954 /// nothing, for the same reason the hard ceiling spares nothing: a saved
7955 /// record whose entry is gone still renders from the PDS record as a link.
7956 #[tokio::test]
7957 async fn the_archive_ceiling_reaps_a_publication_entry_past_it() -> Result<()> {
7958 let pool = init_url("sqlite::memory:").await?;
7959 // An RSS twin, to pin that this pass is SCOPED. Verified needed: dropping
7960 // the `kind NOT IN` clause from it left all 909 tests passing, and that
7961 // mutation quietly re-enables age-based eviction for RSS on an instance
7962 // whose operator set both RSS knobs to zero.
7963 let rss = upsert_feed(
7964 &pool,
7965 &NewFeed {
7966 url: "https://not-swept.example/feed.xml".to_string(),
7967 ..Default::default()
7968 },
7969 )
7970 .await?;
7971 let publication = upsert_feed(
7972 &pool,
7973 &NewFeed {
7974 url: "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab"
7975 .to_string(),
7976 ..Default::default()
7977 },
7978 )
7979 .await?;
7980 let ancient = (chrono::Utc::now() - chrono::Duration::days(400))
7981 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
7982 let recent = now_rfc3339();
7983 insert_entries(
7984 &pool,
7985 publication,
7986 &[
7987 NewEntry {
7988 guid: "past-the-ceiling".into(),
7989 published: Some(ancient.clone()),
7990 fetched_at: Some(ancient),
7991 ..Default::default()
7992 },
7993 NewEntry {
7994 guid: "inside-the-ceiling".into(),
7995 published: Some(recent.clone()),
7996 fetched_at: Some(recent),
7997 ..Default::default()
7998 },
7999 ],
8000 0,
8001 )
8002 .await?;
8003 let long_ago = (chrono::Utc::now() - chrono::Duration::days(400))
8004 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
8005 insert_entries(
8006 &pool,
8007 rss,
8008 &[NewEntry {
8009 guid: "rss-past-the-archive-ceiling".into(),
8010 published: Some(long_ago.clone()),
8011 fetched_at: Some(long_ago),
8012 ..Default::default()
8013 }],
8014 0,
8015 )
8016 .await?;
8017
8018 // STARRED, so this also pins that the ceiling spares nothing.
8019 replace_sub_refs(&pool, "did:plc:reader", &[rss, publication]).await?;
8020 for id in sqlx::query_scalar::<_, i64>("SELECT id FROM entries ORDER BY id")
8021 .fetch_all(&pool)
8022 .await?
8023 {
8024 mark_starred(&pool, "did:plc:reader", id, true).await?;
8025 }
8026
8027 // Rolling window and hard ceiling off: the archive ceiling is the only
8028 // thing that can delete here.
8029 let deleted = prune_old_entries(&pool, 0, 0, 365).await?;
8030 assert_eq!(
8031 deleted, 1,
8032 "the entry past the archive ceiling was not reaped"
8033 );
8034 let mut guids: Vec<String> = sqlx::query_scalar("SELECT guid FROM entries")
8035 .fetch_all(&pool)
8036 .await?;
8037 guids.sort();
8038 assert_eq!(
8039 guids,
8040 vec![
8041 "inside-the-ceiling".to_string(),
8042 "rss-past-the-archive-ceiling".to_string(),
8043 ],
8044 "the archive ceiling must reap the publication's over-age entry and \
8045 ONLY that — an RSS entry on an instance with both RSS knobs at zero \
8046 is one the operator chose to keep",
8047 );
8048
8049 // And zero disables it, consistently with the other two knobs.
8050 assert_eq!(
8051 prune_old_entries(&pool, 0, 0, 0).await?,
8052 0,
8053 "publication_retention_days = 0 still deleted something",
8054 );
8055 Ok(())
8056 }
8057
8058 /// **A retention window too large to be a date must disable that pass, not
8059 /// kill the sweeper.**
8060 ///
8061 /// Every knob parses from a `u32` with no upper bound, and `Duration::days` /
8062 /// `DateTime - TimeDelta` both panic out of range — measured, anything past
8063 /// roughly 96 million days, and `u32::MAX` is. A unit slip (seconds or
8064 /// milliseconds typed into a days field) reaches it.
8065 ///
8066 /// The old failure was quiet: this runs in a spawned task, so tokio catches
8067 /// the panic and the sweeper stops for the life of the process, taking the
8068 /// release valve for `db_size_watermark_bytes` with it — the one thing that
8069 /// stops polling for every reader on the instance.
8070 ///
8071 /// `standard_site::ingest_floor` already answers the same input with "no
8072 /// floor", and `Config::retention_for` exists to keep the two agreeing, so
8073 /// this is also the end of a disagreement: unrepresentable meant "store
8074 /// everything" on one side and "panic" on the other.
8075 #[tokio::test]
8076 async fn an_unrepresentable_retention_window_disables_the_pass_it_belongs_to() -> Result<()> {
8077 let pool = init_url("sqlite::memory:").await?;
8078 let feed_id = upsert_feed(
8079 &pool,
8080 &NewFeed {
8081 url: "https://absurd.example/feed.xml".to_string(),
8082 ..Default::default()
8083 },
8084 )
8085 .await?;
8086 let ancient = (chrono::Utc::now() - chrono::Duration::days(1_000))
8087 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
8088 insert_entries(
8089 &pool,
8090 feed_id,
8091 &[NewEntry {
8092 guid: "ancient".into(),
8093 published: Some(ancient.clone()),
8094 fetched_at: Some(ancient),
8095 ..Default::default()
8096 }],
8097 0,
8098 )
8099 .await?;
8100
8101 // Each knob in turn, since each computes its own cutoff.
8102 let absurd = u32::MAX as i64;
8103 assert_eq!(
8104 prune_old_entries(&pool, absurd, 0, 0).await?,
8105 0,
8106 "an absurd rolling window deleted something",
8107 );
8108 assert_eq!(
8109 prune_old_entries(&pool, 0, absurd, 0).await?,
8110 0,
8111 "an absurd hard ceiling deleted something",
8112 );
8113 assert_eq!(
8114 prune_old_entries(&pool, 0, 0, absurd).await?,
8115 0,
8116 "an absurd archive ceiling deleted something",
8117 );
8118 assert_eq!(
8119 count_entries(&pool).await?,
8120 1,
8121 "the entry went away under a window that cannot even be expressed",
8122 );
8123
8124 // And the sweep still works for the same knobs at a sane value — a
8125 // function that returned early on every input would satisfy the above.
8126 assert_eq!(
8127 prune_old_entries(&pool, 30, 0, 0).await?,
8128 1,
8129 "a 30-day window did not delete a 1000-day-old entry",
8130 );
8131 Ok(())
8132 }
8133
8134 /// The SQL list and the Rust slice are asserted equal, for the same reason
8135 /// [`POLLABLE_KINDS_SQL`] is: a literal here and a slice there is the drift
8136 /// the `kind` column was introduced to end.
8137 #[test]
8138 fn the_sql_aged_kind_list_matches_the_rust_one() {
8139 let expected = crate::feed::FeedKind::AGED
8140 .iter()
8141 .map(|k| format!("'{}'", k.as_str()))
8142 .collect::<Vec<_>>()
8143 .join(", ");
8144 assert_eq!(AGED_KINDS_SQL, expected);
8145 }
8146
8147 #[tokio::test]
8148 async fn prune_old_entries_deletes_only_old_and_cascades_entry_state() -> Result<()> {
8149 let pool = init_url("sqlite::memory:").await?;
8150 let feed_id = upsert_feed(
8151 &pool,
8152 &NewFeed {
8153 url: "https://ret.example/feed.xml".to_string(),
8154 ..Default::default()
8155 },
8156 )
8157 .await?;
8158
8159 let recent = now_rfc3339();
8160 let ancient = (chrono::Utc::now() - chrono::Duration::days(365))
8161 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
8162
8163 // One fresh (published now), one ancient (published a year ago), and one
8164 // UNDATED-but-freshly-fetched (published NULL, fetched_at now) — the last
8165 // must survive because COALESCE falls back to fetched_at, not to "old".
8166 insert_entries(
8167 &pool,
8168 feed_id,
8169 &[
8170 NewEntry {
8171 guid: "fresh".into(),
8172 published: Some(recent.clone()),
8173 fetched_at: Some(recent.clone()),
8174 ..Default::default()
8175 },
8176 NewEntry {
8177 guid: "ancient".into(),
8178 published: Some(ancient.clone()),
8179 fetched_at: Some(ancient.clone()),
8180 ..Default::default()
8181 },
8182 NewEntry {
8183 guid: "undated-fresh".into(),
8184 published: None,
8185 fetched_at: Some(recent.clone()),
8186 ..Default::default()
8187 },
8188 ],
8189 0,
8190 )
8191 .await?;
8192 assert_eq!(count_entries(&pool).await?, 3);
8193 // Subscribe so mark_read is authorized to write an entry_state row.
8194 replace_sub_refs(&pool, "did:plc:reader", &[feed_id]).await?;
8195
8196 // Give the ancient entry an entry_state row so we can prove the FK cascade.
8197 let ancient_id: i64 = sqlx::query_scalar("SELECT id FROM entries WHERE guid = 'ancient'")
8198 .fetch_one(&pool)
8199 .await?;
8200 let wrote = mark_read(&pool, "did:plc:reader", ancient_id, true).await?;
8201 assert!(wrote, "mark_read must write with a sub_ref in place");
8202 let state_before: i64 =
8203 sqlx::query_scalar("SELECT COUNT(*) FROM entry_state WHERE entry_id = ?1")
8204 .bind(ancient_id)
8205 .fetch_one(&pool)
8206 .await?;
8207 assert_eq!(state_before, 1);
8208
8209 // Prune at a 90-day window: only the ancient entry is old.
8210 let deleted = prune_old_entries(&pool, 90, 3650, 0).await?;
8211 assert_eq!(deleted, 1, "only the year-old entry should be pruned");
8212 assert_eq!(
8213 count_entries(&pool).await?,
8214 2,
8215 "fresh + undated-fresh survive"
8216 );
8217
8218 // The surviving guids are exactly the two fresh ones.
8219 let surviving: Vec<String> = sqlx::query_scalar("SELECT guid FROM entries ORDER BY guid")
8220 .fetch_all(&pool)
8221 .await?;
8222 assert_eq!(surviving, vec!["fresh", "undated-fresh"]);
8223
8224 // entry_state for the deleted entry cascaded away via the FK.
8225 let state_after: i64 =
8226 sqlx::query_scalar("SELECT COUNT(*) FROM entry_state WHERE entry_id = ?1")
8227 .bind(ancient_id)
8228 .fetch_one(&pool)
8229 .await?;
8230 assert_eq!(state_after, 0, "entry_state must cascade on entry delete");
8231
8232 // days == 0 disables the rolling WINDOW. The 3650-day ceiling still runs
8233 // (see `a_disabled_window_does_not_disable_the_ceiling`); it deletes
8234 // nothing here because both survivors are fresh.
8235 assert_eq!(prune_old_entries(&pool, 0, 3650, 0).await?, 0);
8236 assert_eq!(count_entries(&pool).await?, 2);
8237 Ok(())
8238 }
8239
8240 #[tokio::test]
8241 async fn prune_removes_orphan_ids_from_read_cursor() -> Result<()> {
8242 let pool = init_url("sqlite::memory:").await?;
8243 let did = "did:plc:reader";
8244 let feed_url = "https://orphan.example/feed.xml";
8245 let feed_id = upsert_feed(
8246 &pool,
8247 &NewFeed {
8248 url: feed_url.to_string(),
8249 ..Default::default()
8250 },
8251 )
8252 .await?;
8253 // Caller subscribes so mark-read is authorized to project into the cursor.
8254 replace_sub_refs(&pool, did, &[feed_id]).await?;
8255
8256 let ancient = (chrono::Utc::now() - chrono::Duration::days(365))
8257 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
8258 let recent = now_rfc3339();
8259 insert_entries(
8260 &pool,
8261 feed_id,
8262 &[
8263 NewEntry {
8264 guid: "old".into(),
8265 published: Some(ancient.clone()),
8266 fetched_at: Some(ancient.clone()),
8267 ..Default::default()
8268 },
8269 NewEntry {
8270 guid: "new".into(),
8271 published: Some(recent.clone()),
8272 fetched_at: Some(recent.clone()),
8273 ..Default::default()
8274 },
8275 ],
8276 0,
8277 )
8278 .await?;
8279 let old_id: i64 = sqlx::query_scalar("SELECT id FROM entries WHERE guid = 'old'")
8280 .fetch_one(&pool)
8281 .await?;
8282 let new_id: i64 = sqlx::query_scalar("SELECT id FROM entries WHERE guid = 'new'")
8283 .fetch_one(&pool)
8284 .await?;
8285
8286 // Mark BOTH read — the cursor's read_ids now references both entry ids.
8287 mark_read(&pool, did, old_id, true).await?;
8288 mark_read(&pool, did, new_id, true).await?;
8289 let before = get_cursor(&pool, did, feed_url).await?.unwrap();
8290 let ids_before: Vec<String> = serde_json::from_str(&before.read_ids)?;
8291 assert!(ids_before.contains(&old_id.to_string()));
8292 assert!(ids_before.contains(&new_id.to_string()));
8293
8294 // Prune the old entry — its id must be scrubbed from the cursor's id-set.
8295 let deleted = prune_old_entries(&pool, 90, 3650, 0).await?;
8296 assert_eq!(deleted, 1);
8297 let after = get_cursor(&pool, did, feed_url).await?.unwrap();
8298 let ids_after: Vec<String> = serde_json::from_str(&after.read_ids)?;
8299 assert_eq!(
8300 ids_after,
8301 vec![new_id.to_string()],
8302 "orphaned (deleted) entry id must be removed; live id kept"
8303 );
8304 // The scrub re-dirties the cursor so the flusher resyncs the PDS record.
8305 assert!(
8306 after.dirty,
8307 "cursor must be marked dirty after orphan scrub"
8308 );
8309 Ok(())
8310 }
8311
8312 #[tokio::test]
8313 async fn insert_entries_trim_scrubs_orphan_cursor_ids() -> Result<()> {
8314 // The per-feed max_entries trim path must ALSO scrub orphaned cursor ids.
8315 let pool = init_url("sqlite::memory:").await?;
8316 let did = "did:plc:reader";
8317 let feed_url = "https://trim.example/feed.xml";
8318 let feed_id = upsert_feed(
8319 &pool,
8320 &NewFeed {
8321 url: feed_url.to_string(),
8322 ..Default::default()
8323 },
8324 )
8325 .await?;
8326 replace_sub_refs(&pool, did, &[feed_id]).await?;
8327
8328 // Two entries, cap of 2 for now (no trim yet).
8329 insert_entries(
8330 &pool,
8331 feed_id,
8332 &[
8333 NewEntry {
8334 guid: "a".into(),
8335 published: Some("2026-01-01T00:00:00Z".into()),
8336 ..Default::default()
8337 },
8338 NewEntry {
8339 guid: "b".into(),
8340 published: Some("2026-01-02T00:00:00Z".into()),
8341 ..Default::default()
8342 },
8343 ],
8344 2,
8345 )
8346 .await?;
8347 let a_id: i64 = sqlx::query_scalar("SELECT id FROM entries WHERE guid = 'a'")
8348 .fetch_one(&pool)
8349 .await?;
8350 mark_read(&pool, did, a_id, true).await?;
8351
8352 // Insert a newer entry with cap=1 → the oldest ('a') is trimmed away.
8353 insert_entries(
8354 &pool,
8355 feed_id,
8356 &[NewEntry {
8357 guid: "c".into(),
8358 published: Some("2026-01-03T00:00:00Z".into()),
8359 ..Default::default()
8360 }],
8361 1,
8362 )
8363 .await?;
8364 // 'a' is gone.
8365 let a_still: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries WHERE guid = 'a'")
8366 .fetch_one(&pool)
8367 .await?;
8368 assert_eq!(a_still, 0, "oldest entry trimmed by the per-feed cap");
8369
8370 // The cursor no longer references the trimmed id.
8371 let cursor = get_cursor(&pool, did, feed_url).await?.unwrap();
8372 let ids: Vec<String> = serde_json::from_str(&cursor.read_ids)?;
8373 assert!(
8374 !ids.contains(&a_id.to_string()),
8375 "trimmed entry id must be scrubbed from the cursor"
8376 );
8377 Ok(())
8378 }
8379
8380 /// A sweep spanning several batches must still delete everything.
8381 ///
8382 /// The batching exists to make the write-lock hold interruptible, not to
8383 /// make the sweep partial — so the obvious way to get it wrong is an
8384 /// off-by-one that leaves a batch behind, or a loop that exits on the first
8385 /// short batch instead of the first empty one.
8386 #[tokio::test]
8387 async fn a_sweep_larger_than_one_batch_still_drains() -> Result<()> {
8388 let pool = init_url("sqlite::memory:").await?;
8389 let feed_id = upsert_feed(
8390 &pool,
8391 &NewFeed {
8392 url: "https://bulk.example/f.xml".to_string(),
8393 ..Default::default()
8394 },
8395 )
8396 .await?;
8397 let old = (chrono::Utc::now() - chrono::Duration::days(400))
8398 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
8399 // Deliberately not a multiple of PRUNE_BATCH, so the final batch is
8400 // short and the loop has to keep going to the empty one.
8401 let count = (PRUNE_BATCH * 2 + 137) as usize;
8402 let entries: Vec<NewEntry> = (0..count)
8403 .map(|i| NewEntry {
8404 guid: format!("bulk-{i}"),
8405 published: Some(old.clone()),
8406 ..Default::default()
8407 })
8408 .collect();
8409 insert_entries(&pool, feed_id, &entries, 0).await?;
8410 assert_eq!(count_entries(&pool).await? as usize, count);
8411
8412 let deleted = prune_old_entries(&pool, 30, 180, 0).await?;
8413 assert_eq!(deleted as usize, count, "the sweep left rows behind");
8414 assert_eq!(count_entries(&pool).await?, 0);
8415 Ok(())
8416 }
8417
8418 /// What a sweep driven by a lock test actually did.
8419 ///
8420 /// `Contended` is NOT a failure. `SQLITE_BUSY` on the pruner is an outcome
8421 /// production expects and handles — `scheduler.rs` logs it and the next tick
8422 /// retries — so a test that treats it as a regression is stricter than the
8423 /// system it guards, and fails for a reason its own assertions are not
8424 /// about. See #146.
8425 enum SweepOutcome {
8426 Completed(u64),
8427 Contended,
8428 }
8429
8430 /// True for the `SQLITE_BUSY` FAMILY anywhere in the chain.
8431 ///
8432 /// Matched on the DRIVER CODE, not on the message text: "database is
8433 /// locked" is a string another error could plausibly carry, and this
8434 /// decides whether a test failure is suppressed.
8435 ///
8436 /// **Masked to the primary code.** sqlx-sqlite's `code()` returns
8437 /// `sqlite3_extended_errcode` verbatim, so comparing it to `"5"` matches
8438 /// only bare `SQLITE_BUSY` and treats the WAL variants as hard failures:
8439 /// `BUSY_RECOVERY` (261), `BUSY_SNAPSHOT` (517), `BUSY_TIMEOUT` (773).
8440 /// This database runs in WAL mode and `store.rs` already documents hitting
8441 /// `SQLITE_BUSY_SNAPSHOT`, so that gap is not hypothetical — the narrowing
8442 /// would have rejected the very class this tolerance exists for.
8443 ///
8444 /// `& 0xFF` is how SQLite defines the relationship: the low byte of an
8445 /// extended code IS the primary code.
8446 fn is_sqlite_busy(err: &anyhow::Error) -> bool {
8447 err.chain().any(|e| {
8448 e.downcast_ref::<sqlx::Error>().is_some_and(|e| match e {
8449 sqlx::Error::Database(db) => db
8450 .code()
8451 .and_then(|c| c.parse::<i32>().ok())
8452 .is_some_and(is_busy_code),
8453 _ => false,
8454 })
8455 })
8456 }
8457
8458 /// The classification, split out so the WAL variants are TESTABLE.
8459 ///
8460 /// A `BUSY_SNAPSHOT` cannot be produced on demand in a test, so without
8461 /// this the claim that 261/517/773 are tolerated would be a comment and
8462 /// nothing else. The wiring — that `is_sqlite_busy` consults this at all —
8463 /// is pinned separately by `a_busy_sweep_is_reported_as_contended_not_as_a_failure`,
8464 /// which drives a real `SQLITE_BUSY` end to end.
8465 fn is_busy_code(code: i32) -> bool {
8466 code & 0xFF == 5
8467 }
8468
8469 /// **The whole `SQLITE_BUSY` family, and nothing else.**
8470 #[test]
8471 fn busy_codes_cover_the_wal_variants() {
8472 for code in [
8473 5, // SQLITE_BUSY
8474 261, // SQLITE_BUSY_RECOVERY
8475 517, // SQLITE_BUSY_SNAPSHOT
8476 773, // SQLITE_BUSY_TIMEOUT
8477 ] {
8478 assert!(
8479 is_busy_code(code),
8480 "{code} is in the BUSY family but would be treated as a hard failure"
8481 );
8482 }
8483 for code in [
8484 0, // SQLITE_OK
8485 1, // SQLITE_ERROR
8486 6, // SQLITE_LOCKED — adjacent, and deliberately NOT tolerated
8487 262, // SQLITE_LOCKED_SHAREDCACHE
8488 11, // SQLITE_CORRUPT
8489 ] {
8490 assert!(
8491 !is_busy_code(code),
8492 "{code} is not contention, but would be swallowed as though it were"
8493 );
8494 }
8495 }
8496
8497 /// Run the batched delete, separating "the write lock was contended" from
8498 /// "the loop misbehaved". Only the second is this test's subject.
8499 async fn sweep_tolerating_busy(
8500 pool: &SqlitePool,
8501 select_ids: &str,
8502 cutoff: &str,
8503 label: &str,
8504 ) -> Result<SweepOutcome> {
8505 match delete_in_batches(pool, select_ids, cutoff, label).await {
8506 Ok(n) => Ok(SweepOutcome::Completed(n)),
8507 // Contended, not broken. Narrowed to SQLITE_BUSY on purpose: every
8508 // other error still fails the caller, so this is not a blanket
8509 // `let _ =` that would delete the test while keeping its name.
8510 Err(err) if is_sqlite_busy(&err) => Ok(SweepOutcome::Contended),
8511 Err(err) => Err(err),
8512 }
8513 }
8514
8515 /// **A sweep that loses the write lock is inconclusive, not a failure.**
8516 ///
8517 /// CI hit this on `main` at `d05a716`: the sweeper took `SQLITE_BUSY` and
8518 /// the test reported a regression, on a tree whose only changes were two
8519 /// version strings and a changelog.
8520 ///
8521 /// Forced deterministically rather than waiting for a contended runner — it
8522 /// did not reproduce in 48 local runs — by holding a write transaction open
8523 /// and giving the sweep a `busy_timeout` short enough to give up at once.
8524 #[tokio::test]
8525 async fn a_busy_sweep_is_reported_as_contended_not_as_a_failure() -> Result<()> {
8526 struct TempDb(std::path::PathBuf);
8527 impl Drop for TempDb {
8528 fn drop(&mut self) {
8529 for suffix in ["", "-wal", "-shm"] {
8530 std::fs::remove_file(format!("{}{suffix}", self.0.display())).ok();
8531 }
8532 }
8533 }
8534 let path = std::env::temp_dir().join(format!("fr-busysweep-{}.db", std::process::id()));
8535 drop(TempDb(path.clone()));
8536 let _tmp = TempDb(path.clone());
8537 let url = format!("sqlite://{}", path.display());
8538 let pool = init_url(&url).await?;
8539
8540 let feed_id = upsert_feed(
8541 &pool,
8542 &NewFeed {
8543 url: "https://busy.example/f.xml".to_string(),
8544 ..Default::default()
8545 },
8546 )
8547 .await?;
8548 let old = (chrono::Utc::now() - chrono::Duration::days(400))
8549 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
8550 let entries: Vec<NewEntry> = (0..4)
8551 .map(|i| NewEntry {
8552 guid: format!("busy-{i}"),
8553 url: Some(format!("https://busy.example/{i}")),
8554 title: Some(format!("e{i}")),
8555 published: Some(old.clone()),
8556 ..Default::default()
8557 })
8558 .collect();
8559 insert_entries(&pool, feed_id, &entries, 1_000).await?;
8560
8561 // A sweep pool that gives up on a contended write immediately.
8562 let sweep_pool = SqlitePoolOptions::new()
8563 .max_connections(1)
8564 .connect_with(
8565 url.parse::<sqlx::sqlite::SqliteConnectOptions>()?
8566 .busy_timeout(std::time::Duration::from_millis(2)),
8567 )
8568 .await?;
8569
8570 // Hold the write lock for the duration of the sweep below.
8571 let mut blocker = pool.acquire().await?;
8572 sqlx::query("BEGIN IMMEDIATE")
8573 .execute(&mut *blocker)
8574 .await?;
8575
8576 let cutoff = (chrono::Utc::now() - chrono::Duration::days(180))
8577 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
8578 let outcome = sweep_tolerating_busy(
8579 &sweep_pool,
8580 "SELECT id FROM entries WHERE COALESCE(published, fetched_at) < ?1",
8581 &cutoff,
8582 "busy-sweep-test",
8583 )
8584 .await;
8585
8586 sqlx::query("ROLLBACK").execute(&mut *blocker).await.ok();
8587
8588 match outcome {
8589 Ok(SweepOutcome::Contended) => Ok(()),
8590 Ok(SweepOutcome::Completed(n)) => panic!(
8591 "the sweep completed ({n} rows) while the write lock was held — \
8592 the fixture is not actually contending, so this test proves nothing"
8593 ),
8594 Err(err) => panic!(
8595 "a contended sweep was reported as a failure rather than as \
8596 inconclusive; production logs this and retries on the next \
8597 tick (scheduler.rs): {err:#}"
8598 ),
8599 }
8600 }
8601
8602 /// **A sweep error that is NOT `SQLITE_BUSY` must still fail.**
8603 ///
8604 /// `sweep_tolerating_busy` claims to narrow its tolerance to contention.
8605 /// Without this, that claim is unenforced: widening the arm to `Err(_) =>
8606 /// Contended` swallows every sweep error — a malformed query, a missing
8607 /// table, a corrupt file — and the whole suite stays green. Measured, not
8608 /// assumed: that mutation passed 733 tests before this test existed.
8609 #[tokio::test]
8610 async fn a_non_busy_sweep_error_still_fails() -> Result<()> {
8611 let pool = init_url("sqlite::memory:").await?;
8612 // A table that does not exist: SQLITE_ERROR (1), not SQLITE_BUSY (5).
8613 let outcome = sweep_tolerating_busy(
8614 &pool,
8615 "SELECT id FROM no_such_table WHERE created < ?1",
8616 "2026-01-01T00:00:00Z",
8617 "bad-query-test",
8618 )
8619 .await;
8620
8621 match outcome {
8622 Err(err) => {
8623 assert!(
8624 !is_sqlite_busy(&err),
8625 "fixture drifted: this must be a non-BUSY error, got {err:#}"
8626 );
8627 Ok(())
8628 }
8629 Ok(SweepOutcome::Contended) => panic!(
8630 "a malformed sweep was reported as lock contention — the \
8631 tolerance is a blanket error swallow, not a narrowing"
8632 ),
8633 Ok(SweepOutcome::Completed(n)) => {
8634 panic!("a sweep over a missing table reported {n} rows deleted")
8635 }
8636 }
8637 }
8638
8639 /// **An UNCONTENDED sweep must report `Completed`.**
8640 ///
8641 /// This exists to stop the `Contended` arm above becoming a way to never
8642 /// run the hand-off assertions. Make `sweep_tolerating_busy` return
8643 /// `Contended` unconditionally and the sweep-lock test still passes — it
8644 /// just silently stops testing anything. This one fails instead.
8645 ///
8646 /// That is the difference between tolerating a real contention loss and
8647 /// deleting a test while keeping its name.
8648 #[tokio::test]
8649 async fn a_sweep_with_no_contention_completes() -> Result<()> {
8650 struct TempDb(std::path::PathBuf);
8651 impl Drop for TempDb {
8652 fn drop(&mut self) {
8653 for suffix in ["", "-wal", "-shm"] {
8654 std::fs::remove_file(format!("{}{suffix}", self.0.display())).ok();
8655 }
8656 }
8657 }
8658 let path = std::env::temp_dir().join(format!("fr-calmsweep-{}.db", std::process::id()));
8659 drop(TempDb(path.clone()));
8660 let _tmp = TempDb(path.clone());
8661 let pool = init_url(&format!("sqlite://{}", path.display())).await?;
8662
8663 let feed_id = upsert_feed(
8664 &pool,
8665 &NewFeed {
8666 url: "https://calm.example/f.xml".to_string(),
8667 ..Default::default()
8668 },
8669 )
8670 .await?;
8671 let old = (chrono::Utc::now() - chrono::Duration::days(400))
8672 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
8673 let entries: Vec<NewEntry> = (0..3)
8674 .map(|i| NewEntry {
8675 guid: format!("calm-{i}"),
8676 published: Some(old.clone()),
8677 ..Default::default()
8678 })
8679 .collect();
8680 insert_entries(&pool, feed_id, &entries, 1_000).await?;
8681
8682 let cutoff = (chrono::Utc::now() - chrono::Duration::days(180))
8683 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
8684 match sweep_tolerating_busy(
8685 &pool,
8686 "SELECT id FROM entries WHERE COALESCE(published, fetched_at) < ?1",
8687 &cutoff,
8688 "calm-sweep-test",
8689 )
8690 .await?
8691 {
8692 SweepOutcome::Completed(n) => {
8693 assert_eq!(n, 3, "the uncontended sweep did not delete the fixture");
8694 Ok(())
8695 }
8696 SweepOutcome::Contended => panic!(
8697 "nothing was holding the write lock, yet the sweep reported \
8698 contention — every test that skips on `Contended` is now \
8699 skipping unconditionally"
8700 ),
8701 }
8702 }
8703
8704 /// **The sweep must not lock other writers out for its duration.**
8705 ///
8706 /// The whole sweep used to be one transaction — both deletes plus a global
8707 /// cursor scrub that loads every `read_cursor` row and then issues a
8708 /// per-cursor live-ids query. SQLite is single-writer with a 5 s
8709 /// `busy_timeout`, so every mark-read, login write and cursor flush failed
8710 /// for that whole span.
8711 ///
8712 /// On-disk (WAL) because the in-memory pool is deliberately
8713 /// single-connection, which would make a concurrency test meaningless.
8714 ///
8715 /// **The writer runs on its own pool with a short `busy_timeout`, and the
8716 /// runtime is multi-thread.** Both are load-bearing — a 5 s `busy_timeout`
8717 /// on a shared runtime is what made this test flake on CI. See the comment
8718 /// on the writer pool and the `attempts` assertion.
8719 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8720 async fn a_writer_gets_through_while_the_sweep_runs() -> Result<()> {
8721 // **Cleanup on EVERY exit, including a panicking assertion.**
8722 //
8723 // The three `remove_file` calls used to sit after the assertions, so any
8724 // failure leaked the database and its `-wal`/`-shm` — 2.7–12.8 MB a time,
8725 // and this test is deliberately the one most likely to fail. Worse, setup
8726 // removed only the `.db`, so a recycled PID paired a fresh database with a
8727 // stale WAL. A guard drops on the unwind path too and takes all three.
8728 struct TempDb(std::path::PathBuf);
8729 impl Drop for TempDb {
8730 fn drop(&mut self) {
8731 for suffix in ["", "-wal", "-shm"] {
8732 std::fs::remove_file(format!("{}{suffix}", self.0.display())).ok();
8733 }
8734 }
8735 }
8736 let dir = std::env::temp_dir();
8737 let path = dir.join(format!("fr-sweeplock-{}.db", std::process::id()));
8738 // Drops the previous run's leftovers, WAL and all, before opening.
8739 drop(TempDb(path.clone()));
8740 let _tmp = TempDb(path.clone());
8741 let url = format!("sqlite://{}", path.display());
8742 let pool = init_url(&url).await?;
8743
8744 let feed_id = upsert_feed(
8745 &pool,
8746 &NewFeed {
8747 url: "https://lock.example/f.xml".to_string(),
8748 ..Default::default()
8749 },
8750 )
8751 .await?;
8752 // **The fixture is DERIVED from the batch count, not described by it.**
8753 //
8754 // Every assertion below reasons about "ten hand-off windows". That was
8755 // prose — a `const BATCHES: u32 = 10` sitting next to a `PRUNE_BATCH *
8756 // 10` fixture with nothing tying them together. Editing the fixture
8757 // alone to `PRUNE_BATCH * 4` left the floor still demanding ten
8758 // hand-offs' worth of time from a four-batch loop, and correct code was
8759 // accused of not handing the lock over at all (1 run in 6). Now the
8760 // compiler carries the coupling.
8761 const BATCHES: i64 = 10;
8762 // **The 50% ceiling below is only safe because BATCHES is large.**
8763 //
8764 // `max_refused_run / attempts` is bounded by roughly `1 / BATCHES` only
8765 // because the fixture opens that many hand-off windows. Shrink it and
8766 // correct code walks into the ceiling: measured with production code
8767 // untouched and the per-batch hold grown 10x, `BATCHES = 4` gives ratios
8768 // of 0.21–0.35 and `BATCHES = 2` gives 0.45–0.56, **failing 3 runs in
8769 // 5**. The comment above invites editing this fixture; this stops that
8770 // edit from silently turning the assertion against the code it guards.
8771 const _: () = assert!(
8772 BATCHES >= 5,
8773 "the 50% ceiling assumes ~1/BATCHES; below 5 batches correct code false-fails",
8774 );
8775 let old = (chrono::Utc::now() - chrono::Duration::days(400))
8776 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
8777 let entries: Vec<NewEntry> = (0..(PRUNE_BATCH * BATCHES) as usize)
8778 .map(|i| NewEntry {
8779 guid: format!("lock-{i}"),
8780 published: Some(old.clone()),
8781 ..Default::default()
8782 })
8783 .collect();
8784 insert_entries(&pool, feed_id, &entries, 0).await?;
8785
8786 // **The writer gets its OWN pool, with a SHORT `busy_timeout`.**
8787 //
8788 // This is the fix for the CI flake described on the `attempts` assertion
8789 // below, and it is two separate changes.
8790 //
8791 // *Its own pool*, so the only thing that can block a write is SQLite's
8792 // write lock — the thing under test. Sharing the 5-connection pool with
8793 // the sweep meant a write could also stall waiting to ACQUIRE a pooled
8794 // connection the sweep was holding, which is a confounder that looks
8795 // identical from the outside.
8796 //
8797 // *A short `busy_timeout`*, so a contended write FAILS FAST and the loop
8798 // takes another shot. At the production 5 s, SQLite's busy handler backs
8799 // off internally — 1, 2, 5, 10, 25, 50, 100 ms and up — all inside a
8800 // single `execute()`. The writer therefore gets ONE attempt per blocked
8801 // write, and once the ladder reaches 100 ms it sleeps straight past the
8802 // `PRUNE_BATCH_HANDOFF` windows `delete_in_batches` opens. Failing fast
8803 // turns one low-probability attempt into hundreds of independent ones:
8804 // measured 54 attempts at 5 ms, 517 at 2 ms, over the same sweep.
8805 const WRITER_BUSY_TIMEOUT: std::time::Duration = std::time::Duration::from_millis(2);
8806 let writer_pool = SqlitePoolOptions::new()
8807 .min_connections(1)
8808 .max_connections(1)
8809 .connect_with(
8810 SqliteConnectOptions::from_str(&url)?
8811 .foreign_keys(true)
8812 .busy_timeout(WRITER_BUSY_TIMEOUT)
8813 .log_statements(tracing::log::LevelFilter::Debug),
8814 )
8815 .await?;
8816
8817 let done = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false));
8818 let writer_done = std::sync::Arc::clone(&done);
8819 let writer = tokio::spawn(async move {
8820 // `(when the attempt STARTED, whether it landed)`.
8821 //
8822 // The ORDER is what the assertions read, not the timestamps: they
8823 // count consecutive failures. The instants serve only to select the
8824 // attempts made inside the measured window — the writer is spawned
8825 // before `t0`, so a plain counter would fold in attempts that can
8826 // never appear in `during`.
8827 //
8828 // (An earlier version of this comment, left behind by the switch away
8829 // from elapsed time, said completions were recorded and that "the
8830 // timestamps are the load-bearing part". Neither is true now.)
8831 let mut outcomes: Vec<(std::time::Instant, bool)> = Vec::new();
8832 // Kept for the failure message: if the writes are failing for a
8833 // reason that is NOT lock contention, nothing lands and the test
8834 // fails — this is what says why. Timestamped so the test can drop it
8835 // when it describes an attempt OUTSIDE the measured window; the
8836 // writer starts before `t0`, so the very first error is usually from
8837 // an attempt the assertions never look at.
8838 let mut first_err: Option<(std::time::Instant, String)> = None;
8839 while !writer_done.load(std::sync::atomic::Ordering::Relaxed) {
8840 let started = std::time::Instant::now();
8841 match grant_access(
8842 &writer_pool,
8843 &format!("did:plc:writer{}", outcomes.len()),
8844 None,
8845 "sweep-test",
8846 None,
8847 )
8848 .await
8849 {
8850 Ok(()) => outcomes.push((started, true)),
8851 // Expected: the sweep holds the write lock right now.
8852 // Retrying is the entire point, so this is counted, not
8853 // fatal. A `?` here would abort the writer on the first
8854 // contended write and destroy the measurement.
8855 Err(err) => {
8856 outcomes.push((started, false));
8857 if first_err.is_none() {
8858 first_err = Some((started, format!("{err:#}")));
8859 }
8860 }
8861 }
8862 tokio::task::yield_now().await;
8863 }
8864 writer_pool.close().await;
8865 (outcomes, first_err)
8866 });
8867
8868 // **Drive `delete_in_batches` directly, not `prune_old_entries`.**
8869 //
8870 // The subject is the batched delete loop and whether it hands the write
8871 // lock over between batches. `prune_old_entries` wraps it in work that
8872 // is not that — two delete passes plus `prune_orphan_cursor_ids` — so
8873 // timing the whole call measures a window in which the lock was never
8874 // meant to be held throughout, and writes landing outside the loop
8875 // count as though the loop had handed the lock over.
8876 //
8877 // A correction to what this comment first claimed. It said the cursor
8878 // scrub was a tail that "grows with the number of rows deleted", and
8879 // that this explained a `42 of 358` measurement. **That is false, and
8880 // measured to be false**: this fixture creates no `read_cursor` rows at
8881 // all, so the scrub does one `SELECT` over an empty table and loops zero
8882 // times — 0.16–2 ms, 0.03–0.5% of the window, at any fixture size. It
8883 // cannot explain anything. Narrowing the window is still right, for the
8884 // reason above; the mechanism originally given for it was not real.
8885 //
8886 // The consequence worth stating: because the fixture has no cursors, the
8887 // old form never covered the scrub's locking either — it only appeared
8888 // to. Nothing here regressed. `prune_orphan_cursor_ids` holding the lock
8889 // across a whole pass is a real production invariant (see its own doc)
8890 // and remains untested; that needs a test with actual cursors, not this
8891 // one.
8892 let hard_cutoff = (chrono::Utc::now() - chrono::Duration::days(180))
8893 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
8894 let t0 = std::time::Instant::now();
8895 let outcome = sweep_tolerating_busy(
8896 &pool,
8897 "SELECT id FROM entries WHERE COALESCE(published, fetched_at) < ?1",
8898 &hard_cutoff,
8899 "sweep-lock-test",
8900 )
8901 .await?;
8902 let sweep = t0.elapsed();
8903
8904 // **Teardown happens on BOTH paths, before the outcome is inspected.**
8905 //
8906 // The contended arm below used to carry its own copy of these two lines.
8907 // A probe proved that arm is never reached by the suite — a `panic!` in
8908 // it failed nothing — so it was five lines of unexercised teardown that
8909 // would run for the first time on a contended CI runner, which is
8910 // exactly when it has to work. Hoisting leaves the arm with nothing that
8911 // can be wrong.
8912 done.store(true, std::sync::atomic::Ordering::Relaxed);
8913 let (outcomes, first_err) = writer.await?;
8914
8915 let deleted = match outcome {
8916 SweepOutcome::Completed(n) => n,
8917 // **Inconclusive, not a regression.** The sweeper lost the write
8918 // lock, which says nothing about whether it hands the lock over
8919 // between batches — the property below. Production logs this and
8920 // retries on the next tick (`scheduler.rs`), so a test that failed
8921 // here would be stricter than the system it guards. Observed on CI
8922 // at `d05a716`, on a tree with no `.rs` change at all.
8923 //
8924 // `a_sweep_with_no_contention_completes` is what stops this arm
8925 // becoming a way to never run the assertions.
8926 SweepOutcome::Contended => {
8927 eprintln!(
8928 "sweep-lock test INCONCLUSIVE: the sweeper took SQLITE_BUSY; \
8929 the hand-off assertions did not run"
8930 );
8931 return Ok(());
8932 }
8933 };
8934 let sweep_end = t0 + sweep;
8935 // Attempts actually made inside the measured window, in order.
8936 let inside: Vec<bool> = outcomes
8937 .iter()
8938 .filter(|(t, _)| *t >= t0 && *t < sweep_end)
8939 .map(|(_, ok)| *ok)
8940 .collect();
8941 let attempts = inside.len();
8942 let during = inside.iter().filter(|ok| **ok).count();
8943 // **The longest unbroken run of REFUSED attempts.**
8944 //
8945 // Counted in attempts, not elapsed time — see the note on the assertion
8946 // for why that distinction is the whole point.
8947 let max_refused_run = {
8948 let (mut worst, mut run) = (0usize, 0usize);
8949 for ok in &inside {
8950 run = if *ok { 0 } else { run + 1 };
8951 worst = worst.max(run);
8952 }
8953 worst
8954 };
8955 let why = first_err
8956 .filter(|(t, _)| *t >= t0 && *t < sweep_end)
8957 .map(|(_, e)| format!(" (first in-window write error: {e})"))
8958 .unwrap_or_default();
8959
8960 assert_eq!(deleted as usize, entries.len());
8961 // **The sweep has to BE batched before anything downstream means
8962 // anything, and this floor is derived, not calibrated.**
8963 //
8964 // The fixture is `PRUNE_BATCH * 10` rows, all older than the hard
8965 // ceiling, so the hard-ceiling delete drains them in ten full batches
8966 // and stands down `PRUNE_BATCH_HANDOFF` after each. A genuinely batched
8967 // sweep therefore cannot finish in under `10 * PRUNE_BATCH_HANDOFF` on
8968 // any machine, however fast its disk — the sleeps are a floor the
8969 // hardware cannot undercut, and the deletes themselves only add to it.
8970 //
8971 // A loop that has LOST its batching is faster, not slower: measured at
8972 // 56 ms with the `LIMIT` dropped, against 305 ms batched. That is why
8973 // this fires before the two assertions below — without it, removing the
8974 // batching starves the writer of attempts and gets reported as "invalid
8975 // measurement", blaming the test for the defect it just detected.
8976 //
8977 // **Partial coverage, measured rather than asserted.** Two mutations
8978 // that keep the loop looking roughly batched are caught only sometimes:
8979 //
8980 // `LIMIT` dropped (no batching at all) 3-4 runs in 5-6, MOSTLY by
8981 // the refusal assertion below,
8982 // not by this floor
8983 // `PRUNE_BATCH_HANDOFF` sleep removed 1 run in 5-6, by this floor
8984 //
8985 // (An earlier version attributed both to this floor. Re-measured: of
8986 // four catches of the `LIMIT` mutation in six runs, three panicked at
8987 // the refusal assertion and one here.)
8988 //
8989 // Both were caught more often — 5/5 and 3/5 — by the wall-clock form
8990 // this replaced. That is a real coverage loss and it was taken on
8991 // purpose: the wall-clock form FALSE-FAILED correct code, which is a
8992 // worse defect than missing a deliberate deletion of a commented line.
8993 // See the note on the assertion below for the measurement.
8994 //
8995 // Nothing here is tuned to make those two reliable. Doing so means
8996 // thresholding a rate, which is what this test has now been wrong about
8997 // three separate times.
8998 let handoff_floor = PRUNE_BATCH_HANDOFF * BATCHES as u32;
8999 assert!(
9000 sweep > handoff_floor,
9001 "the delete loop finished in {sweep:?}, under the {handoff_floor:?} that \
9002 {BATCHES} batches of `PRUNE_BATCH_HANDOFF` alone would take — it is not \
9003 handing the write lock over between batches at all"
9004 );
9005 // **Assert a RATIO OF TWO DURATIONS THAT SCALE TOGETHER.**
9006 //
9007 // Three thresholds have now failed here, each for the same reason: they
9008 // compared something machine-scaled against something fixed.
9009 //
9010 // `worst * 3 < sweep` — broke when the sweep got FASTER (the
9011 // `NOT EXISTS` rewrite, 1.49x) and tightened a
9012 // threshold calibrated against the slow version.
9013 // `wrote >= 10` — a raw count is writes-per-unit-time, so it
9014 // measured the runner. Flaked on CI at 4 writes.
9015 // `during * 2 >=` — a success FRACTION, which I claimed was
9016 // `attempts` scale-free. It is not, and this is the
9017 // important one, because the argument sounds
9018 // right. Successes come from the FIXED
9019 // `BATCHES * PRUNE_BATCH_HANDOFF` of open
9020 // window divided by write latency; failures
9021 // come from the machine-scaled lock hold
9022 // divided by the FIXED `WRITER_BUSY_TIMEOUT`.
9023 // Slow the machine by k and the fraction decays
9024 // as roughly 1/(1 + k²c) — quadratically,
9025 // toward failure. Measured with production code
9026 // fully correct and only the per-batch hold
9027 // grown 10x: **188/949 (19.8%) and 383/1028
9028 // (37.3%), two false failures in three runs**,
9029 // at loop durations of 3.4 s. CPU saturation
9030 // cannot find this — it slows writer and
9031 // sweeper together, which is the wrong axis.
9032 //
9033 // `max_gap * 2 <` — the longest WALL-CLOCK stretch with no write
9034 // `sweep` landing, against the loop's duration. Both
9035 // sides scale with the machine, which fixed the
9036 // fraction's problem and introduced a new one:
9037 // a gap opens when the writer is DESCHEDULED
9038 // just as surely as when the lock is held.
9039 // Observed under 4x CPU saturation, full suite:
9040 // `went 319.95ms of 609.83ms` — while **622 of
9041 // 626 attempts landed**. The lock was fine; the
9042 // writer task simply did not run for 320 ms.
9043 //
9044 // So count REFUSALS, not time. The longest unbroken run of `SQLITE_BUSY`
9045 // against the number of attempts made:
9046 //
9047 // handed over : the lock is free for `PRUNE_BATCH_HANDOFF` after every
9048 // batch, so the longest refused run is bounded by about
9049 // one batch's worth of attempts.
9050 // held across : every attempt in the window is refused — 100%.
9051 //
9052 // **The ~10% this comment used to quote for the handed-over case is not
9053 // what the shipped configuration produces.** Measured here: 0.001–0.05,
9054 // and in roughly a quarter of runs the writer is refused ZERO times
9055 // (`max_refused_run == 0`, every attempt landing), so the assertion is
9056 // vacuously true and certifies the hand-off by never observing one. That
9057 // is a weak test, not a wrong one — but it is worth knowing that the
9058 // enormous margin comes from the writer rarely colliding at all, not
9059 // from a measured 10%. 10% is what appears only once the per-batch hold
9060 // dominates the hand-off (`PRUNE_BATCH` x10 gives 0.115–0.143).
9061 //
9062 // This is immune to descheduling in a way no wall-clock measure can be:
9063 // a starved writer makes no attempts, so it contributes to neither side
9064 // of the ratio. Machine speed still cancels, because both sides are
9065 // counts of the same attempts. Re-checked against the failure above:
9066 // 622 of 626 landing means a refused run of at most 4, nowhere near the
9067 // 313 it would take to trip.
9068 //
9069 // **Detection is near all-or-nothing, and that is a known limit rather
9070 // than an oversight.** Holding one transaction across only the FIRST
9071 // HALF of the batches — production code otherwise correct — is not
9072 // caught at all:
9073 //
9074 // batches held in one tx (of 10) runs failing
9075 // 5 0 of 6 (ratios 0.05-0.27)
9076 // 7 2 of 6
9077 // 9 5 of 5
9078 // 10 22 of 22
9079 //
9080 // The ratio systematically UNDERSTATES the wall-clock fraction the lock
9081 // was held, because a refused attempt costs ~2 ms and leaves the sweeper
9082 // running uncontended, while a successful write actively blocks it and
9083 // stretches the loop. So attempts pile up during free time. The
9084 // "10% vs 100%" framing above describes the endpoints, not the curve.
9085 //
9086 // Closing that would mean measuring the wall-clock SPAN of a refusal run
9087 // rather than its length — which is most of the way back to `max_gap`,
9088 // the form that false-failed correct code on a descheduled writer. Given
9089 // this assertion has now been wrong four times in a row, and the current
9090 // one has zero false failures across 134 runs in six environments while
9091 // catching the real defect 22/22, a fifth redesign to catch a
9092 // half-transaction — a mutation no plausible edit produces — is not a
9093 // trade worth making. Stated here so the next reader knows the gap is
9094 // chosen, not missed.
9095 //
9096 // Measured, with the apparatus verified before each run:
9097 //
9098 // correct, 1x / 10x per-batch hold passes
9099 // one tx across batches, 1x CAUGHT — refused 128 of 129
9100 // one tx across batches, 10x CAUGHT — refused 1841 of 1843
9101 // `LIMIT` dropped caught 3 runs in 5 (by the floor)
9102 // hand-off sleep removed caught 1 run in 5 (by the floor)
9103 //
9104 // The last two were 5/5 and 3/5 under the wall-clock form. Losing that
9105 // is the price of not false-failing correct code, and it is the right
9106 // way round: the named defect is now caught by two orders of magnitude,
9107 // and the mutations that got weaker are deliberate deletions of lines
9108 // that carry their own explanation.
9109 assert!(
9110 attempts >= 20,
9111 "the writer only got {attempts} attempts inside a {sweep:?} delete loop \
9112 — too few for the ratio below to mean anything. That is USUALLY an \
9113 invalid measurement rather than a held lock, but note that a loop \
9114 holding the lock throughout is itself one cause of a starved writer, \
9115 so check {during} (landed) before concluding the test is at \
9116 fault{why}"
9117 );
9118 assert!(
9119 max_refused_run * 2 < attempts,
9120 "the delete loop refused {max_refused_run} consecutive write attempts out \
9121 of {attempts} ({during} landed) — a loop that hands the write lock over \
9122 between batches refuses at most about one batch's worth in a row; one \
9123 that holds the lock across them refuses nearly every attempt it sees{why}"
9124 );
9125
9126 pool.close().await;
9127 // `_tmp` removes the database, WAL and shm as it drops — on this path
9128 // and on the unwind from any assertion above.
9129 Ok(())
9130 }
9131
9132 /// **The sparing predicate must quantify over ALL DIDs, not just one.**
9133 ///
9134 /// `entry_state`'s primary key is `(did, entry_id)`, so several readers can
9135 /// hold rows on the same shared entry. The window spares an entry when ANY of
9136 /// them has starred it or left it unread — one person's star protects the
9137 /// cached copy everyone reads.
9138 ///
9139 /// This is the ONLY case where `id NOT IN (…)` and the correlated
9140 /// `NOT EXISTS` that replaced it could diverge, and it had no test. Every
9141 /// other retention test writes one `entry_state` row per entry under a single
9142 /// DID, where the two forms are trivially identical — so the claim that the
9143 /// suite made the equivalence executable was false when it was written. It is
9144 /// true now.
9145 #[tokio::test]
9146 async fn sparing_honours_every_did_not_just_one() -> Result<()> {
9147 let pool = init_url("sqlite::memory:").await?;
9148 let feed_id = upsert_feed(
9149 &pool,
9150 &NewFeed {
9151 url: "https://shared.example/f.xml".to_string(),
9152 ..Default::default()
9153 },
9154 )
9155 .await?;
9156 let old = (chrono::Utc::now() - chrono::Duration::days(400))
9157 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
9158 let guids = [
9159 "nobody-touched", // no state row at all -> evicted
9160 "both-read-unstarred", // two DIDs, both read+unstarred -> evicted
9161 "one-starred", // A read+unstarred, B starred -> SPARED by B
9162 "one-unread", // A read+unstarred, B unread -> SPARED by B
9163 ];
9164 let entries: Vec<NewEntry> = guids
9165 .iter()
9166 .map(|g| NewEntry {
9167 guid: (*g).to_string(),
9168 published: Some(old.clone()),
9169 ..Default::default()
9170 })
9171 .collect();
9172 insert_entries(&pool, feed_id, &entries, 0).await?;
9173
9174 let id_of = |g: &'static str| {
9175 let pool = pool.clone();
9176 async move {
9177 sqlx::query_scalar::<_, i64>("SELECT id FROM entries WHERE guid = ?1")
9178 .bind(g)
9179 .fetch_one(&pool)
9180 .await
9181 .unwrap()
9182 }
9183 };
9184 // (did, entry, read, starred)
9185 let rows: [(&str, &'static str, i64, i64); 6] = [
9186 ("did:plc:a", "both-read-unstarred", 1, 0),
9187 ("did:plc:b", "both-read-unstarred", 1, 0),
9188 ("did:plc:a", "one-starred", 1, 0),
9189 ("did:plc:b", "one-starred", 1, 1),
9190 ("did:plc:a", "one-unread", 1, 0),
9191 ("did:plc:b", "one-unread", 0, 0),
9192 ];
9193 for (did, guid, read, starred) in rows {
9194 let id = id_of(guid).await;
9195 sqlx::query(
9196 "INSERT INTO entry_state (did, entry_id, read, starred, updated_at) \
9197 VALUES (?1, ?2, ?3, ?4, '2026-01-01T00:00:00Z')",
9198 )
9199 .bind(did)
9200 .bind(id)
9201 .bind(read)
9202 .bind(starred)
9203 .execute(&pool)
9204 .await?;
9205 }
9206
9207 // Window only — no ceiling, so nothing is swept for age alone.
9208 let deleted = prune_old_entries(&pool, 30, 0, 0).await?;
9209 assert_eq!(
9210 deleted, 2,
9211 "expected the untouched and the all-read entries to go"
9212 );
9213
9214 let left: Vec<String> = sqlx::query_scalar("SELECT guid FROM entries ORDER BY guid")
9215 .fetch_all(&pool)
9216 .await?;
9217 assert_eq!(
9218 left,
9219 vec!["one-starred".to_string(), "one-unread".to_string()],
9220 "a second reader's star or unread mark must spare the SHARED entry"
9221 );
9222 Ok(())
9223 }
9224
9225 /// **A mark-read landing during the scrub must not be overwritten.**
9226 ///
9227 /// Moving the scrub out of the sweep's transaction removed a multi-minute
9228 /// write-lock hold and introduced a lost update in its place: the id-sets
9229 /// were read into a snapshot up front and written back unguarded, so a
9230 /// `mark_read` arriving mid-pass had its id silently dropped — and the
9231 /// rewrite set `dirty = 1`, so the flusher pushed the truncated set to the
9232 /// PDS as authoritative. Local `entry_state` still said read, so the loss was
9233 /// invisible here and visible only in every other atproto client.
9234 ///
9235 /// **⚠️ THIS TEST DOES NOT PROVE THAT, AND THE NAME NO LONGER CLAIMS IT.**
9236 ///
9237 /// The mark-read below lands BEFORE the scrub is called, not during it — so
9238 /// a snapshot-then-write implementation taking its snapshot at the top of
9239 /// `prune_orphan_cursor_ids` would see it too, and pass. The discriminator
9240 /// does not discriminate; what is actually pinned is the ordinary outcome:
9241 /// orphaned ids go, live ids stay.
9242 ///
9243 /// What the lost-update shape is really prevented by is a TYPE fact, not
9244 /// this test: `scrub_one_cursor(pool, did, feed_url)` is handed no id-sets,
9245 /// so it cannot write back anything but what it read itself, and
9246 /// re-introducing the bug means changing its signature.
9247 ///
9248 /// Proving it by test needs a real interleave — hold the write lock on a
9249 /// second connection, let the scrub block on it, commit a `mark_read`, then
9250 /// release — which needs a file-backed database and, without a hook inside
9251 /// the pass, a sleep to be sure the key snapshot has already run. A sleep is
9252 /// how this suite gets flaky in CI, and a flaky test is worse than an honest
9253 /// one, so it is left undone and written down instead.
9254 #[tokio::test]
9255 async fn the_cursor_scrub_drops_orphans_and_keeps_live_ids() -> Result<()> {
9256 let pool = init_url("sqlite::memory:").await?;
9257 let did = "did:plc:race";
9258 let feed_url = "https://race.example/f.xml";
9259 let feed_id = upsert_feed(
9260 &pool,
9261 &NewFeed {
9262 url: feed_url.to_string(),
9263 ..Default::default()
9264 },
9265 )
9266 .await?;
9267 insert_entries(
9268 &pool,
9269 feed_id,
9270 &[
9271 NewEntry {
9272 guid: "live".to_string(),
9273 ..Default::default()
9274 },
9275 NewEntry {
9276 guid: "doomed".to_string(),
9277 ..Default::default()
9278 },
9279 ],
9280 0,
9281 )
9282 .await?;
9283 replace_sub_refs(&pool, did, &[feed_id]).await?;
9284 let live_id: i64 = sqlx::query_scalar("SELECT id FROM entries WHERE guid = 'live'")
9285 .fetch_one(&pool)
9286 .await?;
9287 let doomed_id: i64 = sqlx::query_scalar("SELECT id FROM entries WHERE guid = 'doomed'")
9288 .fetch_one(&pool)
9289 .await?;
9290
9291 // A cursor holding only the id that is about to be deleted.
9292 upsert_cursor(
9293 &pool,
9294 &ReadCursor {
9295 did: did.to_string(),
9296 feed_url: feed_url.to_string(),
9297 read_through: None,
9298 read_ids: format!("[\"{doomed_id}\"]"),
9299 unread_ids: "[]".to_string(),
9300 dirty: false,
9301 pds_created: false,
9302 updated_at: now_rfc3339(),
9303 },
9304 )
9305 .await?;
9306 sqlx::query("DELETE FROM entries WHERE guid = 'doomed'")
9307 .execute(&pool)
9308 .await?;
9309
9310 // A reader marks the surviving entry read. NOTE this lands before the
9311 // scrub, not during it — see the caveat on this test. It is here because
9312 // the live id must survive the pass, not because it catches the race.
9313 mark_read(&pool, did, live_id, true).await?;
9314
9315 assert_eq!(prune_orphan_cursor_ids(&pool, None).await?, 1);
9316
9317 let cursor = get_cursor(&pool, did, feed_url).await?.expect("cursor");
9318 let ids: Vec<String> = serde_json::from_str(&cursor.read_ids)?;
9319 assert_eq!(
9320 ids,
9321 vec![live_id.to_string()],
9322 "the scrub dropped a live id"
9323 );
9324 assert!(
9325 !ids.contains(&doomed_id.to_string()),
9326 "the orphaned id survived the scrub"
9327 );
9328 Ok(())
9329 }
9330
9331 /// The cursor scrub still happens — it just no longer rides inside the
9332 /// delete transaction. Moving it out is only safe because it is idempotent;
9333 /// this pins that it still runs at all, which is the thing a "move it out"
9334 /// refactor can silently drop.
9335 #[tokio::test]
9336 async fn the_sweep_still_scrubs_orphaned_cursor_ids() -> Result<()> {
9337 let pool = init_url("sqlite::memory:").await?;
9338 let did = "did:plc:scrub";
9339 let feed_url = "https://scrub.example/f.xml";
9340 let feed_id = upsert_feed(
9341 &pool,
9342 &NewFeed {
9343 url: feed_url.to_string(),
9344 ..Default::default()
9345 },
9346 )
9347 .await?;
9348 let old = (chrono::Utc::now() - chrono::Duration::days(400))
9349 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
9350 insert_entries(
9351 &pool,
9352 feed_id,
9353 &[NewEntry {
9354 guid: "doomed".to_string(),
9355 published: Some(old),
9356 ..Default::default()
9357 }],
9358 0,
9359 )
9360 .await?;
9361 let doomed = entries_for_feed(&pool, did, feed_id).await;
9362 // `entries_for_feed` is sub_ref-scoped; read the id directly instead.
9363 drop(doomed);
9364 let doomed_id: i64 = sqlx::query_scalar("SELECT id FROM entries WHERE guid = 'doomed'")
9365 .fetch_one(&pool)
9366 .await?;
9367
9368 upsert_cursor(
9369 &pool,
9370 &ReadCursor {
9371 did: did.to_string(),
9372 feed_url: feed_url.to_string(),
9373 read_through: None,
9374 read_ids: format!("[\"{doomed_id}\"]"),
9375 unread_ids: "[]".to_string(),
9376 dirty: false,
9377 pds_created: false,
9378 updated_at: now_rfc3339(),
9379 },
9380 )
9381 .await?;
9382
9383 assert_eq!(prune_old_entries(&pool, 30, 180, 0).await?, 1);
9384
9385 let cursor = get_cursor(&pool, did, feed_url).await?.expect("cursor");
9386 let ids: Vec<String> = serde_json::from_str(&cursor.read_ids)?;
9387 assert!(
9388 ids.is_empty(),
9389 "the deleted entry's id survived in the cursor: {ids:?}"
9390 );
9391 assert!(cursor.dirty, "a rewritten cursor must be re-flushed");
9392 Ok(())
9393 }
9394
9395 #[tokio::test]
9396 async fn prune_and_reclaim_drops_db_size() -> Result<()> {
9397 // On-disk DB so VACUUM has a file to shrink.
9398 let dir = std::env::temp_dir();
9399 let path = dir.join(format!("fr-prune-{}.db", std::process::id()));
9400 let url = format!("sqlite://{}", path.display());
9401 let pool = init_url(&url).await?;
9402
9403 let feed_id = upsert_feed(
9404 &pool,
9405 &NewFeed {
9406 url: "https://bulk.example/feed.xml".to_string(),
9407 ..Default::default()
9408 },
9409 )
9410 .await?;
9411 let ancient = (chrono::Utc::now() - chrono::Duration::days(365))
9412 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
9413 let entries: Vec<NewEntry> = (0..2000)
9414 .map(|i| NewEntry {
9415 guid: format!("guid-{i}"),
9416 content_html: Some("<p>".to_string() + &"x".repeat(400) + "</p>"),
9417 published: Some(ancient.clone()),
9418 fetched_at: Some(ancient.clone()),
9419 ..Default::default()
9420 })
9421 .collect();
9422 insert_entries(&pool, feed_id, &entries, 0).await?;
9423 let full = db_size_bytes(&pool).await?;
9424 assert!(full > 0);
9425
9426 // A retention sweep prunes every (year-old) entry, then reclaim shrinks.
9427 let deleted = prune_old_entries(&pool, 90, 3650, 0).await?;
9428 assert_eq!(deleted, 2000);
9429 reclaim(&pool).await?;
9430 let after = db_size_bytes(&pool).await?;
9431 assert!(
9432 after < full,
9433 "prune + reclaim must shrink db_size_bytes: {after} !< {full}"
9434 );
9435
9436 drop(pool);
9437 let _ = std::fs::remove_file(&path);
9438 let _ = std::fs::remove_file(format!("{}-wal", path.display()));
9439 let _ = std::fs::remove_file(format!("{}-shm", path.display()));
9440 Ok(())
9441 }
9442
9443 // ---- B1: an existing PRE-0.2.2 invite_codes table (no intended_did) must
9444 // migrate cleanly, not crash-loop boot. ----------------------------------
9445
9446 #[tokio::test]
9447 async fn migrates_pre_intended_did_invite_codes_table() -> Result<()> {
9448 // Build an on-disk DB whose `invite_codes` table has the OLD 0.2.1 shape
9449 // (NO `intended_did` column, and therefore no `intended_did` index), then
9450 // run init_schema/migrations against it — this is exactly the existing-prod
9451 // volume that blocker B1 crash-looped (the SCHEMA's `CREATE INDEX ...
9452 // (intended_did, ...)` fired before the ALTER TABLE added the column).
9453 let dir = std::env::temp_dir();
9454 let path = dir.join(format!("fr-b1-{}.db", std::process::id()));
9455 let url = format!("sqlite://{}", path.display());
9456
9457 // Open a raw pool WITHOUT init_schema and hand-build the old table shape.
9458 let opts = SqliteConnectOptions::from_str(&url)?
9459 .create_if_missing(true)
9460 .foreign_keys(true);
9461 let pool = SqlitePoolOptions::new()
9462 .min_connections(1)
9463 .max_connections(1)
9464 .connect_with(opts)
9465 .await?;
9466 sqlx::query(
9467 r#"CREATE TABLE invite_codes (
9468 code TEXT PRIMARY KEY,
9469 creator_did TEXT NOT NULL,
9470 status TEXT NOT NULL,
9471 invitee_did TEXT,
9472 created_at INTEGER NOT NULL,
9473 expires_at INTEGER NOT NULL,
9474 redeemed_at INTEGER
9475 );"#,
9476 )
9477 .execute(&pool)
9478 .await?;
9479 // Seed a legacy active code so the migration runs against real data.
9480 sqlx::query(
9481 "INSERT INTO invite_codes (code, creator_did, status, created_at, expires_at) \
9482 VALUES ('FEATHER-LEGACY00', 'did:plc:old', 'active', 1, 9999999999)",
9483 )
9484 .execute(&pool)
9485 .await?;
9486
9487 // The column is genuinely absent to start with (pre-condition of B1).
9488 let cols: Vec<String> = sqlx::query("PRAGMA table_info(invite_codes)")
9489 .fetch_all(&pool)
9490 .await?
9491 .iter()
9492 .map(|r| r.get::<String, _>("name"))
9493 .collect();
9494 assert!(
9495 !cols.iter().any(|c| c == "intended_did"),
9496 "pre-condition: legacy table must lack intended_did"
9497 );
9498
9499 // THE FIX: init_schema must succeed (not error with "no such column").
9500 init_schema(&pool)
9501 .await
9502 .expect("init_schema on a pre-0.2.2 invite_codes table must not crash");
9503
9504 // Post-condition: the column now exists, both indexes were created, and the
9505 // legacy row is intact.
9506 let cols: Vec<String> = sqlx::query("PRAGMA table_info(invite_codes)")
9507 .fetch_all(&pool)
9508 .await?
9509 .iter()
9510 .map(|r| r.get::<String, _>("name"))
9511 .collect();
9512 assert!(cols.iter().any(|c| c == "intended_did"));
9513 let idx: Vec<String> = sqlx::query(
9514 "SELECT name FROM sqlite_master WHERE type='index' AND tbl_name='invite_codes'",
9515 )
9516 .fetch_all(&pool)
9517 .await?
9518 .iter()
9519 .map(|r| r.get::<String, _>("name"))
9520 .collect();
9521 assert!(idx.iter().any(|n| n == "idx_invite_codes_intended"));
9522 assert!(idx.iter().any(|n| n == "idx_invite_codes_intended_active"));
9523
9524 // Idempotent: running it again is a no-op, not an error.
9525 init_schema(&pool)
9526 .await
9527 .expect("re-running init_schema must be idempotent");
9528
9529 // **The OAuth tables must exist too.** They live in this database, and
9530 // creating them only when the Rust backend is selected would make the
9531 // first request after a cutover flip fail with "no such table" -- at the
9532 // one moment nobody wants to find out a migration was missed. They are
9533 // empty and harmless while the sidecar is serving.
9534 let tables: Vec<String> =
9535 sqlx::query_scalar("SELECT name FROM sqlite_master WHERE type = 'table'")
9536 .fetch_all(&pool)
9537 .await
9538 .unwrap();
9539 for table in ["oauth_state", "oauth_session", "oauth_nonce"] {
9540 assert!(
9541 tables.iter().any(|t| t == table),
9542 "{table} is missing, so the rust backend would fail on its first request: {tables:?}"
9543 );
9544 }
9545
9546 // The legacy code still redeems (NULL intended_did → open, as before).
9547 let out = redeem_code(&pool, "FEATHER-LEGACY00", "did:plc:new", None, 100).await?;
9548 assert_eq!(out, Ok(()));
9549
9550 drop(pool);
9551 let _ = std::fs::remove_file(&path);
9552 let _ = std::fs::remove_file(format!("{}-wal", path.display()));
9553 let _ = std::fs::remove_file(format!("{}-shm", path.display()));
9554 Ok(())
9555 }
9556
9557 // ---- 0.3.9: the schema a RELEASED binary left behind must upgrade. ------
9558 //
9559 // B1 above hand-built the old shape of ONE table, so it could only catch the
9560 // mistake it was written for. 0.3.9 made the same mistake on `feeds` — an
9561 // index in the base SCHEMA on `kind`, a column only `apply_migrations` adds
9562 // — and crash-looped production on its first boot, while every test here
9563 // passed, because every other test starts from an empty file. These start
9564 // from the schema a released binary actually created (dumped, not
9565 // transcribed), so they cover every table at once: v0.3.8, the release
9566 // before the bug, and v0.2.0, the oldest and furthest-migrated shape.
9567
9568 /// A fresh in-memory pool on ONE connection that never expires. The bug
9569 /// class is DDL order, which does not depend on a file, and a file named by
9570 /// pid leaks on a failed run and then fails the next run whose pid matches,
9571 /// at the fixture's first CREATE TABLE, before it tests anything.
9572 async fn upgrade_test_pool() -> Result<SqlitePool> {
9573 let opts = SqliteConnectOptions::from_str("sqlite::memory:")?.foreign_keys(true);
9574 Ok(SqlitePoolOptions::new()
9575 .min_connections(1)
9576 .max_connections(1)
9577 .idle_timeout(None)
9578 .max_lifetime(None)
9579 .connect_with(opts)
9580 .await?)
9581 }
9582
9583 /// Every table's columns (with type, NOT NULL, default and pk) and every
9584 /// index (with uniqueness, partiality and its columns in order), as one
9585 /// comparable set. Column ORDER is left out on purpose: `ALTER TABLE ADD
9586 /// COLUMN` appends, so a migrated table legitimately orders differently
9587 /// from a fresh one.
9588 async fn schema_shape(pool: &SqlitePool) -> Result<std::collections::BTreeSet<String>> {
9589 let mut shape = std::collections::BTreeSet::new();
9590 let tables: Vec<String> = sqlx::query_scalar(
9591 "SELECT name FROM sqlite_master WHERE type = 'table' AND name NOT LIKE 'sqlite_%'",
9592 )
9593 .fetch_all(pool)
9594 .await?;
9595 for t in tables {
9596 for r in sqlx::query(
9597 r#"SELECT name, type, "notnull", dflt_value, pk FROM pragma_table_info(?)"#,
9598 )
9599 .bind(&t)
9600 .fetch_all(pool)
9601 .await?
9602 {
9603 shape.insert(format!(
9604 "column {t}.{} {} notnull={} default={:?} pk={}",
9605 r.get::<String, _>("name"),
9606 r.get::<String, _>("type"),
9607 r.get::<i64, _>("notnull"),
9608 r.get::<Option<String>, _>("dflt_value"),
9609 r.get::<i64, _>("pk"),
9610 ));
9611 }
9612 for r in sqlx::query(r#"SELECT name, "unique", partial FROM pragma_index_list(?)"#)
9613 .bind(&t)
9614 .fetch_all(pool)
9615 .await?
9616 {
9617 let name: String = r.get("name");
9618 let cols: Vec<String> =
9619 sqlx::query_scalar("SELECT name FROM pragma_index_info(?) ORDER BY seqno")
9620 .bind(&name)
9621 .fetch_all(pool)
9622 .await?;
9623 shape.insert(format!(
9624 "index {t}.{name} unique={} partial={} ({})",
9625 r.get::<i64, _>("unique"),
9626 r.get::<i64, _>("partial"),
9627 cols.join(", "),
9628 ));
9629 }
9630 }
9631 Ok(shape)
9632 }
9633
9634 /// Load `fixture`, seed rows the way an old binary inserted them, run the
9635 /// current `init_schema`, and require the result to be indistinguishable
9636 /// in shape from a fresh database, with `kind` back-filled correctly.
9637 async fn assert_upgrades_from(version: &str, fixture: &'static str) -> Result<()> {
9638 let pool = upgrade_test_pool().await?;
9639 sqlx::raw_sql(fixture).execute(&pool).await?;
9640
9641 let has_kind = |pool: SqlitePool| async move {
9642 Ok::<_, anyhow::Error>(
9643 sqlx::query_scalar::<_, i64>(
9644 "SELECT count(*) FROM pragma_table_info('feeds') WHERE name = 'kind'",
9645 )
9646 .fetch_one(&pool)
9647 .await?
9648 == 1,
9649 )
9650 };
9651 assert!(
9652 !has_kind(pool.clone()).await?,
9653 "pre-condition: a {version} feeds table has no kind column"
9654 );
9655
9656 // One row of each kind, inserted the way the old binary did: without `kind`.
9657 let publication = "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab";
9658 for u in ["https://example.com/feed.xml", publication] {
9659 sqlx::query("INSERT INTO feeds (url) VALUES (?)")
9660 .bind(u)
9661 .execute(&pool)
9662 .await?;
9663 }
9664
9665 init_schema(&pool)
9666 .await
9667 .unwrap_or_else(|e| panic!("init_schema must upgrade a {version} database: {e:#}"));
9668
9669 let kinds: Vec<(String, String)> =
9670 sqlx::query_as("SELECT url, kind FROM feeds ORDER BY id")
9671 .fetch_all(&pool)
9672 .await?;
9673 assert_eq!(
9674 kinds,
9675 vec![
9676 (
9677 "https://example.com/feed.xml".to_string(),
9678 "rss".to_string()
9679 ),
9680 (publication.to_string(), "publication".to_string()),
9681 ],
9682 "{version}: existing rows are back-filled from their URL"
9683 );
9684 // What it indexes, not only its name: an `idx_feeds_kind` on the wrong
9685 // column passed a name check. (The shape comparison below also covers
9686 // this; this one names the bug that shipped.)
9687 let indexed: Vec<String> = sqlx::query_scalar(
9688 "SELECT name FROM pragma_index_info('idx_feeds_kind') ORDER BY seqno",
9689 )
9690 .fetch_all(&pool)
9691 .await?;
9692 assert_eq!(
9693 indexed,
9694 vec!["kind".to_string()],
9695 "{version}: idx_feeds_kind exists, on feeds(kind), after the column"
9696 );
9697
9698 // The general check: anything a fresh database has that the upgraded
9699 // one lacks, or the reverse, is a migration gap.
9700 let fresh = upgrade_test_pool().await?;
9701 init_schema(&fresh).await?;
9702 let (want, got) = (schema_shape(&fresh).await?, schema_shape(&pool).await?);
9703 assert!(
9704 want == got,
9705 "{version}: upgraded schema differs from a fresh one\n missing: {:#?}\n extra: {:#?}",
9706 want.difference(&got).collect::<Vec<_>>(),
9707 got.difference(&want).collect::<Vec<_>>(),
9708 );
9709
9710 // And a second boot over the upgraded database is a no-op, not an error.
9711 init_schema(&pool)
9712 .await
9713 .unwrap_or_else(|e| panic!("{version}: re-running init_schema failed: {e:#}"));
9714 Ok(())
9715 }
9716
9717 #[tokio::test]
9718 async fn a_v0_3_8_database_upgrades_to_the_current_schema() -> Result<()> {
9719 assert_upgrades_from(
9720 "v0.3.8",
9721 include_str!("../tests/fixtures/schema-v0.3.8.sql"),
9722 )
9723 .await
9724 }
9725
9726 #[tokio::test]
9727 async fn a_v0_2_0_database_upgrades_to_the_current_schema() -> Result<()> {
9728 assert_upgrades_from(
9729 "v0.2.0",
9730 include_str!("../tests/fixtures/schema-v0.2.0.sql"),
9731 )
9732 .await
9733 }
9734
9735 // ---- B2: a code minted FOR a specific DID is redeemable ONLY by that DID. --
9736
9737 #[tokio::test]
9738 async fn redeem_enforces_intended_did_binding() -> Result<()> {
9739 let pool = init_url("sqlite::memory:").await?;
9740 // Mint a claim FOR did:plc:A (the follower the bot posted the link to).
9741 let code = mint_code_for_did(&pool, "did:bot:fr", 3600, "did:plc:A").await?;
9742
9743 // A DIFFERENT DID (a throwaway that stole the public link) is refused as if
9744 // the code didn't exist — no seat granted, code still active.
9745 let stolen = redeem_code(&pool, &code, "did:plc:B", Some("thief.bsky"), 100).await?;
9746 assert_eq!(stolen, Err(RedeemError::NotFound));
9747 assert!(!has_beta_access(&pool, "did:plc:B").await?);
9748 assert_eq!(count_active_codes(&pool).await?, 1, "code must stay active");
9749
9750 // The INTENDED DID redeems successfully.
9751 let ok = redeem_code(&pool, &code, "did:plc:A", Some("alice.bsky"), 100).await?;
9752 assert_eq!(ok, Ok(()));
9753 assert!(has_beta_access(&pool, "did:plc:A").await?);
9754
9755 // A NULL-intended (admin/browser) code stays open to anyone (unchanged).
9756 let open = mint_code(&pool, "did:plc:admin", 3600).await?;
9757 let anyone = redeem_code(&pool, &open, "did:plc:C", None, 100).await?;
9758 assert_eq!(anyone, Ok(()));
9759 assert!(has_beta_access(&pool, "did:plc:C").await?);
9760 Ok(())
9761 }
9762
9763 // ---- S4: at most one ACTIVE code per intended DID; a concurrent second mint
9764 // hits the partial-unique index, and is_intended_active_conflict recognises it.
9765
9766 #[tokio::test]
9767 async fn intended_active_partial_unique_index_blocks_double_mint() -> Result<()> {
9768 let pool = init_url("sqlite::memory:").await?;
9769 // First mint for the DID succeeds.
9770 mint_code_for_did(&pool, "did:bot:fr", 3600, "did:plc:dup").await?;
9771 // A SECOND active mint for the SAME DID violates the partial unique index.
9772 let err = mint_code_for_did(&pool, "did:bot:fr", 3600, "did:plc:dup")
9773 .await
9774 .expect_err("second active mint for the same DID must fail the unique index");
9775 assert!(
9776 is_intended_active_conflict(&err),
9777 "the conflict must be recognised so the web layer can recover: {err:?}"
9778 );
9779 // Still exactly one active code for the DID.
9780 assert!(find_active_code_for_did(&pool, "did:plc:dup")
9781 .await?
9782 .is_some());
9783
9784 // Once the first code is redeemed (no longer active), a fresh mint for the
9785 // DID is allowed again (partial index only constrains active rows).
9786 let existing = find_active_code_for_did(&pool, "did:plc:dup")
9787 .await?
9788 .unwrap();
9789 redeem_code(&pool, &existing, "did:plc:dup", None, 100).await??;
9790 mint_code_for_did(&pool, "did:bot:fr", 3600, "did:plc:dup")
9791 .await
9792 .expect("a new mint is allowed after the prior one is redeemed");
9793
9794 // And the conflict helper does NOT fire on an unrelated error (a PRIMARY KEY
9795 // clash on `code`, i.e. a different constraint).
9796 sqlx::query(
9797 "INSERT INTO invite_codes (code, creator_did, status, created_at, expires_at) \
9798 VALUES ('FEATHER-DUPEKEY0', 'did:x', 'active', 1, 9999999999)",
9799 )
9800 .execute(&pool)
9801 .await?;
9802 let pk_err = sqlx::query(
9803 "INSERT INTO invite_codes (code, creator_did, status, created_at, expires_at) \
9804 VALUES ('FEATHER-DUPEKEY0', 'did:x', 'active', 1, 9999999999)",
9805 )
9806 .execute(&pool)
9807 .await
9808 .expect_err("duplicate PRIMARY KEY must error");
9809 let as_anyhow = anyhow::Error::new(pk_err);
9810 assert!(
9811 !is_intended_active_conflict(&as_anyhow),
9812 "a non-intended-index conflict must NOT be mistaken for the recover-able one"
9813 );
9814 Ok(())
9815 }
9816
9817 #[tokio::test]
9818 async fn purge_expires_orphaned_active_intended_code() -> Result<()> {
9819 // Cheap nit: purging a DID that is the TARGET of an active claim must both
9820 // NULL intended_did AND expire the (now orphaned) active code, so it stops
9821 // counting against the mint cap for its full TTL.
9822 let pool = init_url("sqlite::memory:").await?;
9823 let code = mint_code_for_did(&pool, "did:bot:fr", 3600, "did:plc:leaver").await?;
9824 assert_eq!(count_active_codes(&pool).await?, 1);
9825
9826 purge_did_data(&pool, "did:plc:leaver").await?;
9827
9828 // The code is no longer active (expired), so it no longer counts.
9829 assert_eq!(
9830 count_active_codes(&pool).await?,
9831 0,
9832 "orphaned code must be expired by purge, not left active"
9833 );
9834 // And intended_did was scrubbed.
9835 let intended: Option<String> =
9836 sqlx::query("SELECT intended_did FROM invite_codes WHERE code = ?1")
9837 .bind(&code)
9838 .fetch_one(&pool)
9839 .await?
9840 .get("intended_did");
9841 assert!(intended.is_none(), "intended_did must be NULLed");
9842 Ok(())
9843 }
9844
9845 /// A `(key, source)` observation upserts in place: two writes for the same
9846 /// relay leave ONE row, carrying the newer value.
9847 #[tokio::test]
9848 async fn network_stat_upserts_per_source() -> Result<()> {
9849 let pool = init_url("sqlite::memory:").await?;
9850 let mut stat = NetworkStat {
9851 key: ADOPTION_STAT_KEY.to_string(),
9852 source: "https://relay1.us-west.bsky.network".to_string(),
9853 value: 1,
9854 truncated: false,
9855 observed_at: "2026-08-12T00:00:00Z".to_string(),
9856 };
9857 record_network_stat(&pool, &stat).await?;
9858 stat.value = 4;
9859 stat.observed_at = "2026-08-13T00:00:00Z".to_string();
9860 record_network_stat(&pool, &stat).await?;
9861
9862 let rows: i64 = sqlx::query("SELECT COUNT(*) AS n FROM network_stat")
9863 .fetch_one(&pool)
9864 .await?
9865 .get("n");
9866 assert_eq!(rows, 1, "the same relay must update, not duplicate");
9867 let latest = latest_network_stat(&pool, ADOPTION_STAT_KEY)
9868 .await?
9869 .expect("a stat");
9870 assert_eq!(latest.value, 4);
9871 assert_eq!(latest.observed_at, "2026-08-13T00:00:00Z");
9872 Ok(())
9873 }
9874
9875 /// **Regression (v0.2.9 review).** Once a slow walk can return a PARTIAL
9876 /// count, a plain upsert lets it overwrite a complete, larger one — moving
9877 /// the published "at least N" DOWN because a relay was slow, not because
9878 /// adoption fell. A truncated observation may only ever raise the floor.
9879 #[tokio::test]
9880 async fn a_truncated_observation_never_lowers_a_stored_count() -> Result<()> {
9881 let pool = init_url("sqlite::memory:").await?;
9882 let mut stat = NetworkStat {
9883 key: ADOPTION_STAT_KEY.to_string(),
9884 source: "https://relay1.us-west.bsky.network".to_string(),
9885 value: 2000,
9886 truncated: false,
9887 observed_at: "2026-08-13T00:00:00Z".to_string(),
9888 };
9889 record_network_stat(&pool, &stat).await?;
9890
9891 // A budget-truncated walk that only got one page in.
9892 stat.value = 500;
9893 stat.truncated = true;
9894 stat.observed_at = "2026-08-14T00:00:00Z".to_string();
9895 record_network_stat(&pool, &stat).await?;
9896
9897 let kept = latest_network_stat(&pool, ADOPTION_STAT_KEY)
9898 .await?
9899 .expect("a stat");
9900 assert_eq!(kept.value, 2000, "a partial walk must not lower the count");
9901 assert!(!kept.truncated, "and must not mark the kept row truncated");
9902 assert_eq!(kept.observed_at, "2026-08-13T00:00:00Z");
9903
9904 // A truncated observation that RAISES the floor is still accepted...
9905 stat.value = 3000;
9906 record_network_stat(&pool, &stat).await?;
9907 assert_eq!(
9908 latest_network_stat(&pool, ADOPTION_STAT_KEY)
9909 .await?
9910 .expect("a stat")
9911 .value,
9912 3000
9913 );
9914
9915 // ...and a COMPLETE observation wins even when it is smaller, because
9916 // repos genuinely can go away and a full walk is authoritative.
9917 stat.value = 42;
9918 stat.truncated = false;
9919 record_network_stat(&pool, &stat).await?;
9920 assert_eq!(
9921 latest_network_stat(&pool, ADOPTION_STAT_KEY)
9922 .await?
9923 .expect("a stat")
9924 .value,
9925 42,
9926 "a complete walk is authoritative even when it shrinks"
9927 );
9928
9929 // An EQUAL-valued truncated observation must not downgrade the row
9930 // either: it proves nothing the stored complete count did not already
9931 // prove, but flipping `truncated` would silently degrade /about from
9932 // "42" to "at least 42" with no change in actual adoption. The strict
9933 // `<` in the guard let exactly this through — the equal case is the one
9934 // the two assertions above cannot reach, because both move the value.
9935 stat.truncated = true;
9936 stat.observed_at = "2026-08-15T00:00:00Z".to_string();
9937 record_network_stat(&pool, &stat).await?;
9938 let kept = latest_network_stat(&pool, ADOPTION_STAT_KEY)
9939 .await?
9940 .expect("a stat");
9941 assert_eq!(kept.value, 42);
9942 assert!(
9943 !kept.truncated,
9944 "an equal truncated observation must not mark the kept row truncated"
9945 );
9946 assert_eq!(
9947 kept.observed_at, "2026-08-14T00:00:00Z",
9948 "the rejected observation must not have rewritten the row at all"
9949 );
9950 Ok(())
9951 }
9952
9953 /// Relays disagree by design (non-archival indexes); the max is surfaced.
9954 #[tokio::test]
9955 async fn latest_network_stat_picks_the_max_across_sources() -> Result<()> {
9956 let pool = init_url("sqlite::memory:").await?;
9957 for (source, value, truncated) in [
9958 ("https://relay1.us-west.bsky.network", 2i64, false),
9959 ("https://relay1.us-east.bsky.network", 40i64, true),
9960 ] {
9961 record_network_stat(
9962 &pool,
9963 &NetworkStat {
9964 key: ADOPTION_STAT_KEY.to_string(),
9965 source: source.to_string(),
9966 value,
9967 truncated,
9968 observed_at: "2026-08-13T00:00:00Z".to_string(),
9969 },
9970 )
9971 .await?;
9972 }
9973 let latest = latest_network_stat(&pool, ADOPTION_STAT_KEY)
9974 .await?
9975 .expect("a stat");
9976 assert_eq!(latest.value, 40);
9977 assert_eq!(latest.source, "https://relay1.us-east.bsky.network");
9978 // `truncated` round-trips as a bool.
9979 assert!(latest.truncated);
9980 Ok(())
9981 }
9982
9983 #[tokio::test]
9984 async fn latest_network_stat_is_none_on_an_empty_table() -> Result<()> {
9985 let pool = init_url("sqlite::memory:").await?;
9986 assert!(latest_network_stat(&pool, ADOPTION_STAT_KEY)
9987 .await?
9988 .is_none());
9989 Ok(())
9990 }
9991
9992 /// **The lookup is keyed.** Every existing network-stat test writes only
9993 /// `ADOPTION_STAT_KEY`, so the `WHERE key = ?1` never discriminated; with
9994 /// it widened to `OR 1=1` the suite stayed green. The public `/stats`
9995 /// page asks for the adoption count, and unkeyed it would render the
9996 /// largest value of ANY stat as the network size.
9997 #[tokio::test]
9998 async fn latest_network_stat_ignores_other_keys() -> Result<()> {
9999 let pool = init_url("sqlite::memory:").await?;
10000 for (key, source, value) in [
10001 (ADOPTION_STAT_KEY, "https://relay1.example", 40),
10002 ("some.other.metric", "https://relay1.example", 9_999),
10003 ] {
10004 record_network_stat(
10005 &pool,
10006 &NetworkStat {
10007 key: key.to_string(),
10008 source: source.to_string(),
10009 value,
10010 truncated: false,
10011 observed_at: now_rfc3339(),
10012 },
10013 )
10014 .await?;
10015 }
10016 let latest = latest_network_stat(&pool, ADOPTION_STAT_KEY)
10017 .await?
10018 .expect("the adoption stat was recorded");
10019 assert_eq!(
10020 latest.value, 40,
10021 "another key's value was returned as the adoption count"
10022 );
10023 Ok(())
10024 }
10025
10026 // ── poll health (the public stats page) ─────────────────────────────────
10027
10028 /// Seed a feed row **through the real writer**, so its `kind` is whatever
10029 /// production would store.
10030 ///
10031 /// This used to be a raw `INSERT`, which took the `kind` column's
10032 /// `DEFAULT 'rss'`. That is correct for an http(s) URL and silently wrong
10033 /// for an `at://` one — the helper claimed to seed a row the poller skips
10034 /// while seeding one it selects.
10035 async fn feed_polled(
10036 pool: &SqlitePool,
10037 url: &str,
10038 last_polled: Option<&str>,
10039 next_poll: Option<&str>,
10040 ) {
10041 upsert_feed(
10042 pool,
10043 &NewFeed {
10044 url: url.to_string(),
10045 last_polled: last_polled.map(str::to_string),
10046 next_poll: next_poll.map(str::to_string),
10047 ..Default::default()
10048 },
10049 )
10050 .await
10051 .unwrap();
10052 }
10053
10054 /// The numbers on the public page must describe the poller's actual state.
10055 #[tokio::test]
10056 async fn poll_health_counts_tracked_recent_and_overdue() -> anyhow::Result<()> {
10057 let pool = init_url("sqlite::memory:").await?;
10058 let now = "2026-01-01T12:00:00Z";
10059 let hour_ago = "2026-01-01T11:00:00Z";
10060
10061 // Polled 10 minutes ago, due in 50 minutes: healthy.
10062 feed_polled(
10063 &pool,
10064 "https://a.example/f",
10065 Some("2026-01-01T11:50:00Z"),
10066 Some("2026-01-01T12:50:00Z"),
10067 )
10068 .await;
10069 // Polled 3 hours ago and overdue: the backlog case.
10070 feed_polled(
10071 &pool,
10072 "https://b.example/f",
10073 Some("2026-01-01T09:00:00Z"),
10074 Some("2026-01-01T10:00:00Z"),
10075 )
10076 .await;
10077 // Never polled: counts as overdue (next_poll IS NULL), and must not
10078 // corrupt the "oldest poll" figure with a NULL.
10079 feed_polled(&pool, "https://c.example/f", None, None).await;
10080
10081 let h = poll_health(&pool, now, hour_ago).await?;
10082 assert_eq!(h.feeds_tracked, 3);
10083 assert_eq!(
10084 h.polled_last_hour, 1,
10085 "only the 11:50 poll is within the hour"
10086 );
10087 assert_eq!(h.overdue, 2, "the stale feed and the never-polled one");
10088 assert_eq!(
10089 h.last_poll_secs_ago,
10090 Some(600),
10091 "most recent poll was 10 minutes ago"
10092 );
10093 // **A never-polled feed IS the worst staleness.**
10094 //
10095 // This originally asserted `Some(10_800)` — the oldest FINITE age — and
10096 // in doing so pinned a defect: `MIN` skips NULLs, so the page reported
10097 // "3h ago" while a quarter of the feeds had never been fetched at all.
10098 // The figure read healthiest in the most degraded state, which is the
10099 // opposite of what a health page is for.
10100 assert_eq!(
10101 h.oldest_poll_secs_ago, None,
10102 "a never-polled feed must outrank any finite age"
10103 );
10104 assert_eq!(h.never_polled, 1);
10105
10106 // With every feed polled, the finite worst case is reported again.
10107 sqlx::query("UPDATE feeds SET last_polled = ?1 WHERE last_polled IS NULL")
10108 .bind("2026-01-01T09:00:00Z")
10109 .execute(&pool)
10110 .await?;
10111 let h = poll_health(&pool, now, hour_ago).await?;
10112 assert_eq!(h.never_polled, 0);
10113 assert_eq!(h.oldest_poll_secs_ago, Some(10_800));
10114 Ok(())
10115 }
10116
10117 /// **`/stats` measures the poller, so it counts only what the poller sees.**
10118 ///
10119 /// `due_feeds` skips `at://` rows; nothing ever advances their `next_poll`
10120 /// or sets `last_polled`. Counted, they read as overdue and never-polled
10121 /// forever, and force "oldest poll" to `never` — the same "unsupported
10122 /// shown as broken" the exclusion exists to end, moved to different rows on
10123 /// a public page. The same predicate decides both queries so they cannot
10124 /// drift.
10125 #[tokio::test]
10126 async fn poll_health_ignores_unpollable_at_uri_rows() -> anyhow::Result<()> {
10127 let pool = init_url("sqlite::memory:").await?;
10128 let now = "2026-01-01T12:00:00Z";
10129 let hour_ago = "2026-01-01T11:00:00Z";
10130 feed_polled(
10131 &pool,
10132 "https://a.example/f",
10133 Some("2026-01-01T11:50:00Z"),
10134 Some("2026-01-01T12:50:00Z"),
10135 )
10136 .await;
10137 // Never polled, never due: the shape every at:// row has.
10138 feed_polled(
10139 &pool,
10140 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/app.bsky.feed.post/3lab",
10141 None,
10142 None,
10143 )
10144 .await;
10145
10146 let h = poll_health(&pool, now, hour_ago).await?;
10147 assert_eq!(
10148 h.feeds_tracked, 1,
10149 "an unpollable row was counted as tracked"
10150 );
10151 assert_eq!(h.overdue, 0, "an unpollable row was counted as overdue");
10152 assert_eq!(
10153 h.never_polled, 0,
10154 "an unpollable row was counted as never polled"
10155 );
10156 assert_eq!(
10157 h.oldest_poll_secs_ago,
10158 Some(600),
10159 "an unpollable row forced the oldest poll to `never`"
10160 );
10161 assert_eq!(h.polled_last_hour, 1);
10162 Ok(())
10163 }
10164
10165 /// **The admin's failing-feeds list is the poller's too.** `failing_feeds`
10166 /// feeds `/admin/metrics`; it was not given the exclusion both `/stats`
10167 /// queries got. An `at://` row that carries errors — from a rollback to a
10168 /// build that polled them, say — would then sit at the top of the one page
10169 /// an operator uses to diagnose "unsupported shown as broken", with no
10170 /// poll ever coming to clear it and the one-shot migration already spent.
10171 #[tokio::test]
10172 async fn failing_feeds_ignores_unpollable_at_uri_rows() -> anyhow::Result<()> {
10173 let pool = init_url("sqlite::memory:").await?;
10174 for url in [
10175 "https://broken.example/feed.xml",
10176 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/app.bsky.feed.post/3lab",
10177 ] {
10178 upsert_feed(
10179 &pool,
10180 &NewFeed {
10181 url: url.to_string(),
10182 ..Default::default()
10183 },
10184 )
10185 .await?;
10186 bump_feed_errors(&pool, url, crate::feed::FailureKind::Fetch, "down").await?;
10187 }
10188 let failing = failing_feeds(&pool, 10).await?;
10189 let urls: Vec<&str> = failing.iter().map(|f| f.url.as_str()).collect();
10190 assert_eq!(
10191 urls,
10192 vec!["https://broken.example/feed.xml"],
10193 "an unpollable row was listed as a failing feed"
10194 );
10195 Ok(())
10196 }
10197
10198 /// **The clearing is idempotent by predicate, not by stamp.** It touches
10199 /// only rows that have never been polled successfully: `bump_feed_errors`
10200 /// never sets `last_polled`, both success paths do. So a row a wired
10201 /// reader has fetched once keeps its later failures across restarts, and
10202 /// a row that only ever failed under our own refusal is cleared at every
10203 /// boot — including after a rollback to a build that polled it. No
10204 /// version stamp, nothing for a test to rewind.
10205 #[tokio::test]
10206 async fn the_at_uri_error_clearing_spares_a_row_that_has_been_polled() -> anyhow::Result<()> {
10207 let pool = init_url("sqlite::memory:").await?;
10208 let polled = "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/app.bsky.feed.post/polled";
10209 let never = "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/app.bsky.feed.post/never";
10210 for url in [polled, never] {
10211 upsert_feed(
10212 &pool,
10213 &NewFeed {
10214 url: url.to_string(),
10215 ..Default::default()
10216 },
10217 )
10218 .await?;
10219 bump_feed_errors(&pool, url, crate::feed::FailureKind::Fetch, "down").await?;
10220 }
10221 // A wired reader fetched this one once, then it started failing.
10222 sqlx::query("UPDATE feeds SET last_polled = '2026-01-01T00:00:00Z' WHERE url = ?1")
10223 .bind(polled)
10224 .execute(&pool)
10225 .await?;
10226
10227 for boot in 1..=2 {
10228 apply_migrations(&pool).await?;
10229 let mut errors = std::collections::HashMap::new();
10230 for url in [polled, never] {
10231 let n: i64 =
10232 sqlx::query_scalar("SELECT consecutive_errors FROM feeds WHERE url = ?1")
10233 .bind(url)
10234 .fetch_one(&pool)
10235 .await?;
10236 errors.insert(url, n);
10237 }
10238 assert_eq!(
10239 errors[polled], 1,
10240 "boot {boot} wiped a polled row's failure"
10241 );
10242 assert_eq!(
10243 errors[never], 0,
10244 "boot {boot} left a never-polled row failing"
10245 );
10246 }
10247 Ok(())
10248 }
10249
10250 /// **The SQL kind list and the Rust one are the same list.** A literal in
10251 /// SQL and a slice in Rust is the drift the column exists to end; wiring
10252 /// the standard.site reader changes both, and this is what makes
10253 /// forgetting one a failure rather than a silently dormant feature.
10254 #[test]
10255 fn the_sql_kind_list_matches_the_rust_one() {
10256 let expected = crate::feed::FeedKind::POLLABLE
10257 .iter()
10258 .map(|k| format!("'{}'", k.as_str()))
10259 .collect::<Vec<_>>()
10260 .join(", ");
10261 assert_eq!(POLLABLE_KINDS_SQL, expected);
10262 }
10263
10264 /// **A feed's kind is recorded at insert, not re-derived from its URL.**
10265 ///
10266 /// "Can the poller fetch this?" was a substring predicate spliced into
10267 /// four statements, and a review found a fifth reader that had drifted
10268 /// from it. A column the writers set cannot drift: the Rust side decides
10269 /// once, SQL reads a value.
10270 #[tokio::test]
10271 async fn a_feed_row_records_its_kind_at_insert() -> anyhow::Result<()> {
10272 let pool = init_url("sqlite::memory:").await?;
10273 for (url, want) in [
10274 ("https://real.example/feed.xml", crate::feed::FeedKind::Rss),
10275 ("http://real.example/feed.xml", crate::feed::FeedKind::Rss),
10276 (
10277 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
10278 crate::feed::FeedKind::Publication,
10279 ),
10280 ] {
10281 upsert_feed(
10282 &pool,
10283 &NewFeed {
10284 url: url.to_string(),
10285 ..Default::default()
10286 },
10287 )
10288 .await?;
10289 let got: String = sqlx::query_scalar("SELECT kind FROM feeds WHERE url = ?1")
10290 .bind(url)
10291 .fetch_one(&pool)
10292 .await?;
10293 assert_eq!(got, want.as_str(), "wrong kind recorded for {url}");
10294 }
10295 Ok(())
10296 }
10297
10298 /// **A row written before the column existed is back-filled from its URL.**
10299 /// That back-fill is the LAST use of the string predicate; every reader
10300 /// keys on `kind` afterwards.
10301 #[tokio::test]
10302 async fn the_migration_backfills_kind_from_the_url() -> anyhow::Result<()> {
10303 let pool = init_url("sqlite::memory:").await?;
10304 // A table that predates the column, with both shapes in it.
10305 sqlx::query("DROP TABLE feeds").execute(&pool).await?;
10306 sqlx::query(
10307 "CREATE TABLE feeds (
10308 id INTEGER PRIMARY KEY AUTOINCREMENT,
10309 url TEXT NOT NULL UNIQUE,
10310 title TEXT, site_url TEXT, etag TEXT, last_modified TEXT,
10311 last_polled TEXT, next_poll TEXT,
10312 consecutive_errors INTEGER NOT NULL DEFAULT 0,
10313 last_error_kind TEXT, last_error TEXT
10314 )",
10315 )
10316 .execute(&pool)
10317 .await?;
10318 for url in [
10319 "https://real.example/feed.xml",
10320 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
10321 "At://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lac",
10322 ] {
10323 sqlx::query("INSERT INTO feeds (url) VALUES (?1)")
10324 .bind(url)
10325 .execute(&pool)
10326 .await?;
10327 }
10328
10329 apply_migrations(&pool).await?;
10330
10331 let kinds: Vec<(String, String)> =
10332 sqlx::query_as("SELECT url, kind FROM feeds ORDER BY url")
10333 .fetch_all(&pool)
10334 .await?;
10335 let by_url: std::collections::HashMap<_, _> = kinds.into_iter().collect();
10336 assert_eq!(by_url["https://real.example/feed.xml"], "rss");
10337 assert_eq!(
10338 by_url["at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab"],
10339 "publication"
10340 );
10341 // Recognised as an at-URI (not `rss`), like every other guard does —
10342 // and, since 0.4.0 polls publications, classed `unsupported`: storage
10343 // refuses this spelling (#183), so polling it would fail every tick.
10344 assert_eq!(
10345 by_url["At://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lac"],
10346 "unsupported",
10347 "the back-fill must recognise a non-canonical spelling, and not poll it"
10348 );
10349 Ok(())
10350 }
10351
10352 /// **`feeds.kind` is derived from the URL, so it has to be re-derivable.**
10353 ///
10354 /// The back-fill translated one direction only — a row the Rust side would
10355 /// call `rss` was never touched — which is correct for a one-time migration
10356 /// and wrong for a column that has to survive the rule changing. A kind that
10357 /// disagrees with its own URL is currently permanent: nothing re-reads it.
10358 #[tokio::test]
10359 async fn the_back_fill_corrects_a_kind_that_disagrees_with_the_url() -> anyhow::Result<()> {
10360 let pool = init_url("sqlite::memory:").await?;
10361 let at = "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab";
10362 for (url, wrong) in [
10363 ("https://real.example/feed.xml", "publication"),
10364 (at, "rss"),
10365 ] {
10366 sqlx::query("INSERT INTO feeds (url, kind) VALUES (?1, ?2)")
10367 .bind(url)
10368 .bind(wrong)
10369 .execute(&pool)
10370 .await?;
10371 }
10372
10373 apply_migrations(&pool).await?;
10374
10375 let by_url: std::collections::HashMap<String, String> =
10376 sqlx::query_as("SELECT url, kind FROM feeds")
10377 .fetch_all(&pool)
10378 .await?
10379 .into_iter()
10380 .collect();
10381 assert_eq!(
10382 by_url["https://real.example/feed.xml"], "rss",
10383 "an http feed marked as a publication stayed one, and nothing polls it"
10384 );
10385 assert_eq!(by_url[at], "publication", "the at:// direction regressed");
10386 Ok(())
10387 }
10388
10389 /// **Taking a row out of the poller orphans its poll state, so clear it.**
10390 ///
10391 /// `last_polled` is set here on purpose: the migration's other cleanup step
10392 /// only clears rows we never polled, so a row that HAS been polled proves
10393 /// this reset is the one doing the work. An error count left on a row the
10394 /// scheduler will never select again is hidden from `/stats`, which filters
10395 /// on kind — and if a later rule change readmits the row, it resumes at a
10396 /// backoff earned under a classification that no longer applies.
10397 #[tokio::test]
10398 async fn a_row_taken_out_of_the_poller_loses_the_poll_state_it_cannot_use() -> anyhow::Result<()>
10399 {
10400 let pool = init_url("sqlite::memory:").await?;
10401 let at = "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/app.bsky.feed.post/3lab";
10402 sqlx::query(
10403 "INSERT INTO feeds (url, kind, consecutive_errors, last_error_kind, last_error, \
10404 next_poll, last_polled) \
10405 VALUES (?1, 'rss', 7, 'fetch', 'connection refused', ?2, ?3)",
10406 )
10407 .bind(at)
10408 .bind("2026-09-10T00:00:00Z")
10409 .bind("2026-09-01T00:00:00Z")
10410 .execute(&pool)
10411 .await?;
10412
10413 apply_migrations(&pool).await?;
10414
10415 let (kind, errors, error_kind, error, next_poll): (
10416 String,
10417 i64,
10418 Option<String>,
10419 Option<String>,
10420 Option<String>,
10421 ) = sqlx::query_as(
10422 "SELECT kind, consecutive_errors, last_error_kind, last_error, next_poll \
10423 FROM feeds WHERE url = ?1",
10424 )
10425 .bind(at)
10426 .fetch_one(&pool)
10427 .await?;
10428 assert_eq!(kind, "unsupported", "the row was not reclassified at all");
10429 assert_eq!(
10430 (errors, error_kind, error, next_poll),
10431 (0, None, None, None),
10432 "a row the scheduler will never select again kept its backoff and failure history"
10433 );
10434 Ok(())
10435 }
10436
10437 /// **A row we cannot read must not stop the process from starting.**
10438 ///
10439 /// This runs on the boot path. Refusing to start is a strictly worse
10440 /// outcome than declining to have an opinion about one row, and it is a
10441 /// failure mode the SQL predicate this replaced did not have: it evaluated
10442 /// a non-text `url` happily and returned false.
10443 #[tokio::test]
10444 async fn an_unreadable_feeds_row_does_not_stop_the_boot() -> anyhow::Result<()> {
10445 let pool = init_url("sqlite::memory:").await?;
10446 sqlx::query("INSERT INTO feeds (url, kind) VALUES (X'ff41', 'rss')")
10447 .execute(&pool)
10448 .await?;
10449 sqlx::query("INSERT INTO feeds (url, kind) VALUES (?1, 'publication')")
10450 .bind("https://real.example/feed.xml")
10451 .execute(&pool)
10452 .await?;
10453
10454 apply_migrations(&pool).await?;
10455
10456 let corrected: String = sqlx::query_scalar("SELECT kind FROM feeds WHERE url = ?1")
10457 .bind("https://real.example/feed.xml")
10458 .fetch_one(&pool)
10459 .await?;
10460 assert_eq!(
10461 corrected, "rss",
10462 "one unreadable row aborted the pass before the readable ones were corrected"
10463 );
10464 let untouched: String =
10465 sqlx::query_scalar("SELECT kind FROM feeds WHERE typeof(url) = 'blob'")
10466 .fetch_one(&pool)
10467 .await?;
10468 assert_eq!(
10469 untouched, "rss",
10470 "a row we declined to classify was classified anyway"
10471 );
10472 Ok(())
10473 }
10474
10475 /// Re-subscribing must re-derive the kind, not preserve whatever is there.
10476 ///
10477 /// `upsert_feed` binds `FeedKind::of` on the way in, but its conflict clause
10478 /// never carried `kind`, so the value a row was first written with is the
10479 /// value it keeps. Harmless while the rule is fixed; the rule is about to
10480 /// change.
10481 #[tokio::test]
10482 async fn a_re_upsert_re_derives_the_kind() -> anyhow::Result<()> {
10483 let pool = init_url("sqlite::memory:").await?;
10484 let url = "https://real.example/feed.xml";
10485 let feed = NewFeed {
10486 url: url.to_string(),
10487 ..Default::default()
10488 };
10489 upsert_feed(&pool, &feed).await?;
10490 sqlx::query("UPDATE feeds SET kind = 'publication' WHERE url = ?1")
10491 .bind(url)
10492 .execute(&pool)
10493 .await?;
10494
10495 upsert_feed(&pool, &feed).await?;
10496
10497 let kind: String = sqlx::query_scalar("SELECT kind FROM feeds WHERE url = ?1")
10498 .bind(url)
10499 .fetch_one(&pool)
10500 .await?;
10501 assert_eq!(
10502 kind, "rss",
10503 "a second subscription to the same URL kept the stale classification"
10504 );
10505 Ok(())
10506 }
10507
10508 /// **The readers key on `kind`, not on the URL.** A row whose kind says
10509 /// publication is unpollable even if its URL looks ordinary — which is
10510 /// what makes the column, rather than the string, the source of truth.
10511 #[tokio::test]
10512 async fn the_poller_and_the_pages_key_on_kind() -> anyhow::Result<()> {
10513 let pool = init_url("sqlite::memory:").await?;
10514 upsert_feed(
10515 &pool,
10516 &NewFeed {
10517 url: "https://looks-ordinary.example/feed.xml".to_string(),
10518 ..Default::default()
10519 },
10520 )
10521 .await?;
10522 // Force the kind independently of the URL: only the column should matter.
10523 // `unsupported` because it is the kind no poller reads (publications
10524 // are pollable since 0.4.0).
10525 sqlx::query("UPDATE feeds SET kind = 'unsupported' WHERE url LIKE 'https://looks%'")
10526 .execute(&pool)
10527 .await?;
10528
10529 let due = due_feeds(&pool, "2026-01-01T12:00:00Z", 10).await?;
10530 assert!(due.is_empty(), "due_feeds read the URL, not the kind");
10531 assert_eq!(
10532 unpollable_feeds(&pool).await?,
10533 1,
10534 "unpollable_feeds read the URL"
10535 );
10536
10537 let h = poll_health(&pool, "2026-01-01T12:00:00Z", "2026-01-01T11:00:00Z").await?;
10538 assert_eq!(h.feeds_tracked, 0, "poll_health read the URL, not the kind");
10539 Ok(())
10540 }
10541
10542 /// **SQL and Rust agree on what an at-URI is — case-insensitively.**
10543 ///
10544 /// This test used to pin the opposite, and pinned a bug. It asserted that a
10545 /// mixed-case `At://` row IS handed to the poller, reasoning that the Rust
10546 /// guards use a case-sensitive `strip_prefix` so "every other check treats
10547 /// it as a plain URL". They do not: URL schemes are case-insensitive, so
10548 /// `Url::parse` folds `At://` to scheme `at`, which `net::check_scheme`
10549 /// refuses — and the DID form does not parse at all. Such a row can only
10550 /// fail, every tick, forever, and be published in the `fetch` bucket as an
10551 /// unreachable publisher. That is the exact conflation the exclusion exists
10552 /// to end.
10553 ///
10554 /// Recognition is case-insensitive on both sides now. Storing one is still
10555 /// refused: `feeds.url` is UNIQUE, so two spellings of one publication are
10556 /// two rows — the same rule the canonical-handle check applies.
10557 #[tokio::test]
10558 async fn a_mixed_case_at_uri_is_unpollable_on_both_sides() -> anyhow::Result<()> {
10559 let pool = init_url("sqlite::memory:").await?;
10560 let odd = "At://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab";
10561 assert!(
10562 !crate::feed::is_storable_feed_url(odd, true),
10563 "a non-canonical spelling must not be storable"
10564 );
10565 upsert_feed(
10566 &pool,
10567 &NewFeed {
10568 url: odd.to_string(),
10569 ..Default::default()
10570 },
10571 )
10572 .await?;
10573 let due = due_feeds(&pool, "2026-01-01T12:00:00Z", 10).await?;
10574 assert!(
10575 due.is_empty(),
10576 "a row nothing can fetch was handed to the poller: {:?}",
10577 due.iter().map(|f| &f.url).collect::<Vec<_>>()
10578 );
10579
10580 // And the boot-time clearing reaches it, so a legacy row that already
10581 // accrued errors stops counting as a broken publisher.
10582 bump_feed_errors(&pool, odd, crate::feed::FailureKind::Fetch, "refused").await?;
10583 apply_migrations(&pool).await?;
10584 let n: i64 = sqlx::query_scalar("SELECT consecutive_errors FROM feeds WHERE url = ?1")
10585 .bind(odd)
10586 .fetch_one(&pool)
10587 .await?;
10588 assert_eq!(n, 0, "the clearing skipped a mixed-case at-URI row");
10589 Ok(())
10590 }
10591
10592 /// **The global feeds ceiling counts every row, including unpollable ones
10593 /// — deliberately, and visibly.**
10594 ///
10595 /// `count_feeds` is a fifth reader of "is this an at-URI" that does NOT use
10596 /// the unpollable kinds, and that is the right call: the ceiling bounds
10597 /// STORAGE on a small box, and an unpollable row occupies a row. What was
10598 /// wrong is that the capacity it consumed appeared on no surface — `/stats`
10599 /// measures the poller and excludes them, so an operator could be at the
10600 /// cap while every page said otherwise. `unpollable_feeds` is what
10601 /// `/admin/metrics` renders to close that gap.
10602 #[tokio::test]
10603 async fn the_ceiling_counts_unpollable_rows_and_they_are_countable() -> anyhow::Result<()> {
10604 let pool = init_url("sqlite::memory:").await?;
10605 for url in [
10606 "https://real.example/feed.xml",
10607 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/app.bsky.feed.post/3lab",
10608 "At://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lac",
10609 ] {
10610 upsert_feed(
10611 &pool,
10612 &NewFeed {
10613 url: url.to_string(),
10614 ..Default::default()
10615 },
10616 )
10617 .await?;
10618 }
10619 assert_eq!(
10620 count_feeds(&pool).await?,
10621 3,
10622 "the ceiling must bound storage, so every row counts"
10623 );
10624 assert_eq!(
10625 unpollable_feeds(&pool).await?,
10626 2,
10627 "both at-URI spellings are unpollable and must be countable"
10628 );
10629 Ok(())
10630 }
10631
10632 /// A fresh instance has no polls yet. The page must say so rather than
10633 /// rendering a zero that reads as "polled just now".
10634 #[tokio::test]
10635 async fn poll_health_on_an_empty_instance_reports_no_polls() -> anyhow::Result<()> {
10636 let pool = init_url("sqlite::memory:").await?;
10637 let h = poll_health(&pool, "2026-01-01T12:00:00Z", "2026-01-01T11:00:00Z").await?;
10638 assert_eq!(h.feeds_tracked, 0);
10639 assert_eq!(h.last_poll_secs_ago, None);
10640 assert_eq!(h.oldest_poll_secs_ago, None);
10641 Ok(())
10642 }
10643
10644 /// A poll timestamped in the future — clock skew, or a restored backup —
10645 /// reads as "just now", never as a negative age.
10646 #[tokio::test]
10647 async fn a_future_poll_timestamp_does_not_go_negative() -> anyhow::Result<()> {
10648 let pool = init_url("sqlite::memory:").await?;
10649 feed_polled(
10650 &pool,
10651 "https://a.example/f",
10652 Some("2026-01-01T13:00:00Z"),
10653 None,
10654 )
10655 .await;
10656 let h = poll_health(&pool, "2026-01-01T12:00:00Z", "2026-01-01T11:00:00Z").await?;
10657 assert_eq!(h.last_poll_secs_ago, Some(0));
10658 Ok(())
10659 }
10660
10661 // ── retention is a CACHE policy, not a data-retention policy ────────────
10662
10663 async fn aged_entry(pool: &SqlitePool, url: &str, days_old: i64) -> i64 {
10664 let when = (chrono::Utc::now() - chrono::Duration::days(days_old))
10665 .to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
10666 sqlx::query("INSERT INTO feeds (url) VALUES (?1) ON CONFLICT(url) DO NOTHING")
10667 .bind("https://f.example/feed")
10668 .execute(pool)
10669 .await
10670 .unwrap();
10671 let feed_id: i64 = sqlx::query_scalar("SELECT id FROM feeds WHERE url = ?1")
10672 .bind("https://f.example/feed")
10673 .fetch_one(pool)
10674 .await
10675 .unwrap();
10676 sqlx::query("INSERT INTO entries (feed_id, guid, url, title, published, fetched_at) VALUES (?1,?2,?3,'t',?4,?4)")
10677 .bind(feed_id).bind(url).bind(url).bind(&when)
10678 .execute(pool).await.unwrap();
10679 sqlx::query_scalar("SELECT id FROM entries WHERE guid = ?1")
10680 .bind(url)
10681 .fetch_one(pool)
10682 .await
10683 .unwrap()
10684 }
10685
10686 async fn mark(pool: &SqlitePool, entry_id: i64, read: i64, starred: i64) {
10687 sqlx::query("INSERT INTO entry_state (did, entry_id, read, starred, updated_at) VALUES ('did:plc:x',?1,?2,?3,'2026-01-01T00:00:00Z')")
10688 .bind(entry_id).bind(read).bind(starred)
10689 .execute(pool).await.unwrap();
10690 }
10691
10692 /// **A STARRED article is never evicted, however old.**
10693 ///
10694 /// The starred view joins `entries`, and `entry_state` cascades on delete,
10695 /// so pruning a starred entry removed it from the starred list entirely —
10696 /// and the content is not recoverable, because a feed serves only its last
10697 /// few dozen items. The PDS keeps the saved RECORD; it has never held the
10698 /// article.
10699 #[tokio::test]
10700 async fn retention_keeps_starred_and_unread_entries() -> anyhow::Result<()> {
10701 let pool = init_url("sqlite::memory:").await?;
10702 let old_read = aged_entry(&pool, "old-read", 30).await;
10703 let old_starred = aged_entry(&pool, "old-starred", 30).await;
10704 let old_unread = aged_entry(&pool, "old-unread", 30).await;
10705 let recent_read = aged_entry(&pool, "recent-read", 1).await;
10706 mark(&pool, old_read, 1, 0).await;
10707 mark(&pool, old_starred, 1, 1).await; // read AND starred
10708 mark(&pool, old_unread, 0, 0).await;
10709 mark(&pool, recent_read, 1, 0).await;
10710
10711 let deleted = prune_old_entries(&pool, 14, 3650, 0).await?;
10712 assert_eq!(deleted, 1, "only the old, read, unstarred entry should go");
10713
10714 let left: Vec<String> = sqlx::query_scalar("SELECT guid FROM entries ORDER BY guid")
10715 .fetch_all(&pool)
10716 .await?;
10717 assert_eq!(left, vec!["old-starred", "old-unread", "recent-read"]);
10718 Ok(())
10719 }
10720
10721 /// An entry nobody has interacted with at all — no `entry_state` row — is
10722 /// still evicted once it ages out. Otherwise the cache never shrinks, since
10723 /// most entries are never opened.
10724 #[tokio::test]
10725 async fn retention_evicts_entries_with_no_reader_state() -> anyhow::Result<()> {
10726 let pool = init_url("sqlite::memory:").await?;
10727 aged_entry(&pool, "untouched-old", 30).await;
10728 aged_entry(&pool, "untouched-new", 1).await;
10729 assert_eq!(prune_old_entries(&pool, 14, 3650, 0).await?, 1);
10730 Ok(())
10731 }
10732
10733 /// **A recently-polled feed is NOT made due again.**
10734 ///
10735 /// `due_feeds` treats NULL as due immediately, so an unbounded nudge from a
10736 /// page handler turned every reload of the starred view into another poll of
10737 /// those feeds — outbound amplification against third-party origins, and one
10738 /// reader monopolising a poll budget that is shared and already the binding
10739 /// constraint on user count.
10740 #[tokio::test]
10741 async fn a_recently_polled_feed_is_not_nudged_again() -> anyhow::Result<()> {
10742 let pool = init_url("sqlite::memory:").await?;
10743 let recent = "2026-01-01T11:59:00Z";
10744 let stale_before = "2026-01-01T11:00:00Z"; // one hour before "now"
10745
10746 sqlx::query("INSERT INTO feeds (url, last_polled, next_poll) VALUES (?1, ?2, ?3)")
10747 .bind("https://fresh.example/f")
10748 .bind(recent)
10749 .bind("2026-01-01T12:59:00Z")
10750 .execute(&pool)
10751 .await?;
10752 // Polled long ago: this one SHOULD be nudged.
10753 sqlx::query("INSERT INTO feeds (url, last_polled, next_poll) VALUES (?1, ?2, ?3)")
10754 .bind("https://stale.example/f")
10755 .bind("2026-01-01T06:00:00Z")
10756 .bind("2026-01-01T07:00:00Z")
10757 .execute(&pool)
10758 .await?;
10759
10760 mark_feed_due(&pool, "https://fresh.example/f", stale_before).await?;
10761 mark_feed_due(&pool, "https://stale.example/f", stale_before).await?;
10762
10763 let fresh: Option<String> =
10764 sqlx::query_scalar("SELECT next_poll FROM feeds WHERE url = 'https://fresh.example/f'")
10765 .fetch_one(&pool)
10766 .await?;
10767 let stale: Option<String> =
10768 sqlx::query_scalar("SELECT next_poll FROM feeds WHERE url = 'https://stale.example/f'")
10769 .fetch_one(&pool)
10770 .await?;
10771
10772 assert!(
10773 fresh.is_some(),
10774 "a feed polled a minute ago was made due again — a reload loop is an \
10775 amplification vector"
10776 );
10777 assert!(stale.is_none(), "a long-unpolled feed should be nudged");
10778 Ok(())
10779 }
10780
10781 /// A feed that has never been polled is always nudgeable — there is no
10782 /// recent fetch to argue it would be wasted.
10783 #[tokio::test]
10784 async fn a_never_polled_feed_is_nudged() -> anyhow::Result<()> {
10785 let pool = init_url("sqlite::memory:").await?;
10786 sqlx::query("INSERT INTO feeds (url, last_polled, next_poll) VALUES (?1, NULL, ?2)")
10787 .bind("https://new.example/f")
10788 .bind("2026-01-01T12:59:00Z")
10789 .execute(&pool)
10790 .await?;
10791 mark_feed_due(&pool, "https://new.example/f", "2026-01-01T11:00:00Z").await?;
10792 let next: Option<String> =
10793 sqlx::query_scalar("SELECT next_poll FROM feeds WHERE url = 'https://new.example/f'")
10794 .fetch_one(&pool)
10795 .await?;
10796 assert!(next.is_none());
10797 Ok(())
10798 }
10799
10800 /// **The hard ceiling is the bound that sparing would otherwise remove.**
10801 ///
10802 /// "Mark unread" is a one-click control and `entries` is shared across every
10803 /// reader, so an unbounded `read = 0` exception lets one person pin rows
10804 /// permanently — and since the poller stops entirely above
10805 /// `db_size_watermark_bytes` with this DELETE as its only release valve,
10806 /// those pins could stop polling for everyone.
10807 #[tokio::test]
10808 async fn the_hard_ceiling_evicts_even_starred_and_unread() -> anyhow::Result<()> {
10809 let pool = init_url("sqlite::memory:").await?;
10810 let ancient_starred = aged_entry(&pool, "ancient-starred", 400).await;
10811 let ancient_unread = aged_entry(&pool, "ancient-unread", 400).await;
10812 let recent_starred = aged_entry(&pool, "recent-starred", 30).await;
10813 mark(&pool, ancient_starred, 1, 1).await;
10814 mark(&pool, ancient_unread, 0, 0).await;
10815 mark(&pool, recent_starred, 1, 1).await;
10816
10817 // 14-day soft window, 180-day hard ceiling.
10818 prune_old_entries(&pool, 14, 180, 0).await?;
10819
10820 let left: Vec<String> = sqlx::query_scalar("SELECT guid FROM entries ORDER BY guid")
10821 .fetch_all(&pool)
10822 .await?;
10823 assert_eq!(
10824 left,
10825 vec!["recent-starred"],
10826 "past the ceiling nothing is pinned — otherwise one reader can stall the poller \
10827 for every reader"
10828 );
10829 Ok(())
10830 }
10831
10832 /// The per-feed trim spares starred entries too. It was fixed in the
10833 /// retention sweep and NOT here, which left the documented guarantee false —
10834 /// and this path runs on every poll of every feed rather than daily.
10835 #[tokio::test]
10836 async fn the_per_feed_trim_spares_starred_entries() -> anyhow::Result<()> {
10837 let pool = init_url("sqlite::memory:").await?;
10838 let old_starred = aged_entry(&pool, "old-starred", 5).await;
10839 mark(&pool, old_starred, 1, 1).await;
10840 for i in 0..5 {
10841 aged_entry(&pool, &format!("filler-{i}"), 1).await;
10842 }
10843 let feed_id: i64 = sqlx::query_scalar("SELECT id FROM feeds LIMIT 1")
10844 .fetch_one(&pool)
10845 .await?;
10846
10847 // Trim hard enough that the older starred entry would be cut. The trim
10848 // runs inside `insert_entries`, so drive it the way production does.
10849 insert_entries(&pool, feed_id, &[], 2).await?;
10850
10851 let left: Vec<String> =
10852 sqlx::query_scalar("SELECT guid FROM entries WHERE guid = 'old-starred'")
10853 .fetch_all(&pool)
10854 .await?;
10855 assert_eq!(
10856 left,
10857 vec!["old-starred"],
10858 "the per-feed trim evicted a starred entry"
10859 );
10860 Ok(())
10861 }
10862
10863 /// **When more entries are starred than the cap, the NEWEST starred ones
10864 /// are spared.** The sparing subquery orders by date and takes `cap`; the
10865 /// existing tests seed one starred row (fewer than the cap, so the order
10866 /// never chooses) or assert only a count. With `DESC` flipped to `ASC` the
10867 /// suite stayed green — and in production the trim would spare the OLDEST
10868 /// starred articles and evict the newest, on every poll of every feed.
10869 #[tokio::test]
10870 async fn the_trim_spares_the_newest_starred_entries_when_over_cap() -> anyhow::Result<()> {
10871 let pool = init_url("sqlite::memory:").await?;
10872 // Five starred entries, one per day, cap of two: only the two newest
10873 // may survive.
10874 let mut ids = Vec::new();
10875 for days_old in 1..=5 {
10876 let id = aged_entry(&pool, &format!("starred-{days_old}"), days_old).await;
10877 mark(&pool, id, 1, 1).await;
10878 ids.push((days_old, id));
10879 }
10880 let feed_id: i64 = sqlx::query_scalar("SELECT feed_id FROM entries WHERE id = ?1")
10881 .bind(ids[0].1)
10882 .fetch_one(&pool)
10883 .await?;
10884 insert_entries(&pool, feed_id, &[], 2).await?;
10885
10886 let mut survivors: Vec<String> = sqlx::query_scalar("SELECT guid FROM entries")
10887 .fetch_all(&pool)
10888 .await?;
10889 survivors.sort();
10890 assert_eq!(
10891 survivors,
10892 vec!["starred-1".to_string(), "starred-2".to_string()],
10893 "the trim spared the wrong starred entries"
10894 );
10895 Ok(())
10896 }
10897}