Skip to main content

feather_reader/
standard_site.rs

1//! Reading `standard.site` publications as feeds.
2//!
3//! A publication is not a feed document — it is a record in somebody's atproto
4//! repo, and its "entries" are separate records in the same repo. So this reads
5//! two collections rather than fetching one URL:
6//!
7//! ```text
8//! at://<did>/site.standard.publication/<rkey>
9//!   ├─ resolve <did> → PDS
10//!   ├─ getRecord   site.standard.publication  → name, url
11//!   └─ listRecords site.standard.document     → paged, filtered on `site`
12//! ```
13//!
14//! **Unauthenticated throughout.** This reads *someone else's* repo with no
15//! session, which is why it cannot reuse [`crate::oauth::xrpc::Repo`]: that type
16//! takes its base URL from the session's PDS, hardcodes `repo` to
17//! `session.sub`, and DPoP-signs every send. None of that survives contact with
18//! "read a stranger's repo".
19//!
20//! **Only `textContent` and `description` are read; `content` is ignored.**
21//! `content` is an open union — measured across 449 real documents it carried
22//! six different wrappers and twenty-two block types from five vendor
23//! namespaces, growing with every platform that adopts the lexicon, and it
24//! would drag an HTML-sanitisation surface over foreign input. A document with
25//! neither field still yields an entry: title, date and a link is what an RSS
26//! reader shows for a title-only feed, and is not a failure state.
27
28use serde::Deserialize;
29
30use crate::lexicon::nsid;
31
32/// Documents per `listRecords` page.
33///
34/// Smaller than the protocol default of 100 because a `site.standard.document`
35/// carries the whole article — ~17 KB measured, and the `content` union this
36/// module ignores is still in the wire bytes — while
37/// [`crate::net::read_capped`] bounds a response at 8 MB. 100 long-form
38/// articles per page can exceed that and fail the walk outright.
39const DOCUMENT_PAGE_SIZE: u32 = 25;
40
41/// A parsed `at://` URI: `at://<authority>/<collection>/<rkey>`.
42///
43/// Parsed by hand rather than with `url::Url`, which **cannot read the form that
44/// matters**: `at://did:plc:…/…` fails with *invalid port number*, because the
45/// colons in the DID are taken as a port separator. The handle form parses
46/// fine, which is what makes the failure easy to miss.
47#[derive(Debug, Clone, PartialEq, Eq)]
48pub struct AtUri {
49    pub authority: String,
50    pub collection: String,
51    pub rkey: String,
52}
53
54impl AtUri {
55    /// Parse, or `None` if this is not a well-formed three-segment at-URI.
56    pub fn parse(uri: &str) -> Option<Self> {
57        let rest = uri.strip_prefix(crate::atproto::AT_URI_PREFIX)?;
58        let mut parts = rest.split('/');
59        let (authority, collection, rkey) = (parts.next()?, parts.next()?, parts.next()?);
60        if parts.next().is_some()
61            || authority.is_empty()
62            || collection.is_empty()
63            || rkey.is_empty()
64        {
65            return None;
66        }
67        Some(Self {
68            authority: authority.to_string(),
69            collection: collection.to_string(),
70            rkey: rkey.to_string(),
71        })
72    }
73}
74
75impl std::fmt::Display for AtUri {
76    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
77        write!(
78            f,
79            "at://{}/{}/{}",
80            self.authority, self.collection, self.rkey
81        )
82    }
83}
84
85/// What one read of a publication produced.
86///
87/// **`complete` is the fact the store cannot recover afterwards.** A truncated
88/// walk and a short publication return the same entries; the difference decides
89/// whether an empty result is a quiet blog or a read that gave up, and the
90/// caller has no other way to tell. `fetch` used to drop it on the floor after
91/// logging a warning.
92#[derive(Debug)]
93pub struct PublicationRead {
94    pub publication: Publication,
95    pub entries: Vec<Entry>,
96    pub complete: bool,
97}
98
99/// The publication record — a pointer, not a feed. Supplies the title and the
100/// base URL that document `path`s are joined onto.
101#[derive(Debug, Clone, PartialEq, Eq)]
102pub struct Publication {
103    pub name: Option<String>,
104    pub url: String,
105}
106
107/// One document, mapped onto the shape the feed pipeline already stores.
108#[derive(Debug, Clone, PartialEq, Eq)]
109pub struct Entry {
110    /// The document's own `at://` URI.
111    ///
112    /// **Not derived from `path`.** `path` is mutable — a publisher who moves an
113    /// article would duplicate their whole archive on the next poll, because
114    /// dedup is `UNIQUE (feed_id, guid)`.
115    pub guid: String,
116    pub title: String,
117    /// When the document was published, re-spelled by `crate::feed::fmt_time`
118    /// into the store's one RFC3339 shape. The reading order sorts on this
119    /// column as a string, so a publisher's spelling cannot go in verbatim.
120    ///
121    /// Taken from `publishedAt` when it parses AND is not in the future,
122    /// otherwise from the TID in the record key, otherwise `None`. See
123    /// `entries_from_records` for why a future date is discarded rather than
124    /// clamped, and why an undated entry is left undated.
125    pub published: Option<String>,
126    /// The joined, **scheme-vetted** permalink — `None` when the document's
127    /// `path` does not resolve to a safe href on the publication's origin.
128    /// `Option` because the guarantee cannot be met unconditionally and the
129    /// store's column is optional too; a title-only entry is not a failure.
130    pub url: Option<String>,
131    /// `description`, else `textContent`, escaped by
132    /// `crate::feed::plain_text_to_html` — both are plain text in the
133    /// lexicon, and the column they land in is rendered as HTML.
134    pub summary: Option<String>,
135}
136
137impl From<Entry> for crate::store::NewEntry {
138    /// The shape the poller stores. Kept here, next to the fields it maps,
139    /// so wiring the reader to the scheduler has nothing left to decide.
140    fn from(e: Entry) -> Self {
141        crate::store::NewEntry {
142            guid: e.guid,
143            url: e.url,
144            title: Some(e.title),
145            author: None,
146            published: e.published,
147            content_html: e.summary,
148            fetched_at: None,
149        }
150    }
151}
152
153#[derive(Debug, Deserialize)]
154struct PublicationValue {
155    name: Option<String>,
156    url: String,
157}
158
159#[derive(Debug, Deserialize)]
160struct DocumentValue {
161    title: String,
162    /// Optional so a document without one is an entry with no date, the same
163    /// answer a garbage one gets — the strictness ran the other way, making
164    /// the field this module is willing to DISCARD the one whose absence was
165    /// fatal to the whole record.
166    #[serde(rename = "publishedAt")]
167    published_at: Option<String>,
168    /// Optional for the same reason as `publishedAt`: a document with no
169    /// `path` keeps its title, date and summary rather than vanishing from
170    /// the feed entirely. `Entry.url` is already `Option`.
171    path: Option<String>,
172    /// The at-URI of the publication this document belongs to.
173    ///
174    /// **Load-bearing.** A repo can hold several publications — measured, some
175    /// do — so documents must be filtered by this rather than assumed to belong
176    /// to the one being polled.
177    site: String,
178    #[serde(rename = "textContent")]
179    text_content: Option<String>,
180    description: Option<String>,
181}
182
183/// Find the publication named by `rkey` among a repo's publication records.
184///
185/// Returns its **canonical** at-URI — the one the PDS itself minted — alongside
186/// the record. That canonical URI is what documents reference in their `site`
187/// field, so it is the key the filter uses, rather than the string the reader
188/// subscribed with. Storage is DID-form only (#164), so today the two agree;
189/// taking the PDS's spelling keeps them agreeing if they ever stop.
190pub fn publication_from_records(
191    rkey: &str,
192    records: &[crate::atproto::RecordEntry],
193) -> Option<(String, Publication)> {
194    let entry = records
195        .iter()
196        .find(|r| AtUri::parse(&r.uri).is_some_and(|u| u.rkey == rkey))?;
197    let value: PublicationValue = serde_json::from_value(entry.value.clone()).ok()?;
198    // **The base of every Entry.url, so it is vetted as the href it becomes.**
199    // `net::safe_link` is the same check the RSS entry pipeline applies at
200    // `feed.rs`, and the reason `safe_link.rs` exists as a type at all: the
201    // procedural version of this guarantee was deleted once with a green suite.
202    let url = crate::net::safe_link(&value.url)?;
203    Some((
204        entry.uri.clone(),
205        Publication {
206            name: value.name,
207            url,
208        },
209    ))
210}
211
212/// Map a repo's document records onto entries, keeping only those belonging to
213/// `canonical_site`.
214pub fn entries_from_records(
215    canonical_site: &str,
216    publication: &Publication,
217    records: &[crate::atproto::RecordEntry],
218) -> Vec<Entry> {
219    // **Normalised to a directory.** `Url::join` is RFC-3986: against a base of
220    // `https://example.com/blog`, a relative `posts/a` resolves to
221    // `/posts/a`, silently dropping the subpath every permalink needs. A
222    // trailing slash makes the base a directory, which is what a publication
223    // URL means.
224    let base = url::Url::parse(&publication.url).ok().map(|mut u| {
225        if !u.path().ends_with('/') {
226            u.set_path(&format!("{}/", u.path()));
227        }
228        u
229    });
230    // One "now" for the whole batch, so two documents of the same poll are
231    // judged against the same instant rather than one being called credible and
232    // an identical one not. (`tid_timestamp` reads the clock again for its own
233    // ceiling; that only ever tightens a bound five minutes away, so it needs
234    // no such agreement.)
235    let now = chrono::Utc::now();
236    // The SAME allowance the record-key branch gets. A publisher's clock runs
237    // ahead of ours as readily as a PDS's does, and judging a stated date
238    // against a bare `now` while the rkey below gets five minutes discarded a
239    // perfectly good date and left the newest post undated — which, ordered on
240    // a bare `published DESC`, puts it at the bottom of the list.
241    let ceiling = now + chrono::Duration::seconds(crate::atproto::CLOCK_SKEW_GRACE_SECS);
242    records
243        .iter()
244        .filter_map(|record| {
245            // A document that does not deserialise is SKIPPED, not fatal: one
246            // malformed record must not cost a publisher its whole feed.
247            let doc: DocumentValue = serde_json::from_value(record.value.clone()).ok()?;
248            if doc.site != canonical_site {
249                return None;
250            }
251            Some(Entry {
252                guid: record.uri.clone(),
253                title: doc.title,
254                // **Three rules, in order: what the publisher credibly said,
255                // then when the record was written, then nothing.**
256                //
257                // A stated date in the future is DISCARDED, not clamped to now.
258                // The store refreshes `published` on every poll but stamps
259                // `fetched_at` only once, so a date derived from the current
260                // clock is rewritten hourly: the row stays the newest thing in
261                // the feed forever, is never swept, is never evicted by the
262                // per-feed cap, and sits at the top of the reading list re-dated
263                // to today. Clamping moved the defect rather than fixing it.
264                // Every fall-through here holds still instead — the TID is the
265                // record's real write time, and an undated row is dated by
266                // `fetched_at`, which never moves. An rkey that is not a TID
267                // leaves the entry undated rather than inventing a date from a
268                // slug.
269                published: doc
270                    .published_at
271                    .as_deref()
272                    .and_then(|raw| chrono::DateTime::parse_from_rfc3339(raw).ok())
273                    .map(|d| d.with_timezone(&chrono::Utc))
274                    .filter(|d| *d <= ceiling)
275                    .or_else(|| {
276                        AtUri::parse(&record.uri)
277                            .and_then(|uri| crate::atproto::tid_timestamp(&uri.rkey))
278                    })
279                    .map(crate::feed::fmt_time),
280                // `non_blank` for the same reason the summary uses it: a blank
281                // path joins to the publication's own base, so a handful of
282                // documents with an empty `path` became a handful of entries
283                // all linking to the site root.
284                url: non_blank(doc.path)
285                    .as_deref()
286                    .and_then(|path| join_path(base.as_ref(), path)),
287                // `description` first — the authored summary — but only when it
288                // actually says something: a blank one must not shadow the body.
289                // Then ESCAPED, not sanitised: both fields are plain text.
290                summary: non_blank(doc.description)
291                    .or_else(|| non_blank(doc.text_content))
292                    .map(|raw| crate::feed::plain_text_to_html(&raw)),
293            })
294        })
295        .collect()
296}
297
298/// The ingest floor: the oldest `published` an entry may carry and still be worth
299/// storing, or `None` for "store anything".
300///
301/// **The floor is the window the SWEEP would use, which is not the smaller of the
302/// two.** `store::prune_old_entries` honours the hard ceiling only when it is
303/// strictly older than the rolling window, or when there is no window at all
304/// (`hard_days > 0 && (days <= 0 || hard_days > days)`); a ceiling inside the
305/// window is logged and ignored there, because the hard delete spares nothing and
306/// would otherwise delete exactly the rows the soft delete exists to spare.
307///
308/// So the earliest thing that can delete a row is `retention_days` when there is
309/// a window, and the ceiling only when there is not.
310///
311/// **The pair comes from [`crate::config::Config::retention_for`], and for a
312/// publication it is not the RSS window.** That function is the one home for which
313/// window applies to which kind, precisely so this and the sweep cannot drift: a
314/// publication gets `(0, publication_retention_days)` — no rolling window, and a
315/// generous archive ceiling — because a 14-day window stored ZERO rows from every
316/// real publication measured, their newest documents being 109 to 241 days old.
317/// Fed that pair, the `days == 0` branch below falls through to the ceiling, which
318/// is exactly the sweep pass that can delete such a row.
319///
320/// `a_publications_floor_is_its_archive_ceiling_not_the_rss_window` asserts the
321/// two halves agree; it is the test that fails if either side is changed alone.
322///
323/// Two wrong versions, both worth naming. Keying on `retention_days` alone left
324/// `retention_days = 0` unfloored while the ceiling still deleted at
325/// `retention_hard_days` — the exact cycle the floor exists to prevent, and a
326/// worse one, since the ceiling spares nothing and a starred entry came back
327/// unstarred rather than merely unread. Reaching for the MINIMUM then
328/// over-corrected: at `days = 180, hard = 30` the sweep ignores the ceiling and
329/// deletes nothing before 180 days, while `min` floored ingest at 30 and silently
330/// discarded five months of a publisher's archive that nothing would have deleted.
331///
332/// **An unrepresentable window is no floor, said directly.** `RETENTION_DAYS`
333/// parses into a `u32` with no upper bound, and `u32::MAX` days is an operator
334/// saying "keep everything"; both the duration and the subtraction can fail, and
335/// either failure means `None` here. The previous shape fell back to a sentinel
336/// instant (`MIN_UTC`) and left the outcome resting on the fact that the row
337/// comparison is LEXICOGRAPHIC: `fmt_time` renders an out-of-range year with a
338/// `+`/`-` sign, which sorts either side of a 4-digit year by ASCII accident
339/// rather than by date. Verified: swapping that fallback to `MAX_UTC` — a floor
340/// of the year 262143, which should discard every entry in existence — changed
341/// nothing, because `"2026…" > "+262143…"`. With `None` the ambiguity is gone,
342/// and the comparison never sees a date outside the range it can order.
343///
344/// `now` is a parameter so this is testable without the clock.
345fn ingest_floor(
346    retention_days: u32,
347    retention_hard_days: u32,
348    now: chrono::DateTime<chrono::Utc>,
349) -> Option<String> {
350    let days = if retention_days > 0 {
351        retention_days
352    } else if retention_hard_days > 0 {
353        retention_hard_days
354    } else {
355        return None;
356    };
357    let window = chrono::Duration::try_days(days.into())?;
358    now.checked_sub_signed(window).map(crate::feed::fmt_time)
359}
360
361/// Persist one publication read, and say what the poll amounted to.
362///
363/// `retention_days` is the ingest floor: an entry whose stated date is already
364/// older than the window is not stored, because storing it means the next sweep
365/// deletes it and the next poll re-inserts it with a new row id — which loses its
366/// read state and arrives unread, on that cycle, forever.
367///
368/// **For a publication the caller passes `Config::retention_for`'s pair, which is
369/// the archive ceiling rather than the 14-day window** — so in practice almost
370/// nothing is floored out here, which is the point: measured, a 14-day floor
371/// dropped every document of every real publication tried. See `ingest_floor`.
372///
373/// **An entry with no date is stored anyway.** That is a decision, not an
374/// oversight: it is dated by `fetched_at` instead, which does not move, and the
375/// alternative is discarding an article the reader can never see.
376///
377/// Three consequences, stated because an earlier version of this comment named
378/// only the first and called it "accepted":
379///
380/// * It resurrects once per retention window. `fetched_at` is stamped at the
381///   first insert and never refreshed, so the row ages out, is swept once the
382///   reader has read it, and is re-inserted unread by the next poll.
383/// * Under the hard ceiling it comes back **unstarred as well**. The soft sweep
384///   spares `starred = 1 OR read = 0`; the ceiling spares nothing.
385/// * Until then it OUTRANKS the publication's real articles. Both the sweep and
386///   the per-feed cap order on `COALESCE(published, fetched_at)`, so an undated
387///   entry sorts by the moment it was fetched — that is, as the newest thing in
388///   the feed — and a publication whose documents are mostly undated can push
389///   dated articles out of `max_entries_per_feed`.
390///
391/// **`new_entries` is an upper bound, not a count of rows that survived.**
392/// `insert_entries` reports what it inserted, and the per-feed cap then trims
393/// within the same call, so a poll that inserted 30 rows into a feed capped at 10
394/// can report 30.
395pub async fn store_publication(
396    pool: &sqlx::SqlitePool,
397    url: &str,
398    read: PublicationRead,
399    max_entries_per_feed: i64,
400    retention_days: u32,
401    retention_hard_days: u32,
402) -> anyhow::Result<crate::feed::PollOutcome> {
403    let offered = read.entries.len();
404
405    // **The failure check runs before anything is written.** Stamping the feed
406    // and then returning `Failed` is the shape that makes a broken publication
407    // read as freshly polled on `/stats`, and it is easy to write by accident
408    // because the upsert is the natural first step.
409    //
410    // **Keyed on what was OFFERED, not on what survives the floor.** A truncated
411    // read that produced nothing means the walk gave up before its first record.
412    // A truncated read whose entries are merely older than the window is a
413    // healthy poll of an old publication, and calling that a failure puts it into
414    // backoff that widens forever.
415    if !read.complete && offered == 0 {
416        return Ok(crate::feed::PollOutcome::Failed {
417            backoff: crate::feed::backoff_for(1),
418            kind: crate::feed::FailureKind::Body,
419            detail: crate::feed::failure_detail(
420                "the publication read stopped before its first document",
421            ),
422        });
423    }
424
425    // See [`ingest_floor`] for which of the two windows this is, and why.
426    let floor = ingest_floor(retention_days, retention_hard_days, chrono::Utc::now());
427    let rows: Vec<crate::store::NewEntry> = read
428        .entries
429        .into_iter()
430        .filter(|e| match (&floor, &e.published) {
431            // Lexicographic on the store's one RFC3339 spelling, which sorts
432            // chronologically by construction.
433            (Some(floor), Some(published)) => published.as_str() >= floor.as_str(),
434            // No floor, or no date. The undated case is the decision the doc
435            // above records: kept, dated by `fetched_at`, and it resurrects once
436            // per window.
437            _ => true,
438        })
439        .map(Into::into)
440        .collect();
441
442    // **Say when the floor emptied the read.** A complete read of three
443    // year-old posts and a complete read of an empty publication both return
444    // `Updated { new_entries: 0 }`, stamp the feed, and look green — so a
445    // subscriber to an archived blog gets a blank feed and nothing anywhere says
446    // why. `offered` is already in hand.
447    if offered > rows.len() {
448        tracing::info!(
449            feed = %url,
450            offered,
451            stored = rows.len(),
452            "the retention floor dropped documents older than the window"
453        );
454    }
455
456    let feed_id = crate::store::upsert_feed(
457        pool,
458        &crate::store::NewFeed {
459            url: url.to_string(),
460            title: read.publication.name.clone(),
461            site_url: Some(read.publication.url.clone()),
462            last_polled: Some(crate::feed::fmt_time(chrono::Utc::now())),
463            ..Default::default()
464        },
465    )
466    .await?;
467
468    let new_entries =
469        crate::store::insert_entries(pool, feed_id, &rows, max_entries_per_feed).await?;
470    Ok(crate::feed::PollOutcome::Updated { new_entries })
471}
472
473fn non_blank(s: Option<String>) -> Option<String> {
474    s.filter(|v| !v.trim().is_empty())
475}
476
477/// Join a document `path` onto the publication's base URL.
478///
479/// **`Url::join`, not concatenation.** Concatenating produced
480/// `https://x.com/https://evil.example/a` for an absolute path and buried the
481/// path inside the query for a base carrying one. `join` also keeps the result
482/// on the publication's own origin for a relative path, which is the only shape
483/// the lexicon describes.
484fn join_path(base: Option<&url::Url>, path: &str) -> Option<String> {
485    // **`safe_link` on the way out, not only on the base.** The scheme
486    // guarantee used to live solely in `publication_from_records`; this
487    // function and `Publication` are both `pub`, so a caller that built a
488    // `Publication` some other way (the step-3 poller, from a stored row) gave
489    // an unparseable base — and the no-base branch then returned the
490    // document's `path` verbatim, putting `javascript:` into an entry link.
491    // **No base, no URL.** This branch used to return `safe_link(path)`, which
492    // vets the scheme but NOT the origin — so a caller holding a `Publication`
493    // it did not build through `publication_from_records` (the step-3 poller,
494    // from a stored row whose `site_url` is NULL or malformed) would publish a
495    // publisher-controlled `https://evil.example/x` as a permalink under that
496    // publication's name. The two branches agree now: off-origin is `None`, and
497    // "no origin to be off" is also `None`.
498    let base = base?;
499    match base.join(path) {
500        // A path that resolves off the publication's origin is not a path, it
501        // is a redirect the publisher smuggled into a field we render as theirs.
502        Ok(joined) if joined.origin() == base.origin() => crate::net::safe_link(joined.as_str()),
503        // **No URL, not the homepage.** Falling back to the base gave every
504        // affected entry the same href pointing at the site root — which is
505        // what a publication on an apex domain whose documents live on `www.`
506        // or a CDN would produce for its whole archive, with nothing to say
507        // anything had been dropped.
508        _ => None,
509    }
510}
511
512/// What the document walk should do with one record.
513///
514/// A pure decision so it can be tested without a PDS: the walk itself is a
515/// closure over the network.
516#[derive(Debug, Clone, Copy, PartialEq, Eq)]
517enum DocumentFate {
518    /// Belongs to the publication being read.
519    Keep,
520    /// Belongs to another publication **in this repo** — normal, and the
521    /// reason the `site` filter exists. Not a signal of anything.
522    Sibling,
523    /// References a publication this repo does not have: a `site` spelling
524    /// nothing can ever match. Indistinguishable from a quiet blog without
525    /// saying so, which is why it is counted.
526    Orphan,
527    /// Not a document this reader understands.
528    Malformed,
529}
530
531fn classify_document(
532    record: &crate::atproto::RecordEntry,
533    canonical_site: &str,
534    known: &std::collections::HashSet<&str>,
535) -> DocumentFate {
536    match serde_json::from_value::<DocumentValue>(record.value.clone()) {
537        Ok(doc) if doc.site == canonical_site => DocumentFate::Keep,
538        Ok(doc) if known.contains(doc.site.as_str()) => DocumentFate::Sibling,
539        Ok(_) => DocumentFate::Orphan,
540        Err(_) => DocumentFate::Malformed,
541    }
542}
543
544/// Read a publication and its documents through the hardened anonymous client.
545///
546/// **Deliberately thin.** Everything that makes this fetch safe already exists
547/// in [`crate::atproto`] and is reused rather than rebuilt:
548///
549/// - [`crate::atproto::resolve_did_to_pds`] runs
550///   [`crate::net::assert_public_target`] on the `serviceEndpoint`, which is a
551///   stranger's string;
552/// - every read goes through [`crate::net::guarded_get_no_privacy`], re-vetting
553///   per hop and pinning the connection, which closes the rebinding window and
554///   supplies the `User-Agent` that 4 of 19 measured endpoints demand;
555/// - [`crate::net::read_capped`] bounds each response;
556/// - [`crate::atproto::PdsClient::list_all_records`] bounds the page count AND
557///   detects a repeated or absent cursor — the trap this module's first draft
558///   walked into, already solved there;
559/// - an XRPC error envelope surfaces as [`crate::atproto::AtProtoError::Xrpc`] rather
560///   than deserialising into an empty page.
561///
562/// The first draft of this module reimplemented all of that, worse. The only
563/// logic left here is the part that is genuinely about standard.site.
564pub async fn fetch(
565    http: &reqwest::Client,
566    plc_directory: &str,
567    uri: &AtUri,
568) -> anyhow::Result<PublicationRead> {
569    use anyhow::Context;
570
571    // The collection is part of the identity of what was subscribed to, and
572    // this function is `pub`: without the check it lists publications and
573    // matches on rkey alone, so `at://did/app.bsky.feed.post/<rkey>` would be
574    // "read as a publication" whenever a publication shares that rkey.
575    anyhow::ensure!(
576        uri.collection == nsid::STANDARD_PUBLICATION,
577        "{uri} is not a {} URI",
578        nsid::STANDARD_PUBLICATION
579    );
580
581    let pds = crate::atproto::resolve_did_to_pds(http, plc_directory, &uri.authority)
582        .await
583        .with_context(|| format!("resolving the PDS for {}", uri.authority))?;
584    let client = crate::atproto::PdsClient::anonymous(http.clone(), pds, uri.authority.clone());
585
586    // **One budget for both walks, because they nest.** The publications are
587    // still held when the documents walk runs — `known` borrows them, and the
588    // `keep` closure captures that — so two independent ceilings let this one
589    // read hold twice the bound the box was sized for. Threading a single budget
590    // makes the limit a property of the read rather than of each walk, and the
591    // borrow checker keeps it that way where a comment would not.
592    let mut budget = crate::atproto::ByteBudget::new(crate::atproto::MAX_LIST_BYTES);
593    let publications = client
594        .list_all_records_within(nsid::STANDARD_PUBLICATION, &mut budget)
595        .await
596        .with_context(|| format!("listing publications for {}", uri.authority))?;
597    let (canonical_site, publication) = publication_from_records(&uri.rkey, &publications)
598        .with_context(|| format!("{uri} is not a readable site.standard.publication"))?;
599
600    // **Filtered inside the walk, so the cap counts THIS publication's
601    // documents.** A repo-wide cap applied before the filter starves a quiet
602    // publication whose busy sibling fills the window — it returns nothing,
603    // permanently, and worse with every post the sibling makes.
604    //
605    // The page is smaller than the protocol default because a document
606    // carries the whole article (~17 KB measured, and the `content` union this
607    // module ignores is still on the wire): 100 long-form articles per page
608    // can exceed `read_capped`'s 8 MB and fail the walk outright.
609    let known: std::collections::HashSet<&str> =
610        publications.iter().map(|p| p.uri.as_str()).collect();
611    let mut orphaned = 0usize;
612    let documents = client
613        .list_recent_matching_within(
614            nsid::STANDARD_DOCUMENT,
615            crate::atproto::MAX_LARGE_RECORDS,
616            &mut budget,
617            DOCUMENT_PAGE_SIZE,
618            // Orphans are counted while WALKING, not over the kept window: a
619            // truncated slice would both miss orphans and stay silent in
620            // exactly the case where the feed went empty structurally.
621            |record| match classify_document(record, &canonical_site, &known) {
622                DocumentFate::Keep => true,
623                DocumentFate::Orphan => {
624                    orphaned += 1;
625                    false
626                }
627                DocumentFate::Sibling | DocumentFate::Malformed => false,
628            },
629        )
630        .await
631        .with_context(|| format!("listing documents for {canonical_site}"))?;
632    Ok(read_from(publication, &canonical_site, documents, orphaned))
633}
634
635/// Assemble a [`PublicationRead`] from a finished walk.
636///
637/// **Extracted so `complete` is testable.** It is the one fact the store cannot
638/// recover afterwards, and reaching the `false` case through `fetch` needs a PLC
639/// directory, a PDS, and a walk that truncates — three mocked hosts for one
640/// boolean. Reaching it here needs a struct.
641fn read_from(
642    publication: Publication,
643    canonical_site: &str,
644    documents: crate::atproto::RecordWalk,
645    orphaned: usize,
646) -> PublicationRead {
647    let entries = entries_from_records(canonical_site, &publication, &documents.records);
648
649    // **A walk that stopped early is not a short archive.** Reading part of a
650    // publication is acceptable; reporting it as the whole of one is not, and
651    // when the part is empty — a quiet publication whose busy sibling fills
652    // every page this reader will fetch — the feed looks healthy and stays
653    // empty forever.
654    // Logged AND returned. Logging it alone is how the fact that the walk gave
655    // up reached nobody who could act on it: a truncated read and a short
656    // publication produce the same entries, and only the caller can tell the
657    // difference between "quiet blog" and "we stopped early".
658    if !documents.complete {
659        tracing::warn!(
660            site = %canonical_site,
661            kept = entries.len(),
662            "stopped reading this publication before its documents ran out"
663        );
664    }
665    // **A spelling mismatch on `site` looks exactly like an empty
666    // publication.** A publication with no documents is normal, so the poller
667    // would call this healthy forever; if the repo HAD documents and none
668    // matched, say so, because that is the shape of a bug rather than of a
669    // quiet blog.
670    if orphaned > 0 {
671        tracing::warn!(
672            site = %canonical_site,
673            orphaned,
674            "documents in this repo reference no publication in it — a `site` spelling nothing matches"
675        );
676    }
677    PublicationRead {
678        publication,
679        entries,
680        complete: documents.complete,
681    }
682}
683
684#[cfg(test)]
685mod tests {
686    use super::*;
687    use crate::atproto::RecordEntry;
688    use serde_json::json;
689
690    const DID: &str = "did:plc:ohutz6x5acjmpuulp3x7wxxc";
691
692    /// **A walk that gave up must not report itself as a whole publication.**
693    ///
694    /// `fetch` used to log that and return only the entries, so the fact reached
695    /// nobody who could act on it — and a truncated read and a quiet blog produce
696    /// exactly the same entries.
697    #[test]
698    fn a_read_carries_whether_the_walk_finished() {
699        let publication = Publication {
700            name: Some("Scan's Lab".to_string()),
701            url: "https://example.com/blog/".to_string(),
702        };
703        for complete in [true, false] {
704            let walk = crate::atproto::RecordWalk {
705                records: Vec::new(),
706                complete,
707            };
708            let read = read_from(publication.clone(), "at://d/c/r", walk, 0);
709            assert_eq!(
710                read.complete, complete,
711                "the walk said complete={complete} and the read said {}",
712                read.complete
713            );
714        }
715    }
716
717    // -- storing a publication read -----------------------------------------
718
719    fn read_of(entries: Vec<Entry>, complete: bool) -> PublicationRead {
720        PublicationRead {
721            publication: Publication {
722                name: Some("Scan's Lab".to_string()),
723                url: "https://example.com/blog/".to_string(),
724            },
725            entries,
726            complete,
727        }
728    }
729
730    fn entry_dated(guid: &str, published: Option<&str>) -> Entry {
731        Entry {
732            guid: guid.to_string(),
733            title: "T".to_string(),
734            published: published.map(str::to_string),
735            url: Some("https://example.com/blog/a".to_string()),
736            summary: None,
737        }
738    }
739
740    fn days_ago(n: i64) -> String {
741        crate::feed::fmt_time(chrono::Utc::now() - chrono::Duration::days(n))
742    }
743
744    const PUB_URL: &str = "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab";
745
746    async fn pool() -> sqlx::SqlitePool {
747        crate::store::init_url("sqlite::memory:").await.unwrap()
748    }
749
750    #[tokio::test]
751    async fn a_complete_read_stores_its_entries_and_reports_them() {
752        let pool = pool().await;
753        let read = read_of(
754            vec![
755                entry_dated("at://d/c/1", Some(&days_ago(1))),
756                entry_dated("at://d/c/2", Some(&days_ago(2))),
757            ],
758            true,
759        );
760        let outcome = store_publication(&pool, PUB_URL, read, 0, 14, 180)
761            .await
762            .unwrap();
763        assert!(
764            matches!(
765                outcome,
766                crate::feed::PollOutcome::Updated { new_entries: 2 }
767            ),
768            "expected two new entries, got {outcome:?}"
769        );
770        let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
771            .fetch_one(&pool)
772            .await
773            .unwrap();
774        assert_eq!(n, 2, "the entries were not stored");
775    }
776
777    /// **Starvation is keyed on what was OFFERED, not on what was stored.**
778    ///
779    /// A truncated read that produced nothing is a failure: the walk gave up
780    /// before the first record. But a truncated read whose entries the retention
781    /// floor dropped is not — the read worked, the entries are simply older than
782    /// the window, and calling that a failure puts a healthy publication into
783    /// backoff forever.
784    #[tokio::test]
785    async fn an_incomplete_read_that_offered_nothing_is_a_failure() {
786        let pool = pool().await;
787        let outcome = store_publication(&pool, PUB_URL, read_of(vec![], false), 0, 14, 180)
788            .await
789            .unwrap();
790        assert!(
791            matches!(outcome, crate::feed::PollOutcome::Failed { .. }),
792            "a truncated read that produced nothing is not a healthy poll: {outcome:?}"
793        );
794    }
795
796    #[tokio::test]
797    async fn an_incomplete_read_whose_entries_the_floor_dropped_is_not_a_failure() {
798        let pool = pool().await;
799        let read = read_of(
800            vec![entry_dated("at://d/c/old", Some(&days_ago(900)))],
801            false,
802        );
803        let outcome = store_publication(&pool, PUB_URL, read, 0, 14, 180)
804            .await
805            .unwrap();
806        assert!(
807            !matches!(outcome, crate::feed::PollOutcome::Failed { .. }),
808            "the read offered an entry; the floor dropping it is not a failed poll: {outcome:?}"
809        );
810    }
811
812    /// A failed poll must not look like a successful one on `/stats`.
813    #[tokio::test]
814    async fn a_failed_read_does_not_stamp_last_polled() {
815        let pool = pool().await;
816        let outcome = store_publication(&pool, PUB_URL, read_of(vec![], false), 0, 14, 180)
817            .await
818            .unwrap();
819        assert!(matches!(outcome, crate::feed::PollOutcome::Failed { .. }));
820        let stamped: Option<String> =
821            sqlx::query_scalar("SELECT last_polled FROM feeds WHERE url = ?1")
822                .bind(PUB_URL)
823                .fetch_optional(&pool)
824                .await
825                .unwrap()
826                .flatten();
827        assert_eq!(
828            stamped, None,
829            "a failed poll stamped last_polled, so the feed reads as freshly polled"
830        );
831    }
832
833    #[tokio::test]
834    async fn an_entry_already_older_than_the_window_is_not_stored() {
835        let pool = pool().await;
836        let read = read_of(
837            vec![
838                entry_dated("at://d/c/fresh", Some(&days_ago(1))),
839                entry_dated("at://d/c/ancient", Some(&days_ago(900))),
840            ],
841            true,
842        );
843        store_publication(&pool, PUB_URL, read, 0, 14, 180)
844            .await
845            .unwrap();
846        let guids: Vec<String> = sqlx::query_scalar("SELECT guid FROM entries ORDER BY guid")
847            .fetch_all(&pool)
848            .await
849            .unwrap();
850        assert_eq!(
851            guids,
852            vec!["at://d/c/fresh".to_string()],
853            "an entry the next sweep would delete was stored anyway"
854        );
855    }
856
857    /// **The floor follows whichever window actually deletes, not the rolling one.**
858    ///
859    /// `retention_days = 0` is a supported configuration meaning "no rolling
860    /// window", and the hard ceiling stays alive independently. Keying the ingest
861    /// floor on the rolling window alone let a 900-day-old entry in, which the
862    /// ceiling then deleted and the next poll re-inserted — and the ceiling spares
863    /// nothing, so a starred entry came back unstarred.
864    #[tokio::test]
865    async fn the_floor_follows_the_hard_ceiling_when_the_window_is_disabled() {
866        let pool = pool().await;
867        let read = read_of(
868            vec![entry_dated("at://d/c/ancient", Some(&days_ago(900)))],
869            true,
870        );
871        store_publication(&pool, PUB_URL, read, 0, 0, 180)
872            .await
873            .unwrap();
874        let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
875            .fetch_one(&pool)
876            .await
877            .unwrap();
878        assert_eq!(
879            n, 0,
880            "an entry the hard ceiling will delete was stored, so it will resurrect"
881        );
882    }
883
884    /// **The WINDOW is the floor when there is one, because it deletes first.**
885    ///
886    /// With a 14-day rolling window and a 180-day ceiling, an entry 100 days old
887    /// is inside the ceiling and outside the window — so the sweep takes it once
888    /// it has been read and the next poll puts it back. Taking the longer of the
889    /// two would store it.
890    #[tokio::test]
891    async fn the_floor_follows_the_window_when_both_are_set() {
892        let pool = pool().await;
893        let read = read_of(
894            vec![entry_dated("at://d/c/hundred", Some(&days_ago(100)))],
895            true,
896        );
897        store_publication(&pool, PUB_URL, read, 0, 14, 180)
898            .await
899            .unwrap();
900        let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
901            .fetch_one(&pool)
902            .await
903            .unwrap();
904        assert_eq!(
905            n, 0,
906            "an entry inside the ceiling but outside the window was stored, so it will cycle"
907        );
908    }
909
910    /// With both windows off there is no floor, because nothing will delete it.
911    /// **The two halves of the retention policy, asserted against each other.**
912    ///
913    /// `Config::retention_for` decides which window a kind gets; `ingest_floor`
914    /// decides what is worth storing; `store::prune_old_entries` decides what is
915    /// deleted. The first two are asserted here against the third's rule, because
916    /// a drift between them is the resurrection cycle — a row the store keeps, the
917    /// sweep deletes, and the next poll re-inserts unread — and nothing else in the
918    /// suite would notice one side changing alone.
919    ///
920    /// The publication case is the one that matters: fed the RSS window a
921    /// publication stores nothing at all, because a real one's newest document is
922    /// months old.
923    #[test]
924    fn a_publications_floor_is_its_archive_ceiling_not_the_rss_window() {
925        let config = crate::config::Config::default();
926        let now = chrono::DateTime::parse_from_rfc3339("2026-09-29T12:00:00Z")
927            .unwrap()
928            .with_timezone(&chrono::Utc);
929
930        let (days, hard) = config.retention_for(crate::feed::FeedKind::Publication);
931        let floor = ingest_floor(days, hard, now).expect("a publication has an ingest floor");
932        let expected = crate::feed::fmt_time(
933            now - chrono::Duration::days(config.publication_retention_days.into()),
934        );
935        assert_eq!(
936            floor, expected,
937            "a publication's floor must be its archive ceiling ({} days), because \
938             that is the only sweep pass that can delete its rows",
939            config.publication_retention_days,
940        );
941
942        // And the number this rules OUT: the RSS window would drop every document
943        // of every real publication measured (newest 109 to 241 days old).
944        let rss_floor = ingest_floor(config.retention_days, config.retention_hard_days, now)
945            .expect("an RSS feed has an ingest floor");
946        assert!(
947            floor < rss_floor,
948            "the publication floor ({floor}) is no older than the RSS one \
949             ({rss_floor}), so a months-old document would still be dropped",
950        );
951        let a_real_publications_newest_document =
952            crate::feed::fmt_time(now - chrono::Duration::days(109));
953        assert!(
954            a_real_publications_newest_document.as_str() >= floor.as_str(),
955            "the newest document a real publication offered would be refused at \
956             ingest: {a_real_publications_newest_document} against a floor of {floor}",
957        );
958        assert!(
959            a_real_publications_newest_document.as_str() < rss_floor.as_str(),
960            "this assertion is only meaningful while the RSS window WOULD have \
961             dropped it, and it no longer does",
962        );
963    }
964
965    /// **The floor, all four configurations, without a database or the clock.**
966    ///
967    /// The end-to-end tests below observe the floor through what gets stored,
968    /// which cannot see the difference between "no floor" and "a floor the
969    /// lexicographic row comparison happens to sort past". This asserts the value
970    /// itself.
971    #[test]
972    fn the_ingest_floor_is_the_window_the_sweep_would_use() {
973        let now = chrono::DateTime::parse_from_rfc3339("2026-09-26T12:00:00Z")
974            .unwrap()
975            .with_timezone(&chrono::Utc);
976
977        assert_eq!(
978            ingest_floor(14, 180, now).as_deref(),
979            Some("2026-09-12T12:00:00Z"),
980            "with both set, the floor is the WINDOW — the thing that deletes first",
981        );
982        assert_eq!(
983            ingest_floor(180, 30, now).as_deref(),
984            Some("2026-03-30T12:00:00Z"),
985            "a ceiling INSIDE the window is one `prune_old_entries` ignores, so it \
986             must not lower the floor — `min` here discarded five months of archive \
987             that nothing would have deleted",
988        );
989        assert_eq!(
990            ingest_floor(0, 30, now).as_deref(),
991            Some("2026-08-27T12:00:00Z"),
992            "with no window the ceiling stands alone, and it still deletes",
993        );
994        assert_eq!(
995            ingest_floor(0, 0, now),
996            None,
997            "with no retention at all there is nothing to floor against",
998        );
999        // `u32::MAX` days is an operator saying "keep everything". Both the
1000        // duration and the subtraction fail there, and the answer is the ABSENCE
1001        // of a floor — not a sentinel instant whose rendering the row comparison
1002        // then has to sort correctly by accident.
1003        assert_eq!(
1004            ingest_floor(u32::MAX, u32::MAX, now),
1005            None,
1006            "an unrepresentable window produced a floor, so the comparison is \
1007             resting on how `fmt_time` renders an out-of-range year",
1008        );
1009    }
1010
1011    /// **A ceiling INSIDE the window is one the sweep ignores, so the floor must
1012    /// ignore it too.**
1013    ///
1014    /// `prune_old_entries` honours `retention_hard_days` only when it is strictly
1015    /// older than `retention_days` (or when there is no window at all): a ceiling
1016    /// inside the window is logged and dropped, because the hard delete spares
1017    /// nothing and would otherwise delete exactly the rows the soft delete exists
1018    /// to spare. So at `days = 180, hard = 30` nothing is deleted before 180 days.
1019    ///
1020    /// An earlier version took the MINIMUM of the two, floored ingest at 30 days,
1021    /// and silently discarded five months of a publisher's archive that nothing
1022    /// would ever have deleted. The older test above passes under both rules —
1023    /// `min(14, 180)` and "the window" are both 14 — which is why this case is
1024    /// the one that had to be written.
1025    #[tokio::test]
1026    async fn a_ceiling_inside_the_window_does_not_lower_the_floor() {
1027        let pool = pool().await;
1028        let read = read_of(
1029            vec![entry_dated("at://d/c/hundred", Some(&days_ago(100)))],
1030            true,
1031        );
1032        store_publication(&pool, PUB_URL, read, 0, 180, 30)
1033            .await
1034            .unwrap();
1035        let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
1036            .fetch_one(&pool)
1037            .await
1038            .unwrap();
1039        assert_eq!(
1040            n, 1,
1041            "an entry inside the 180-day window was dropped because of a 30-day \
1042             ceiling the sweep ignores — five months of archive discarded at ingest \
1043             that nothing would have deleted",
1044        );
1045    }
1046
1047    /// **A complete read of an empty publication is a healthy poll, not a
1048    /// failure.** The failure branch is keyed on `!complete && offered == 0`, and
1049    /// dropping either half of that makes a brand-new or fully-archived
1050    /// publication go into backoff that widens forever.
1051    #[tokio::test]
1052    async fn a_complete_read_of_nothing_is_not_a_failure() {
1053        let pool = pool().await;
1054        let outcome = store_publication(&pool, PUB_URL, read_of(vec![], true), 0, 14, 180)
1055            .await
1056            .unwrap();
1057        assert!(
1058            matches!(
1059                outcome,
1060                crate::feed::PollOutcome::Updated { new_entries: 0 }
1061            ),
1062            "a complete read of an empty publication was not a healthy poll: {outcome:?}"
1063        );
1064    }
1065
1066    /// **The other half of `a_failed_read_does_not_stamp_last_polled`.** That test
1067    /// alone is satisfied by never stamping at all, which would leave every
1068    /// publication permanently due and `/stats` permanently wrong.
1069    #[tokio::test]
1070    async fn a_successful_read_stamps_last_polled() {
1071        let pool = pool().await;
1072        let read = read_of(vec![entry_dated("at://d/c/1", Some(&days_ago(1)))], true);
1073        store_publication(&pool, PUB_URL, read, 0, 14, 180)
1074            .await
1075            .unwrap();
1076        let stamped: Option<String> =
1077            sqlx::query_scalar("SELECT last_polled FROM feeds WHERE url = ?1")
1078                .bind(PUB_URL)
1079                .fetch_one(&pool)
1080                .await
1081                .unwrap();
1082        assert!(
1083            stamped.is_some(),
1084            "a successful read left `last_polled` NULL, so the publication stays \
1085             due forever and `/stats` never shows it as polled",
1086        );
1087    }
1088
1089    /// **The saturating window must saturate in the KEEPING direction.**
1090    ///
1091    /// `an_absurd_retention_window_does_not_panic` only asserts that it returns.
1092    /// A saturation that produced `DateTime::MAX` instead — or a floor of "now" —
1093    /// would satisfy it while silently discarding the publisher's entire archive:
1094    /// `u32::MAX` days is an operator saying "keep everything".
1095    #[tokio::test]
1096    async fn an_absurd_retention_window_keeps_everything_rather_than_nothing() {
1097        let pool = pool().await;
1098        let read = read_of(
1099            vec![entry_dated("at://d/c/ancient", Some(&days_ago(10_000)))],
1100            true,
1101        );
1102        store_publication(&pool, PUB_URL, read, 0, u32::MAX, u32::MAX)
1103            .await
1104            .expect("a huge window is a wide floor, not a crash");
1105        let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
1106            .fetch_one(&pool)
1107            .await
1108            .unwrap();
1109        assert_eq!(
1110            n, 1,
1111            "a window of u32::MAX days dropped a 27-year-old entry, so the \
1112             saturation went the wrong way",
1113        );
1114    }
1115
1116    /// **`max_entries_per_feed` is passed through, and every test above passes
1117    /// `0`.** So a call that dropped the argument, or passed a constant, would
1118    /// have gone unnoticed — and this is the cap that decides which of a
1119    /// publisher's articles a reader keeps.
1120    #[tokio::test]
1121    async fn the_per_feed_cap_is_the_one_the_caller_passed() {
1122        let pool = pool().await;
1123        let read = read_of(
1124            (0..5)
1125                .map(|i| entry_dated(&format!("at://d/c/{i}"), Some(&days_ago(i + 1))))
1126                .collect(),
1127            true,
1128        );
1129        store_publication(&pool, PUB_URL, read, 2, 0, 0)
1130            .await
1131            .unwrap();
1132        let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
1133            .fetch_one(&pool)
1134            .await
1135            .unwrap();
1136        assert_eq!(
1137            n, 2,
1138            "five entries under a cap of two left {n} rows, so the caller's cap is \
1139             not the one being applied",
1140        );
1141    }
1142
1143    #[tokio::test]
1144    async fn no_retention_at_all_means_no_ingest_floor() {
1145        let pool = pool().await;
1146        let read = read_of(
1147            vec![entry_dated("at://d/c/ancient", Some(&days_ago(900)))],
1148            true,
1149        );
1150        store_publication(&pool, PUB_URL, read, 0, 0, 0)
1151            .await
1152            .unwrap();
1153        let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
1154            .fetch_one(&pool)
1155            .await
1156            .unwrap();
1157        assert_eq!(
1158            n, 1,
1159            "with nothing deleting it, an old entry is worth keeping"
1160        );
1161    }
1162
1163    /// A retention window near `u32::MAX` must not panic the poller.
1164    #[tokio::test]
1165    async fn an_absurd_retention_window_does_not_panic() {
1166        let pool = pool().await;
1167        let read = read_of(vec![entry_dated("at://d/c/x", Some(&days_ago(1)))], true);
1168        store_publication(&pool, PUB_URL, read, 0, u32::MAX, u32::MAX)
1169            .await
1170            .expect("a huge window is a wide floor, not a crash");
1171    }
1172
1173    /// The feed row learns the publication's name and homepage — the only reason
1174    /// beyond the timestamp that the upsert is there at all. `upsert_feed`
1175    /// COALESCEs both, so dropping either is silent.
1176    #[tokio::test]
1177    async fn the_feed_row_learns_the_publications_name_and_site() {
1178        let pool = pool().await;
1179        let read = read_of(vec![entry_dated("at://d/c/1", Some(&days_ago(1)))], true);
1180        store_publication(&pool, PUB_URL, read, 0, 14, 180)
1181            .await
1182            .unwrap();
1183        let (title, site): (Option<String>, Option<String>) =
1184            sqlx::query_as("SELECT title, site_url FROM feeds WHERE url = ?1")
1185                .bind(PUB_URL)
1186                .fetch_one(&pool)
1187                .await
1188                .unwrap();
1189        assert_eq!(
1190            title.as_deref(),
1191            Some("Scan's Lab"),
1192            "the name never reached the row"
1193        );
1194        assert_eq!(
1195            site.as_deref(),
1196            Some("https://example.com/blog/"),
1197            "the homepage never reached the row"
1198        );
1199    }
1200
1201    /// **The undated entry is kept, deliberately.** It is dated by `fetched_at`,
1202    /// which holds still, and the alternative is discarding an article the reader
1203    /// can never see. It resurrects once per retention window; that is accepted.
1204    #[tokio::test]
1205    async fn an_undated_entry_is_stored_rather_than_dropped() {
1206        let pool = pool().await;
1207        let read = read_of(vec![entry_dated("at://d/c/undated", None)], true);
1208        store_publication(&pool, PUB_URL, read, 0, 14, 180)
1209            .await
1210            .unwrap();
1211        let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
1212            .fetch_one(&pool)
1213            .await
1214            .unwrap();
1215        assert_eq!(n, 1, "an entry with no date was discarded");
1216    }
1217
1218    /// A record key from a real atproto repo, and the instant it decodes to.
1219    ///
1220    /// **Fixed, not minted.** A test that mints a TID expects "about now", and
1221    /// "about now" is satisfied by any source of the current time — including a
1222    /// fallback that has had the rkey ripped out of it and returns `Utc::now()`
1223    /// instead. Both tests named for the rkey fallback used to survive exactly
1224    /// that mutation. A historical key with a stated answer cannot.
1225    const PAST_TID: &str = "3jzfcijpj2z2a";
1226    const PAST_TID_WRITTEN_AT: &str = "2023-06-30T15:03:01Z";
1227
1228    fn rec(collection: &str, rkey: &str, value: serde_json::Value) -> RecordEntry {
1229        RecordEntry {
1230            uri: format!("at://{DID}/{collection}/{rkey}"),
1231            cid: None,
1232            value,
1233        }
1234    }
1235
1236    fn publication(rkey: &str, url: &str) -> RecordEntry {
1237        rec(
1238            nsid::STANDARD_PUBLICATION,
1239            rkey,
1240            json!({ "name": "Scan's Lab", "url": url }),
1241        )
1242    }
1243
1244    fn document(rkey: &str, site: &str, title: &str, path: &str) -> RecordEntry {
1245        rec(
1246            nsid::STANDARD_DOCUMENT,
1247            rkey,
1248            json!({
1249                "title": title,
1250                "publishedAt": "2026-07-11T00:00:00Z",
1251                "path": path,
1252                "site": site,
1253                "textContent": "body",
1254            }),
1255        )
1256    }
1257
1258    fn canonical(rkey: &str) -> String {
1259        format!("at://{DID}/{}/{rkey}", nsid::STANDARD_PUBLICATION)
1260    }
1261
1262    /// **The `site` filter keys on the URI the PDS minted, not the string the
1263    /// reader subscribed with.** Documents reference their publication by the
1264    /// canonical URI, and every measured document does. An earlier draft
1265    /// compared against the subscribed string; storage was then meant to admit
1266    /// handle-form URIs, and a handle-form subscription found nothing forever
1267    /// while the module declared the feed healthy. #164 made storage DID-only,
1268    /// so the two strings agree today — the canonical one is still the right
1269    /// key, and this pins it.
1270    #[test]
1271    fn the_site_filter_uses_the_uri_the_pds_minted() {
1272        let records = vec![publication("p", "https://scanash.com")];
1273        let (site, pubn) = publication_from_records("p", &records).expect("publication not found");
1274        assert_eq!(site, canonical("p"), "did not take the PDS's canonical URI");
1275
1276        let docs = vec![document("d1", &canonical("p"), "Hello", "/hello")];
1277        let entries = entries_from_records(&site, &pubn, &docs);
1278        assert_eq!(
1279            entries.len(),
1280            1,
1281            "a canonical-site document was not matched"
1282        );
1283    }
1284
1285    /// The mapping into the store's row is total and loses nothing the poller
1286    /// would need — so wiring the reader has nothing to invent.
1287    #[test]
1288    fn an_entry_maps_onto_the_stores_row() {
1289        let records = vec![publication("p", "https://example.com")];
1290        let (site, pubn) = publication_from_records("p", &records).unwrap();
1291        let docs = vec![document("rk1", &site, "Hello", "/hello")];
1292        let row: crate::store::NewEntry = entries_from_records(&site, &pubn, &docs)
1293            .pop()
1294            .unwrap()
1295            .into();
1296        assert_eq!(
1297            row.guid,
1298            format!("at://{DID}/{}/rk1", nsid::STANDARD_DOCUMENT)
1299        );
1300        assert_eq!(row.url.as_deref(), Some("https://example.com/hello"));
1301        assert_eq!(row.title.as_deref(), Some("Hello"));
1302        assert_eq!(row.published.as_deref(), Some("2026-07-11T00:00:00Z"));
1303        assert_eq!(row.content_html.as_deref(), Some("body"));
1304        assert_eq!(row.author, None);
1305        assert_eq!(row.fetched_at, None);
1306    }
1307
1308    /// A repo can hold several publications — measured, some do — and
1309    /// `listRecords` cannot filter server-side.
1310    #[test]
1311    fn documents_are_filtered_by_their_site_field() {
1312        let records = vec![publication("mine", "https://example.com")];
1313        let (site, pubn) = publication_from_records("mine", &records).unwrap();
1314        let docs = vec![
1315            document("a", &site, "Mine", "/a"),
1316            document("b", &canonical("theirs"), "Theirs", "/b"),
1317            document("c", &site, "Mine again", "/c"),
1318        ];
1319        let titles: Vec<String> = entries_from_records(&site, &pubn, &docs)
1320            .into_iter()
1321            .map(|e| e.title)
1322            .collect();
1323        assert_eq!(titles, ["Mine", "Mine again"]);
1324    }
1325
1326    /// **The publication URL is a stranger's string and is vetted as one.**
1327    ///
1328    /// It becomes the base of every `Entry.url`, which is an href. This repo has
1329    /// `safe_link.rs` as a type with a module-private field precisely because
1330    /// the procedural version of this guarantee failed; #143 and #138 are this
1331    /// same bug class on `siteUrl`.
1332    #[test]
1333    fn a_publication_with_a_hostile_url_is_refused() {
1334        for hostile in [
1335            "javascript:alert(1)",
1336            "data:text/html,<script>",
1337            "file:///etc/passwd",
1338            "",
1339        ] {
1340            let records = vec![publication("p", hostile)];
1341            assert!(
1342                publication_from_records("p", &records).is_none(),
1343                "accepted a publication whose url is {hostile:?}",
1344            );
1345        }
1346    }
1347
1348    /// **Entry URLs are joined, not concatenated.**
1349    ///
1350    /// String concatenation produced `https://x.com/https://evil.com/a` for an
1351    /// absolute `path`, and put the path inside the query string for a base
1352    /// carrying one.
1353    #[test]
1354    fn entry_urls_are_joined_against_the_publication_base() {
1355        let records = vec![publication("p", "https://example.com/blog")];
1356        let (site, pubn) = publication_from_records("p", &records).unwrap();
1357        let docs = vec![
1358            document("a", &site, "Relative", "/a"),
1359            document("b", &site, "Absolute-looking", "https://evil.example/x"),
1360        ];
1361        let urls: Vec<Option<String>> = entries_from_records(&site, &pubn, &docs)
1362            .into_iter()
1363            .map(|e| e.url)
1364            .collect();
1365        assert_eq!(urls[0].as_deref(), Some("https://example.com/a"));
1366        // Exactly the publication's base, not merely "not evil": a mutation
1367        // that returned the raw path, or an empty string, passed the weaker
1368        // negative assertion this used to be.
1369        // Off-origin is dropped, not rewritten to the base. This assertion has
1370        // moved twice: it began as "not evil" (a mutant returning the raw path
1371        // passed it), was tightened to the base fallback, and is now `None` —
1372        // the base gave every affected entry the same homepage href.
1373        assert_eq!(
1374            urls[1], None,
1375            "a document path that escapes its publication's origin must yield no URL"
1376        );
1377    }
1378
1379    /// 8% of measured documents (37 of 449) carry neither summary field.
1380    #[test]
1381    fn a_document_with_neither_summary_field_still_yields_an_entry() {
1382        let records = vec![publication("p", "https://example.com/")];
1383        let (site, pubn) = publication_from_records("p", &records).unwrap();
1384        let bare = rec(
1385            nsid::STANDARD_DOCUMENT,
1386            "bare",
1387            json!({
1388                "title": "Bare",
1389                "publishedAt": "2026-07-11T00:00:00Z",
1390                "path": "/bare",
1391                "site": site,
1392            }),
1393        );
1394        let entries = entries_from_records(&site, &pubn, &[bare]);
1395        assert_eq!(entries.len(), 1);
1396        assert_eq!(entries[0].summary, None);
1397        assert_eq!(entries[0].url.as_deref(), Some("https://example.com/bare"));
1398    }
1399
1400    /// `description` is the authored summary; `textContent` is the whole body.
1401    /// An EMPTY description must not shadow a real one.
1402    #[test]
1403    fn an_empty_description_does_not_shadow_the_body() {
1404        let records = vec![publication("p", "https://example.com")];
1405        let (site, pubn) = publication_from_records("p", &records).unwrap();
1406        let doc = rec(
1407            nsid::STANDARD_DOCUMENT,
1408            "d",
1409            json!({
1410                "title": "T",
1411                "publishedAt": "2026-07-11T00:00:00Z",
1412                "path": "/d",
1413                "site": site,
1414                "description": "   ",
1415                "textContent": "the real body",
1416            }),
1417        );
1418        let entries = entries_from_records(&site, &pubn, &[doc]);
1419        assert_eq!(entries[0].summary.as_deref(), Some("the real body"));
1420    }
1421
1422    /// **`publishedAt` is parsed, not passed through.** The store's `published`
1423    /// column is the RFC3339 shape `feed::fmt_time` writes, and the reading
1424    /// order sorts on it as a string. A publisher's string went in verbatim —
1425    /// a garbage value would have sorted arbitrarily among real ones, and a
1426    /// valid-but-differently-spelled one (`+00:00`, fractional seconds) would
1427    /// not have matched the RSS path's spelling for the same instant.
1428    #[test]
1429    fn published_at_is_normalised_or_dropped() {
1430        let records = vec![publication("p", "https://example.com")];
1431        let (site, pubn) = publication_from_records("p", &records).unwrap();
1432        let with = |rkey: &str, published_at: serde_json::Value| {
1433            rec(
1434                nsid::STANDARD_DOCUMENT,
1435                rkey,
1436                json!({ "title": "T", "publishedAt": published_at, "path": "/x", "site": site }),
1437            )
1438        };
1439        // Single-character rkeys on purpose: they are not TIDs, so the
1440        // rkey-derived fallback below does not apply and "unparseable" really
1441        // does mean undated here.
1442        let docs = vec![
1443            with("a", json!("2026-07-11T09:30:00.123+02:00")),
1444            with("b", json!("yesterday-ish")),
1445            with("c", json!("2026-07-11T00:00:00Z")),
1446        ];
1447        let published: Vec<Option<String>> = entries_from_records(&site, &pubn, &docs)
1448            .into_iter()
1449            .map(|e| e.published)
1450            .collect();
1451        assert_eq!(
1452            published,
1453            vec![
1454                Some("2026-07-11T07:30:00Z".to_string()),
1455                None,
1456                Some("2026-07-11T00:00:00Z".to_string()),
1457            ],
1458            "publishedAt was not normalised to the store's spelling"
1459        );
1460    }
1461
1462    /// A document with no usable `publishedAt` is dated from its rkey.
1463    ///
1464    /// **An undated entry is not merely untidy, it is immortal-and-mortal at
1465    /// once.** The store's retention sweep and its per-feed cap both order on
1466    /// `COALESCE(published, fetched_at)`, and an entry inserted with no
1467    /// `published` gets `fetched_at` stamped at insertion. So the sweep deletes
1468    /// it once it is `retention_days` old, the next poll re-inserts it with a
1469    /// fresh `fetched_at` and a new `entries.id`, its read state is gone with
1470    /// the cascade, and it arrives unread — again, on the same cycle, forever.
1471    /// The same reset also sorts it newest in the per-feed cap, where it evicts
1472    /// entries that really are newer.
1473    #[test]
1474    fn an_undated_document_is_dated_from_its_tid_rkey() {
1475        let records = vec![publication("p", "https://example.com")];
1476        let (site, pubn) = publication_from_records("p", &records).unwrap();
1477        let docs = vec![rec(
1478            nsid::STANDARD_DOCUMENT,
1479            PAST_TID,
1480            json!({ "title": "T", "path": "/x", "site": site }),
1481        )];
1482        assert_eq!(
1483            entries_from_records(&site, &pubn, &docs)
1484                .into_iter()
1485                .next()
1486                .expect("the document is an entry")
1487                .published,
1488            Some(PAST_TID_WRITTEN_AT.to_string()),
1489            "the date must come from the record key, and be spelled the way the store spells dates"
1490        );
1491    }
1492
1493    #[test]
1494    fn an_unparseable_published_at_falls_back_to_the_tid_rkey() {
1495        let records = vec![publication("p", "https://example.com")];
1496        let (site, pubn) = publication_from_records("p", &records).unwrap();
1497        let docs = vec![rec(
1498            nsid::STANDARD_DOCUMENT,
1499            PAST_TID,
1500            json!({ "title": "T", "publishedAt": "yesterday-ish", "path": "/x", "site": site }),
1501        )];
1502        assert_eq!(
1503            entries_from_records(&site, &pubn, &docs)
1504                .into_iter()
1505                .next()
1506                .expect("the document is an entry")
1507                .published,
1508            Some(PAST_TID_WRITTEN_AT.to_string()),
1509            "a date the parser cannot read is no date at all, so the rkey must stand in"
1510        );
1511    }
1512
1513    #[test]
1514    fn a_stated_date_outranks_the_rkey() {
1515        let records = vec![publication("p", "https://example.com")];
1516        let (site, pubn) = publication_from_records("p", &records).unwrap();
1517        // Both candidate dates are historical, so nothing here depends on
1518        // what the machine's clock reads.
1519        let docs = vec![rec(
1520            nsid::STANDARD_DOCUMENT,
1521            PAST_TID,
1522            json!({ "title": "T", "publishedAt": "2020-01-02T00:00:00Z", "path": "/x", "site": site }),
1523        )];
1524        assert_eq!(
1525            entries_from_records(&site, &pubn, &docs)
1526                .into_iter()
1527                .next()
1528                .expect("the document is an entry")
1529                .published,
1530            Some("2020-01-02T00:00:00Z".to_string()),
1531            "the rkey records when the file was written, which is not when the post was published"
1532        );
1533    }
1534
1535    /// **A stated date in the future is discarded, not clamped.**
1536    ///
1537    /// Clamping it to "now" looks safe and is not. The store refreshes
1538    /// `published` on every poll, so the row would be re-dated to the current
1539    /// hour forever: never older than the retention cutoff, never outranked in
1540    /// the per-feed cap, permanently first in the reading list. The date must
1541    /// not depend on when the mapping ran, which is why both of these name the
1542    /// exact value they expect rather than comparing against the clock.
1543    #[test]
1544    fn a_future_dated_document_falls_back_to_its_rkey() {
1545        let records = vec![publication("p", "https://example.com")];
1546        let (site, pubn) = publication_from_records("p", &records).unwrap();
1547        let docs = vec![rec(
1548            nsid::STANDARD_DOCUMENT,
1549            PAST_TID,
1550            json!({ "title": "T", "publishedAt": "2999-01-01T00:00:00Z", "path": "/x", "site": site }),
1551        )];
1552        assert_eq!(
1553            entries_from_records(&site, &pubn, &docs)
1554                .into_iter()
1555                .next()
1556                .expect("the document is an entry")
1557                .published,
1558            Some(PAST_TID_WRITTEN_AT.to_string()),
1559            "the date must be the record's write time, not the hour the poll happened to run"
1560        );
1561    }
1562
1563    #[test]
1564    fn a_future_dated_document_without_a_tid_rkey_is_undated() {
1565        let records = vec![publication("p", "https://example.com")];
1566        let (site, pubn) = publication_from_records("p", &records).unwrap();
1567        let docs = vec![rec(
1568            nsid::STANDARD_DOCUMENT,
1569            "self",
1570            json!({ "title": "T", "publishedAt": "2999-01-01T00:00:00Z", "path": "/x", "site": site }),
1571        )];
1572        assert_eq!(
1573            entries_from_records(&site, &pubn, &docs)
1574                .into_iter()
1575                .next()
1576                .expect("the document is an entry")
1577                .published,
1578            None,
1579            "with nothing credible to date it by, the row falls to fetched_at, which holds still"
1580        );
1581    }
1582
1583    /// **A little ahead of our clock is skew, not a lie.**
1584    ///
1585    /// The rkey here is deliberately not a TID, so nothing masks a wrongly
1586    /// discarded date: if the stated one is thrown away the entry is undated,
1587    /// and an undated entry sorts to the bottom of a list ordered on a bare
1588    /// `published DESC`. A publisher a few seconds fast would have had their
1589    /// newest post buried.
1590    #[test]
1591    fn a_stated_date_a_little_ahead_of_our_clock_is_still_believed() {
1592        let records = vec![publication("p", "https://example.com")];
1593        let (site, pubn) = publication_from_records("p", &records).unwrap();
1594        let slightly_ahead =
1595            crate::feed::fmt_time(chrono::Utc::now() + chrono::Duration::seconds(10));
1596        let docs = vec![rec(
1597            nsid::STANDARD_DOCUMENT,
1598            "self",
1599            json!({ "title": "T", "publishedAt": slightly_ahead, "path": "/x", "site": site }),
1600        )];
1601        assert_eq!(
1602            entries_from_records(&site, &pubn, &docs)
1603                .into_iter()
1604                .next()
1605                .expect("the document is an entry")
1606                .published,
1607            Some(slightly_ahead),
1608            "a few seconds of clock skew must not cost the entry its date"
1609        );
1610    }
1611
1612    #[test]
1613    fn a_document_with_neither_a_date_nor_a_tid_rkey_stays_undated() {
1614        let records = vec![publication("p", "https://example.com")];
1615        let (site, pubn) = publication_from_records("p", &records).unwrap();
1616        // Two shapes, rejected by two different checks. "my-first-post" is 13
1617        // characters but carries a `-`, so it never reaches the alphabet's
1618        // arithmetic at all. "abcdefghijklm" is 13 valid s32 characters and
1619        // decodes perfectly well — to the year 2192 — which is the case the
1620        // bound in `tid_timestamp` exists for, and the one a publisher naming
1621        // files by slug actually produces.
1622        let docs = vec![
1623            rec(
1624                nsid::STANDARD_DOCUMENT,
1625                "my-first-post",
1626                json!({ "title": "T", "path": "/x", "site": site }),
1627            ),
1628            rec(
1629                nsid::STANDARD_DOCUMENT,
1630                "abcdefghijklm",
1631                json!({ "title": "T", "path": "/y", "site": site }),
1632            ),
1633        ];
1634        assert_eq!(
1635            entries_from_records(&site, &pubn, &docs)
1636                .into_iter()
1637                .map(|e| e.published)
1638                .collect::<Vec<_>>(),
1639            vec![None, None],
1640            "an invented date is worse than no date; the store decides what to do with undated rows"
1641        );
1642    }
1643
1644    /// **Summaries are plain text and are escaped, not sanitised.**
1645    ///
1646    /// The lexicon defines `textContent` and `description` as plain text, and
1647    /// the store's `content_html` is rendered as HTML, so the text must be
1648    /// escaped on the way in. The first version of this ran `ammonia::clean`
1649    /// over them — the RSS body function — which parses its input as markup
1650    /// and deletes everything after a bare `<`. 321 of 449 measured documents
1651    /// use `textContent` as their summary; any post mentioning `Vec<T>` lost
1652    /// the rest of its summary, silently.
1653    #[test]
1654    fn summaries_are_escaped_as_plain_text_not_sanitised_as_markup() {
1655        let records = vec![publication("p", "https://example.com")];
1656        let (site, pubn) = publication_from_records("p", &records).unwrap();
1657        let doc = |rkey: &str, body: &str| {
1658            rec(
1659                nsid::STANDARD_DOCUMENT,
1660                rkey,
1661                json!({
1662                    "title": "T",
1663                    "publishedAt": "2026-07-11T00:00:00Z",
1664                    "path": "/d",
1665                    "site": site,
1666                    "textContent": body,
1667                }),
1668            )
1669        };
1670        let summaries: Vec<String> = entries_from_records(
1671            &site,
1672            &pubn,
1673            &[
1674                doc("a", "Vec<String> is a type"),
1675                doc("b", "<script>alert(1)</script>"),
1676            ],
1677        )
1678        .into_iter()
1679        .filter_map(|e| e.summary)
1680        .collect();
1681        assert_eq!(
1682            summaries[0], "Vec&lt;String&gt; is a type",
1683            "prose was eaten by an HTML parser"
1684        );
1685        assert!(
1686            !summaries[1].contains("<script"),
1687            "escaping failed: {}",
1688            summaries[1]
1689        );
1690    }
1691
1692    /// **`Entry.url` is a vetted href or nothing.** The scheme guarantee lived
1693    /// only inside `publication_from_records`; `entries_from_records` and
1694    /// `Publication` are both `pub`, so any other constructor — step 3 building
1695    /// one from the stored `feeds` row, say — gave an unparseable base, and the
1696    /// no-base branch then emitted the document's `path` verbatim. A
1697    /// `javascript:` path became the entry link. This is the class `safe_link`
1698    /// exists for.
1699    #[test]
1700    fn an_entry_url_is_never_an_unvetted_path() {
1701        let pubn = Publication {
1702            name: None,
1703            // What a caller that did not go through `publication_from_records`
1704            // can hand this function.
1705            url: "not a url".to_string(),
1706        };
1707        let site = canonical("p");
1708        let docs = vec![
1709            document("a", &site, "Hostile", "javascript:alert(1)"),
1710            document("b", &site, "Fine", "https://example.com/ok"),
1711        ];
1712        let entries = entries_from_records(&site, &pubn, &docs);
1713        assert_eq!(
1714            entries[0].url, None,
1715            "an unvetted path became an entry link"
1716        );
1717        // Contract change: with no parseable base there is no origin to check,
1718        // so a well-formed absolute URL is refused too. `safe_link` alone vets
1719        // the SCHEME; it would have published a publisher-controlled host under
1720        // this publication's name. See `no_parseable_base_means_no_url_not_any_url`.
1721        assert_eq!(
1722            entries[1].url, None,
1723            "an off-origin absolute URL was published under the publication's name"
1724        );
1725    }
1726
1727    /// A document with no `publishedAt` is an entry with no date — the same
1728    /// answer a garbage one gets. The field the module is willing to discard
1729    /// must not be the one whose absence is fatal.
1730    #[test]
1731    fn a_document_without_published_at_is_still_an_entry() {
1732        let records = vec![publication("p", "https://example.com")];
1733        let (site, pubn) = publication_from_records("p", &records).unwrap();
1734        let doc = rec(
1735            nsid::STANDARD_DOCUMENT,
1736            "d",
1737            json!({ "title": "T", "path": "/d", "site": site }),
1738        );
1739        let entries = entries_from_records(&site, &pubn, &[doc]);
1740        assert_eq!(
1741            entries.len(),
1742            1,
1743            "a missing publishedAt dropped the document"
1744        );
1745        assert_eq!(entries[0].published, None);
1746    }
1747
1748    /// **`fetch` refuses a URI naming another collection, before the network.**
1749    /// It lists publications and matches on rkey alone, so without this an
1750    /// `app.bsky.feed.post` URI would be "read as a publication" whenever a
1751    /// publication in that repo shares the rkey. Storage enforces the
1752    /// collection today, but this function is `pub`.
1753    #[tokio::test]
1754    async fn fetch_refuses_a_uri_for_another_collection() {
1755        let uri = AtUri::parse(&format!("at://{DID}/app.bsky.feed.post/3lab")).unwrap();
1756        let err = fetch(&reqwest::Client::new(), "https://plc.example", &uri)
1757            .await
1758            .expect_err("read a feed post as a publication");
1759        assert!(
1760            format!("{err:#}").contains(nsid::STANDARD_PUBLICATION),
1761            "failed for the wrong reason: {err:#}"
1762        );
1763    }
1764
1765    /// **A publication on a subpath keeps it.** `Url::join` is RFC-3986, so a
1766    /// relative `posts/a` against `https://example.com/blog` resolves to
1767    /// `/posts/a` — dropping the subpath every permalink needs, while still
1768    /// passing the origin check. The base is normalised to a directory.
1769    #[test]
1770    fn a_subpath_publication_keeps_its_base_path() {
1771        let records = vec![publication("p", "https://example.com/blog")];
1772        let (site, pubn) = publication_from_records("p", &records).unwrap();
1773        let docs = vec![document("a", &site, "Relative", "posts/a")];
1774        let urls: Vec<Option<String>> = entries_from_records(&site, &pubn, &docs)
1775            .into_iter()
1776            .map(|e| e.url)
1777            .collect();
1778        assert_eq!(urls[0].as_deref(), Some("https://example.com/blog/posts/a"));
1779    }
1780
1781    /// With no parseable base there is no origin to check, so there is no URL
1782    /// — `safe_link` alone vets the scheme and would pass any absolute URL a
1783    /// publisher chose, under this publication's name.
1784    #[test]
1785    fn no_parseable_base_means_no_url_not_any_url() {
1786        let pubn = Publication {
1787            name: None,
1788            url: "not a url".to_string(),
1789        };
1790        let site = canonical("p");
1791        let docs = vec![document("a", &site, "Absolute", "https://evil.example/x")];
1792        let entries = entries_from_records(&site, &pubn, &docs);
1793        assert_eq!(
1794            entries[0].url, None,
1795            "an off-origin absolute URL was published"
1796        );
1797    }
1798
1799    /// **An off-origin path is dropped, not rewritten to the homepage.**
1800    /// Returning the base gave every affected entry the SAME href pointing at
1801    /// the site root — realistic whenever a publication's `url` is the apex
1802    /// and its documents sit on `www.` or a CDN domain. `None` is the honest
1803    /// answer, and the template already has a no-URL branch.
1804    #[test]
1805    fn an_off_origin_path_yields_no_url_rather_than_the_homepage() {
1806        let records = vec![publication("p", "https://example.com/blog")];
1807        let (site, pubn) = publication_from_records("p", &records).unwrap();
1808        let docs = vec![
1809            document("a", &site, "Elsewhere", "https://www.example.com/post"),
1810            document("b", &site, "Home", "/ok"),
1811        ];
1812        let urls: Vec<Option<String>> = entries_from_records(&site, &pubn, &docs)
1813            .into_iter()
1814            .map(|e| e.url)
1815            .collect();
1816        assert_eq!(
1817            urls[0], None,
1818            "an off-origin path was rewritten to the base"
1819        );
1820        assert_eq!(urls[1].as_deref(), Some("https://example.com/ok"));
1821    }
1822
1823    /// **A blank `path` is no URL, not the homepage** — the same answer the
1824    /// off-origin branch now gives, and for the same reason: several
1825    /// documents with an empty `path` otherwise became several entries all
1826    /// linking to the site root.
1827    #[test]
1828    fn a_blank_path_yields_no_url() {
1829        let records = vec![publication("p", "https://example.com/blog")];
1830        let (site, pubn) = publication_from_records("p", &records).unwrap();
1831        let docs = vec![
1832            document("a", &site, "Blank", ""),
1833            document("b", &site, "Spaces", "   "),
1834            document("c", &site, "Real", "/real"),
1835        ];
1836        let urls: Vec<Option<String>> = entries_from_records(&site, &pubn, &docs)
1837            .into_iter()
1838            .map(|e| e.url)
1839            .collect();
1840        assert_eq!(urls[0], None, "a blank path became the homepage");
1841        assert_eq!(urls[1], None, "a whitespace path became the homepage");
1842        assert_eq!(urls[2].as_deref(), Some("https://example.com/real"));
1843    }
1844
1845    /// A document without `path` keeps its title, date and summary — the
1846    /// policy `publishedAt` and `Entry.url` already follow.
1847    #[test]
1848    fn a_document_without_a_path_is_still_an_entry() {
1849        let records = vec![publication("p", "https://example.com")];
1850        let (site, pubn) = publication_from_records("p", &records).unwrap();
1851        let doc = rec(
1852            nsid::STANDARD_DOCUMENT,
1853            "d",
1854            json!({ "title": "T", "publishedAt": "2026-07-11T00:00:00Z", "site": site }),
1855        );
1856        let entries = entries_from_records(&site, &pubn, &[doc]);
1857        assert_eq!(
1858            entries.len(),
1859            1,
1860            "a missing path dropped the whole document"
1861        );
1862        assert_eq!(entries[0].title, "T");
1863        assert_eq!(entries[0].url, None);
1864    }
1865
1866    /// **A sibling is not an orphan.** A repo with an empty publication A and
1867    /// a busy publication B is the exact shape the `site` filter exists for;
1868    /// treating B's documents as a signal would warn on every poll of A
1869    /// forever. The signal is a document referencing a publication this repo
1870    /// does not have — a spelling nothing can ever match.
1871    #[test]
1872    fn documents_are_classified_keep_sibling_or_orphan() {
1873        let pubs = [
1874            publication("a", "https://example.com"),
1875            publication("b", "https://b.example"),
1876        ];
1877        let known: std::collections::HashSet<&str> = pubs.iter().map(|p| p.uri.as_str()).collect();
1878        let mine = canonical("a");
1879        let fate = |d: &crate::atproto::RecordEntry| classify_document(d, &mine, &known);
1880
1881        assert_eq!(
1882            fate(&document("d1", &mine, "Mine", "/1")),
1883            DocumentFate::Keep
1884        );
1885        assert_eq!(
1886            fate(&document("d2", &canonical("b"), "B's", "/2")),
1887            DocumentFate::Sibling,
1888            "a sibling publication's document is not an orphan"
1889        );
1890        assert_eq!(
1891            fate(&document(
1892                "d3",
1893                "at://did:plc:other/site.standard.publication/x",
1894                "?",
1895                "/3"
1896            )),
1897            DocumentFate::Orphan
1898        );
1899        assert_eq!(
1900            fate(&rec(
1901                nsid::STANDARD_DOCUMENT,
1902                "d4",
1903                json!({"title": "no rest"})
1904            )),
1905            DocumentFate::Malformed
1906        );
1907    }
1908
1909    /// `AtUri` uses the crate's one spelling of the prefix.
1910    #[test]
1911    fn at_uri_parsing_uses_the_shared_prefix() {
1912        let uri = format!(
1913            "{}{DID}/{}/abc",
1914            crate::atproto::AT_URI_PREFIX,
1915            nsid::STANDARD_PUBLICATION
1916        );
1917        assert!(AtUri::parse(&uri).is_some());
1918    }
1919
1920    /// The guid is the record's own URI. `path` is mutable; dedup is
1921    /// `UNIQUE (feed_id, guid)`.
1922    #[test]
1923    fn the_guid_is_the_record_uri_not_the_path() {
1924        let records = vec![publication("p", "https://example.com")];
1925        let (site, pubn) = publication_from_records("p", &records).unwrap();
1926        let docs = vec![document("rk1", &site, "T", "/moved")];
1927        let entries = entries_from_records(&site, &pubn, &docs);
1928        assert_eq!(
1929            entries[0].guid,
1930            format!("at://{DID}/{}/rk1", nsid::STANDARD_DOCUMENT)
1931        );
1932    }
1933
1934    /// One unreadable record must not cost a publisher the whole feed.
1935    #[test]
1936    fn a_malformed_document_is_skipped_rather_than_fatal() {
1937        let records = vec![publication("p", "https://example.com")];
1938        let (site, pubn) = publication_from_records("p", &records).unwrap();
1939        let docs = vec![
1940            rec(
1941                nsid::STANDARD_DOCUMENT,
1942                "bad",
1943                json!({ "title": "no rest" }),
1944            ),
1945            document("ok", &site, "Good", "/good"),
1946        ];
1947        let entries = entries_from_records(&site, &pubn, &docs);
1948        assert_eq!(entries.len(), 1);
1949        assert_eq!(entries[0].title, "Good");
1950    }
1951
1952    /// An rkey that is not in the repo is absent, not an error.
1953    #[test]
1954    fn a_missing_publication_is_none() {
1955        let records = vec![publication("other", "https://example.com")];
1956        assert!(publication_from_records("p", &records).is_none());
1957    }
1958
1959    #[test]
1960    fn at_uris_parse_in_both_forms_and_reject_malformed_ones() {
1961        let did = AtUri::parse(&format!("at://{DID}/site.standard.publication/abc")).unwrap();
1962        assert_eq!(did.authority, DID);
1963        assert_eq!(did.rkey, "abc");
1964        assert_eq!(
1965            did.to_string(),
1966            format!("at://{DID}/site.standard.publication/abc")
1967        );
1968        assert!(AtUri::parse("at://alice.example.com/site.standard.publication/abc").is_some());
1969        for bad in [
1970            "at://",
1971            "at://only-authority",
1972            "at://authority/collection",
1973            "at://authority/collection/",
1974            "at:///collection/rkey",
1975            "at://authority/collection/rkey/extra",
1976            "https://example.com/feed.xml",
1977            "at:authority/collection/rkey",
1978        ] {
1979            assert!(AtUri::parse(bad).is_none(), "parsed {bad:?}");
1980        }
1981    }
1982}