feather_reader/standard_site.rs
1//! Reading `standard.site` publications as feeds.
2//!
3//! A publication is not a feed document — it is a record in somebody's atproto
4//! repo, and its "entries" are separate records in the same repo. So this reads
5//! two collections rather than fetching one URL:
6//!
7//! ```text
8//! at://<did>/site.standard.publication/<rkey>
9//! ├─ resolve <did> → PDS
10//! ├─ getRecord site.standard.publication → name, url
11//! └─ listRecords site.standard.document → paged, filtered on `site`
12//! ```
13//!
14//! **Unauthenticated throughout.** This reads *someone else's* repo with no
15//! session, which is why it cannot reuse [`crate::oauth::xrpc::Repo`]: that type
16//! takes its base URL from the session's PDS, hardcodes `repo` to
17//! `session.sub`, and DPoP-signs every send. None of that survives contact with
18//! "read a stranger's repo".
19//!
20//! **Only `textContent` and `description` are read; `content` is ignored.**
21//! `content` is an open union — measured across 449 real documents it carried
22//! six different wrappers and twenty-two block types from five vendor
23//! namespaces, growing with every platform that adopts the lexicon, and it
24//! would drag an HTML-sanitisation surface over foreign input. A document with
25//! neither field still yields an entry: title, date and a link is what an RSS
26//! reader shows for a title-only feed, and is not a failure state.
27
28use serde::Deserialize;
29
30use crate::lexicon::nsid;
31
32/// Documents per `listRecords` page.
33///
34/// Smaller than the protocol default of 100 because a `site.standard.document`
35/// carries the whole article — ~17 KB measured, and the `content` union this
36/// module ignores is still in the wire bytes — while
37/// [`crate::net::read_capped`] bounds a response at 8 MB. 100 long-form
38/// articles per page can exceed that and fail the walk outright.
39const DOCUMENT_PAGE_SIZE: u32 = 25;
40
41/// A parsed `at://` URI: `at://<authority>/<collection>/<rkey>`.
42///
43/// Parsed by hand rather than with `url::Url`, which **cannot read the form that
44/// matters**: `at://did:plc:…/…` fails with *invalid port number*, because the
45/// colons in the DID are taken as a port separator. The handle form parses
46/// fine, which is what makes the failure easy to miss.
47#[derive(Debug, Clone, PartialEq, Eq)]
48pub struct AtUri {
49 pub authority: String,
50 pub collection: String,
51 pub rkey: String,
52}
53
54impl AtUri {
55 /// Parse, or `None` if this is not a well-formed three-segment at-URI.
56 pub fn parse(uri: &str) -> Option<Self> {
57 let rest = uri.strip_prefix(crate::atproto::AT_URI_PREFIX)?;
58 let mut parts = rest.split('/');
59 let (authority, collection, rkey) = (parts.next()?, parts.next()?, parts.next()?);
60 if parts.next().is_some()
61 || authority.is_empty()
62 || collection.is_empty()
63 || rkey.is_empty()
64 {
65 return None;
66 }
67 Some(Self {
68 authority: authority.to_string(),
69 collection: collection.to_string(),
70 rkey: rkey.to_string(),
71 })
72 }
73}
74
75impl std::fmt::Display for AtUri {
76 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
77 write!(
78 f,
79 "at://{}/{}/{}",
80 self.authority, self.collection, self.rkey
81 )
82 }
83}
84
85/// What one read of a publication produced.
86///
87/// **`complete` is the fact the store cannot recover afterwards.** A truncated
88/// walk and a short publication return the same entries; the difference decides
89/// whether an empty result is a quiet blog or a read that gave up, and the
90/// caller has no other way to tell. `fetch` used to drop it on the floor after
91/// logging a warning.
92#[derive(Debug)]
93pub struct PublicationRead {
94 pub publication: Publication,
95 pub entries: Vec<Entry>,
96 pub complete: bool,
97}
98
99/// The publication record — a pointer, not a feed. Supplies the title and the
100/// base URL that document `path`s are joined onto.
101#[derive(Debug, Clone, PartialEq, Eq)]
102pub struct Publication {
103 pub name: Option<String>,
104 pub url: String,
105}
106
107/// One document, mapped onto the shape the feed pipeline already stores.
108#[derive(Debug, Clone, PartialEq, Eq)]
109pub struct Entry {
110 /// The document's own `at://` URI.
111 ///
112 /// **Not derived from `path`.** `path` is mutable — a publisher who moves an
113 /// article would duplicate their whole archive on the next poll, because
114 /// dedup is `UNIQUE (feed_id, guid)`.
115 pub guid: String,
116 pub title: String,
117 /// When the document was published, re-spelled by `crate::feed::fmt_time`
118 /// into the store's one RFC3339 shape. The reading order sorts on this
119 /// column as a string, so a publisher's spelling cannot go in verbatim.
120 ///
121 /// Taken from `publishedAt` when it parses AND is not in the future,
122 /// otherwise from the TID in the record key, otherwise `None`. See
123 /// `entries_from_records` for why a future date is discarded rather than
124 /// clamped, and why an undated entry is left undated.
125 pub published: Option<String>,
126 /// The joined, **scheme-vetted** permalink — `None` when the document's
127 /// `path` does not resolve to a safe href on the publication's origin.
128 /// `Option` because the guarantee cannot be met unconditionally and the
129 /// store's column is optional too; a title-only entry is not a failure.
130 pub url: Option<String>,
131 /// `description`, else `textContent`, escaped by
132 /// `crate::feed::plain_text_to_html` — both are plain text in the
133 /// lexicon, and the column they land in is rendered as HTML.
134 pub summary: Option<String>,
135}
136
137impl From<Entry> for crate::store::NewEntry {
138 /// The shape the poller stores. Kept here, next to the fields it maps,
139 /// so wiring the reader to the scheduler has nothing left to decide.
140 fn from(e: Entry) -> Self {
141 crate::store::NewEntry {
142 guid: e.guid,
143 url: e.url,
144 title: Some(e.title),
145 author: None,
146 published: e.published,
147 content_html: e.summary,
148 fetched_at: None,
149 }
150 }
151}
152
153#[derive(Debug, Deserialize)]
154struct PublicationValue {
155 name: Option<String>,
156 url: String,
157}
158
159#[derive(Debug, Deserialize)]
160struct DocumentValue {
161 title: String,
162 /// Optional so a document without one is an entry with no date, the same
163 /// answer a garbage one gets — the strictness ran the other way, making
164 /// the field this module is willing to DISCARD the one whose absence was
165 /// fatal to the whole record.
166 #[serde(rename = "publishedAt")]
167 published_at: Option<String>,
168 /// Optional for the same reason as `publishedAt`: a document with no
169 /// `path` keeps its title, date and summary rather than vanishing from
170 /// the feed entirely. `Entry.url` is already `Option`.
171 path: Option<String>,
172 /// The at-URI of the publication this document belongs to.
173 ///
174 /// **Load-bearing.** A repo can hold several publications — measured, some
175 /// do — so documents must be filtered by this rather than assumed to belong
176 /// to the one being polled.
177 site: String,
178 #[serde(rename = "textContent")]
179 text_content: Option<String>,
180 description: Option<String>,
181}
182
183/// Find the publication named by `rkey` among a repo's publication records.
184///
185/// Returns its **canonical** at-URI — the one the PDS itself minted — alongside
186/// the record. That canonical URI is what documents reference in their `site`
187/// field, so it is the key the filter uses, rather than the string the reader
188/// subscribed with. Storage is DID-form only (#164), so today the two agree;
189/// taking the PDS's spelling keeps them agreeing if they ever stop.
190pub fn publication_from_records(
191 rkey: &str,
192 records: &[crate::atproto::RecordEntry],
193) -> Option<(String, Publication)> {
194 let entry = records
195 .iter()
196 .find(|r| AtUri::parse(&r.uri).is_some_and(|u| u.rkey == rkey))?;
197 let value: PublicationValue = serde_json::from_value(entry.value.clone()).ok()?;
198 // **The base of every Entry.url, so it is vetted as the href it becomes.**
199 // `net::safe_link` is the same check the RSS entry pipeline applies at
200 // `feed.rs`, and the reason `safe_link.rs` exists as a type at all: the
201 // procedural version of this guarantee was deleted once with a green suite.
202 let url = crate::net::safe_link(&value.url)?;
203 Some((
204 entry.uri.clone(),
205 Publication {
206 name: value.name,
207 url,
208 },
209 ))
210}
211
212/// Map a repo's document records onto entries, keeping only those belonging to
213/// `canonical_site`.
214pub fn entries_from_records(
215 canonical_site: &str,
216 publication: &Publication,
217 records: &[crate::atproto::RecordEntry],
218) -> Vec<Entry> {
219 // **Normalised to a directory.** `Url::join` is RFC-3986: against a base of
220 // `https://example.com/blog`, a relative `posts/a` resolves to
221 // `/posts/a`, silently dropping the subpath every permalink needs. A
222 // trailing slash makes the base a directory, which is what a publication
223 // URL means.
224 let base = url::Url::parse(&publication.url).ok().map(|mut u| {
225 if !u.path().ends_with('/') {
226 u.set_path(&format!("{}/", u.path()));
227 }
228 u
229 });
230 // One "now" for the whole batch, so two documents of the same poll are
231 // judged against the same instant rather than one being called credible and
232 // an identical one not. (`tid_timestamp` reads the clock again for its own
233 // ceiling; that only ever tightens a bound five minutes away, so it needs
234 // no such agreement.)
235 let now = chrono::Utc::now();
236 // The SAME allowance the record-key branch gets. A publisher's clock runs
237 // ahead of ours as readily as a PDS's does, and judging a stated date
238 // against a bare `now` while the rkey below gets five minutes discarded a
239 // perfectly good date and left the newest post undated — which, ordered on
240 // a bare `published DESC`, puts it at the bottom of the list.
241 let ceiling = now + chrono::Duration::seconds(crate::atproto::CLOCK_SKEW_GRACE_SECS);
242 records
243 .iter()
244 .filter_map(|record| {
245 // A document that does not deserialise is SKIPPED, not fatal: one
246 // malformed record must not cost a publisher its whole feed.
247 let doc: DocumentValue = serde_json::from_value(record.value.clone()).ok()?;
248 if doc.site != canonical_site {
249 return None;
250 }
251 Some(Entry {
252 guid: record.uri.clone(),
253 title: doc.title,
254 // **Three rules, in order: what the publisher credibly said,
255 // then when the record was written, then nothing.**
256 //
257 // A stated date in the future is DISCARDED, not clamped to now.
258 // The store refreshes `published` on every poll but stamps
259 // `fetched_at` only once, so a date derived from the current
260 // clock is rewritten hourly: the row stays the newest thing in
261 // the feed forever, is never swept, is never evicted by the
262 // per-feed cap, and sits at the top of the reading list re-dated
263 // to today. Clamping moved the defect rather than fixing it.
264 // Every fall-through here holds still instead — the TID is the
265 // record's real write time, and an undated row is dated by
266 // `fetched_at`, which never moves. An rkey that is not a TID
267 // leaves the entry undated rather than inventing a date from a
268 // slug.
269 published: doc
270 .published_at
271 .as_deref()
272 .and_then(|raw| chrono::DateTime::parse_from_rfc3339(raw).ok())
273 .map(|d| d.with_timezone(&chrono::Utc))
274 .filter(|d| *d <= ceiling)
275 .or_else(|| {
276 AtUri::parse(&record.uri)
277 .and_then(|uri| crate::atproto::tid_timestamp(&uri.rkey))
278 })
279 .map(crate::feed::fmt_time),
280 // `non_blank` for the same reason the summary uses it: a blank
281 // path joins to the publication's own base, so a handful of
282 // documents with an empty `path` became a handful of entries
283 // all linking to the site root.
284 url: non_blank(doc.path)
285 .as_deref()
286 .and_then(|path| join_path(base.as_ref(), path)),
287 // `description` first — the authored summary — but only when it
288 // actually says something: a blank one must not shadow the body.
289 // Then ESCAPED, not sanitised: both fields are plain text.
290 summary: non_blank(doc.description)
291 .or_else(|| non_blank(doc.text_content))
292 .map(|raw| crate::feed::plain_text_to_html(&raw)),
293 })
294 })
295 .collect()
296}
297
298/// The ingest floor: the oldest `published` an entry may carry and still be worth
299/// storing, or `None` for "store anything".
300///
301/// **The floor is the window the SWEEP would use, which is not the smaller of the
302/// two.** `store::prune_old_entries` honours the hard ceiling only when it is
303/// strictly older than the rolling window, or when there is no window at all
304/// (`hard_days > 0 && (days <= 0 || hard_days > days)`); a ceiling inside the
305/// window is logged and ignored there, because the hard delete spares nothing and
306/// would otherwise delete exactly the rows the soft delete exists to spare.
307///
308/// So the earliest thing that can delete a row is `retention_days` when there is
309/// a window, and the ceiling only when there is not.
310///
311/// **The pair comes from [`crate::config::Config::retention_for`], and for a
312/// publication it is not the RSS window.** That function is the one home for which
313/// window applies to which kind, precisely so this and the sweep cannot drift: a
314/// publication gets `(0, publication_retention_days)` — no rolling window, and a
315/// generous archive ceiling — because a 14-day window stored ZERO rows from every
316/// real publication measured, their newest documents being 109 to 241 days old.
317/// Fed that pair, the `days == 0` branch below falls through to the ceiling, which
318/// is exactly the sweep pass that can delete such a row.
319///
320/// `a_publications_floor_is_its_archive_ceiling_not_the_rss_window` asserts the
321/// two halves agree; it is the test that fails if either side is changed alone.
322///
323/// Two wrong versions, both worth naming. Keying on `retention_days` alone left
324/// `retention_days = 0` unfloored while the ceiling still deleted at
325/// `retention_hard_days` — the exact cycle the floor exists to prevent, and a
326/// worse one, since the ceiling spares nothing and a starred entry came back
327/// unstarred rather than merely unread. Reaching for the MINIMUM then
328/// over-corrected: at `days = 180, hard = 30` the sweep ignores the ceiling and
329/// deletes nothing before 180 days, while `min` floored ingest at 30 and silently
330/// discarded five months of a publisher's archive that nothing would have deleted.
331///
332/// **An unrepresentable window is no floor, said directly.** `RETENTION_DAYS`
333/// parses into a `u32` with no upper bound, and `u32::MAX` days is an operator
334/// saying "keep everything"; both the duration and the subtraction can fail, and
335/// either failure means `None` here. The previous shape fell back to a sentinel
336/// instant (`MIN_UTC`) and left the outcome resting on the fact that the row
337/// comparison is LEXICOGRAPHIC: `fmt_time` renders an out-of-range year with a
338/// `+`/`-` sign, which sorts either side of a 4-digit year by ASCII accident
339/// rather than by date. Verified: swapping that fallback to `MAX_UTC` — a floor
340/// of the year 262143, which should discard every entry in existence — changed
341/// nothing, because `"2026…" > "+262143…"`. With `None` the ambiguity is gone,
342/// and the comparison never sees a date outside the range it can order.
343///
344/// `now` is a parameter so this is testable without the clock.
345fn ingest_floor(
346 retention_days: u32,
347 retention_hard_days: u32,
348 now: chrono::DateTime<chrono::Utc>,
349) -> Option<String> {
350 let days = if retention_days > 0 {
351 retention_days
352 } else if retention_hard_days > 0 {
353 retention_hard_days
354 } else {
355 return None;
356 };
357 let window = chrono::Duration::try_days(days.into())?;
358 now.checked_sub_signed(window).map(crate::feed::fmt_time)
359}
360
361/// Persist one publication read, and say what the poll amounted to.
362///
363/// `retention_days` is the ingest floor: an entry whose stated date is already
364/// older than the window is not stored, because storing it means the next sweep
365/// deletes it and the next poll re-inserts it with a new row id — which loses its
366/// read state and arrives unread, on that cycle, forever.
367///
368/// **For a publication the caller passes `Config::retention_for`'s pair, which is
369/// the archive ceiling rather than the 14-day window** — so in practice almost
370/// nothing is floored out here, which is the point: measured, a 14-day floor
371/// dropped every document of every real publication tried. See `ingest_floor`.
372///
373/// **An entry with no date is stored anyway.** That is a decision, not an
374/// oversight: it is dated by `fetched_at` instead, which does not move, and the
375/// alternative is discarding an article the reader can never see.
376///
377/// Three consequences, stated because an earlier version of this comment named
378/// only the first and called it "accepted":
379///
380/// * It resurrects once per retention window. `fetched_at` is stamped at the
381/// first insert and never refreshed, so the row ages out, is swept once the
382/// reader has read it, and is re-inserted unread by the next poll.
383/// * Under the hard ceiling it comes back **unstarred as well**. The soft sweep
384/// spares `starred = 1 OR read = 0`; the ceiling spares nothing.
385/// * Until then it OUTRANKS the publication's real articles. Both the sweep and
386/// the per-feed cap order on `COALESCE(published, fetched_at)`, so an undated
387/// entry sorts by the moment it was fetched — that is, as the newest thing in
388/// the feed — and a publication whose documents are mostly undated can push
389/// dated articles out of `max_entries_per_feed`.
390///
391/// **`new_entries` is an upper bound, not a count of rows that survived.**
392/// `insert_entries` reports what it inserted, and the per-feed cap then trims
393/// within the same call, so a poll that inserted 30 rows into a feed capped at 10
394/// can report 30.
395pub async fn store_publication(
396 pool: &sqlx::SqlitePool,
397 url: &str,
398 read: PublicationRead,
399 max_entries_per_feed: i64,
400 retention_days: u32,
401 retention_hard_days: u32,
402) -> anyhow::Result<crate::feed::PollOutcome> {
403 let offered = read.entries.len();
404
405 // **The failure check runs before anything is written.** Stamping the feed
406 // and then returning `Failed` is the shape that makes a broken publication
407 // read as freshly polled on `/stats`, and it is easy to write by accident
408 // because the upsert is the natural first step.
409 //
410 // **Keyed on what was OFFERED, not on what survives the floor.** A truncated
411 // read that produced nothing means the walk gave up before its first record.
412 // A truncated read whose entries are merely older than the window is a
413 // healthy poll of an old publication, and calling that a failure puts it into
414 // backoff that widens forever.
415 if !read.complete && offered == 0 {
416 return Ok(crate::feed::PollOutcome::Failed {
417 backoff: crate::feed::backoff_for(1),
418 kind: crate::feed::FailureKind::Body,
419 detail: crate::feed::failure_detail(
420 "the publication read stopped before its first document",
421 ),
422 });
423 }
424
425 // See [`ingest_floor`] for which of the two windows this is, and why.
426 let floor = ingest_floor(retention_days, retention_hard_days, chrono::Utc::now());
427 let rows: Vec<crate::store::NewEntry> = read
428 .entries
429 .into_iter()
430 .filter(|e| match (&floor, &e.published) {
431 // Lexicographic on the store's one RFC3339 spelling, which sorts
432 // chronologically by construction.
433 (Some(floor), Some(published)) => published.as_str() >= floor.as_str(),
434 // No floor, or no date. The undated case is the decision the doc
435 // above records: kept, dated by `fetched_at`, and it resurrects once
436 // per window.
437 _ => true,
438 })
439 .map(Into::into)
440 .collect();
441
442 // **Say when the floor emptied the read.** A complete read of three
443 // year-old posts and a complete read of an empty publication both return
444 // `Updated { new_entries: 0 }`, stamp the feed, and look green — so a
445 // subscriber to an archived blog gets a blank feed and nothing anywhere says
446 // why. `offered` is already in hand.
447 if offered > rows.len() {
448 tracing::info!(
449 feed = %url,
450 offered,
451 stored = rows.len(),
452 "the retention floor dropped documents older than the window"
453 );
454 }
455
456 let feed_id = crate::store::upsert_feed(
457 pool,
458 &crate::store::NewFeed {
459 url: url.to_string(),
460 title: read.publication.name.clone(),
461 site_url: Some(read.publication.url.clone()),
462 last_polled: Some(crate::feed::fmt_time(chrono::Utc::now())),
463 ..Default::default()
464 },
465 )
466 .await?;
467
468 let new_entries =
469 crate::store::insert_entries(pool, feed_id, &rows, max_entries_per_feed).await?;
470 Ok(crate::feed::PollOutcome::Updated { new_entries })
471}
472
473fn non_blank(s: Option<String>) -> Option<String> {
474 s.filter(|v| !v.trim().is_empty())
475}
476
477/// Join a document `path` onto the publication's base URL.
478///
479/// **`Url::join`, not concatenation.** Concatenating produced
480/// `https://x.com/https://evil.example/a` for an absolute path and buried the
481/// path inside the query for a base carrying one. `join` also keeps the result
482/// on the publication's own origin for a relative path, which is the only shape
483/// the lexicon describes.
484fn join_path(base: Option<&url::Url>, path: &str) -> Option<String> {
485 // **`safe_link` on the way out, not only on the base.** The scheme
486 // guarantee used to live solely in `publication_from_records`; this
487 // function and `Publication` are both `pub`, so a caller that built a
488 // `Publication` some other way (the step-3 poller, from a stored row) gave
489 // an unparseable base — and the no-base branch then returned the
490 // document's `path` verbatim, putting `javascript:` into an entry link.
491 // **No base, no URL.** This branch used to return `safe_link(path)`, which
492 // vets the scheme but NOT the origin — so a caller holding a `Publication`
493 // it did not build through `publication_from_records` (the step-3 poller,
494 // from a stored row whose `site_url` is NULL or malformed) would publish a
495 // publisher-controlled `https://evil.example/x` as a permalink under that
496 // publication's name. The two branches agree now: off-origin is `None`, and
497 // "no origin to be off" is also `None`.
498 let base = base?;
499 match base.join(path) {
500 // A path that resolves off the publication's origin is not a path, it
501 // is a redirect the publisher smuggled into a field we render as theirs.
502 Ok(joined) if joined.origin() == base.origin() => crate::net::safe_link(joined.as_str()),
503 // **No URL, not the homepage.** Falling back to the base gave every
504 // affected entry the same href pointing at the site root — which is
505 // what a publication on an apex domain whose documents live on `www.`
506 // or a CDN would produce for its whole archive, with nothing to say
507 // anything had been dropped.
508 _ => None,
509 }
510}
511
512/// What the document walk should do with one record.
513///
514/// A pure decision so it can be tested without a PDS: the walk itself is a
515/// closure over the network.
516#[derive(Debug, Clone, Copy, PartialEq, Eq)]
517enum DocumentFate {
518 /// Belongs to the publication being read.
519 Keep,
520 /// Belongs to another publication **in this repo** — normal, and the
521 /// reason the `site` filter exists. Not a signal of anything.
522 Sibling,
523 /// References a publication this repo does not have: a `site` spelling
524 /// nothing can ever match. Indistinguishable from a quiet blog without
525 /// saying so, which is why it is counted.
526 Orphan,
527 /// Not a document this reader understands.
528 Malformed,
529}
530
531fn classify_document(
532 record: &crate::atproto::RecordEntry,
533 canonical_site: &str,
534 known: &std::collections::HashSet<&str>,
535) -> DocumentFate {
536 match serde_json::from_value::<DocumentValue>(record.value.clone()) {
537 Ok(doc) if doc.site == canonical_site => DocumentFate::Keep,
538 Ok(doc) if known.contains(doc.site.as_str()) => DocumentFate::Sibling,
539 Ok(_) => DocumentFate::Orphan,
540 Err(_) => DocumentFate::Malformed,
541 }
542}
543
544/// Read a publication and its documents through the hardened anonymous client.
545///
546/// **Deliberately thin.** Everything that makes this fetch safe already exists
547/// in [`crate::atproto`] and is reused rather than rebuilt:
548///
549/// - [`crate::atproto::resolve_did_to_pds`] runs
550/// [`crate::net::assert_public_target`] on the `serviceEndpoint`, which is a
551/// stranger's string;
552/// - every read goes through [`crate::net::guarded_get_no_privacy`], re-vetting
553/// per hop and pinning the connection, which closes the rebinding window and
554/// supplies the `User-Agent` that 4 of 19 measured endpoints demand;
555/// - [`crate::net::read_capped`] bounds each response;
556/// - [`crate::atproto::PdsClient::list_all_records`] bounds the page count AND
557/// detects a repeated or absent cursor — the trap this module's first draft
558/// walked into, already solved there;
559/// - an XRPC error envelope surfaces as [`crate::atproto::AtProtoError::Xrpc`] rather
560/// than deserialising into an empty page.
561///
562/// The first draft of this module reimplemented all of that, worse. The only
563/// logic left here is the part that is genuinely about standard.site.
564pub async fn fetch(
565 http: &reqwest::Client,
566 plc_directory: &str,
567 uri: &AtUri,
568) -> anyhow::Result<PublicationRead> {
569 use anyhow::Context;
570
571 // The collection is part of the identity of what was subscribed to, and
572 // this function is `pub`: without the check it lists publications and
573 // matches on rkey alone, so `at://did/app.bsky.feed.post/<rkey>` would be
574 // "read as a publication" whenever a publication shares that rkey.
575 anyhow::ensure!(
576 uri.collection == nsid::STANDARD_PUBLICATION,
577 "{uri} is not a {} URI",
578 nsid::STANDARD_PUBLICATION
579 );
580
581 let pds = crate::atproto::resolve_did_to_pds(http, plc_directory, &uri.authority)
582 .await
583 .with_context(|| format!("resolving the PDS for {}", uri.authority))?;
584 let client = crate::atproto::PdsClient::anonymous(http.clone(), pds, uri.authority.clone());
585
586 // **One budget for both walks, because they nest.** The publications are
587 // still held when the documents walk runs — `known` borrows them, and the
588 // `keep` closure captures that — so two independent ceilings let this one
589 // read hold twice the bound the box was sized for. Threading a single budget
590 // makes the limit a property of the read rather than of each walk, and the
591 // borrow checker keeps it that way where a comment would not.
592 let mut budget = crate::atproto::ByteBudget::new(crate::atproto::MAX_LIST_BYTES);
593 let publications = client
594 .list_all_records_within(nsid::STANDARD_PUBLICATION, &mut budget)
595 .await
596 .with_context(|| format!("listing publications for {}", uri.authority))?;
597 let (canonical_site, publication) = publication_from_records(&uri.rkey, &publications)
598 .with_context(|| format!("{uri} is not a readable site.standard.publication"))?;
599
600 // **Filtered inside the walk, so the cap counts THIS publication's
601 // documents.** A repo-wide cap applied before the filter starves a quiet
602 // publication whose busy sibling fills the window — it returns nothing,
603 // permanently, and worse with every post the sibling makes.
604 //
605 // The page is smaller than the protocol default because a document
606 // carries the whole article (~17 KB measured, and the `content` union this
607 // module ignores is still on the wire): 100 long-form articles per page
608 // can exceed `read_capped`'s 8 MB and fail the walk outright.
609 let known: std::collections::HashSet<&str> =
610 publications.iter().map(|p| p.uri.as_str()).collect();
611 let mut orphaned = 0usize;
612 let documents = client
613 .list_recent_matching_within(
614 nsid::STANDARD_DOCUMENT,
615 crate::atproto::MAX_LARGE_RECORDS,
616 &mut budget,
617 DOCUMENT_PAGE_SIZE,
618 // Orphans are counted while WALKING, not over the kept window: a
619 // truncated slice would both miss orphans and stay silent in
620 // exactly the case where the feed went empty structurally.
621 |record| match classify_document(record, &canonical_site, &known) {
622 DocumentFate::Keep => true,
623 DocumentFate::Orphan => {
624 orphaned += 1;
625 false
626 }
627 DocumentFate::Sibling | DocumentFate::Malformed => false,
628 },
629 )
630 .await
631 .with_context(|| format!("listing documents for {canonical_site}"))?;
632 Ok(read_from(publication, &canonical_site, documents, orphaned))
633}
634
635/// Assemble a [`PublicationRead`] from a finished walk.
636///
637/// **Extracted so `complete` is testable.** It is the one fact the store cannot
638/// recover afterwards, and reaching the `false` case through `fetch` needs a PLC
639/// directory, a PDS, and a walk that truncates — three mocked hosts for one
640/// boolean. Reaching it here needs a struct.
641fn read_from(
642 publication: Publication,
643 canonical_site: &str,
644 documents: crate::atproto::RecordWalk,
645 orphaned: usize,
646) -> PublicationRead {
647 let entries = entries_from_records(canonical_site, &publication, &documents.records);
648
649 // **A walk that stopped early is not a short archive.** Reading part of a
650 // publication is acceptable; reporting it as the whole of one is not, and
651 // when the part is empty — a quiet publication whose busy sibling fills
652 // every page this reader will fetch — the feed looks healthy and stays
653 // empty forever.
654 // Logged AND returned. Logging it alone is how the fact that the walk gave
655 // up reached nobody who could act on it: a truncated read and a short
656 // publication produce the same entries, and only the caller can tell the
657 // difference between "quiet blog" and "we stopped early".
658 if !documents.complete {
659 tracing::warn!(
660 site = %canonical_site,
661 kept = entries.len(),
662 "stopped reading this publication before its documents ran out"
663 );
664 }
665 // **A spelling mismatch on `site` looks exactly like an empty
666 // publication.** A publication with no documents is normal, so the poller
667 // would call this healthy forever; if the repo HAD documents and none
668 // matched, say so, because that is the shape of a bug rather than of a
669 // quiet blog.
670 if orphaned > 0 {
671 tracing::warn!(
672 site = %canonical_site,
673 orphaned,
674 "documents in this repo reference no publication in it — a `site` spelling nothing matches"
675 );
676 }
677 PublicationRead {
678 publication,
679 entries,
680 complete: documents.complete,
681 }
682}
683
684#[cfg(test)]
685mod tests {
686 use super::*;
687 use crate::atproto::RecordEntry;
688 use serde_json::json;
689
690 const DID: &str = "did:plc:ohutz6x5acjmpuulp3x7wxxc";
691
692 /// **A walk that gave up must not report itself as a whole publication.**
693 ///
694 /// `fetch` used to log that and return only the entries, so the fact reached
695 /// nobody who could act on it — and a truncated read and a quiet blog produce
696 /// exactly the same entries.
697 #[test]
698 fn a_read_carries_whether_the_walk_finished() {
699 let publication = Publication {
700 name: Some("Scan's Lab".to_string()),
701 url: "https://example.com/blog/".to_string(),
702 };
703 for complete in [true, false] {
704 let walk = crate::atproto::RecordWalk {
705 records: Vec::new(),
706 complete,
707 };
708 let read = read_from(publication.clone(), "at://d/c/r", walk, 0);
709 assert_eq!(
710 read.complete, complete,
711 "the walk said complete={complete} and the read said {}",
712 read.complete
713 );
714 }
715 }
716
717 // -- storing a publication read -----------------------------------------
718
719 fn read_of(entries: Vec<Entry>, complete: bool) -> PublicationRead {
720 PublicationRead {
721 publication: Publication {
722 name: Some("Scan's Lab".to_string()),
723 url: "https://example.com/blog/".to_string(),
724 },
725 entries,
726 complete,
727 }
728 }
729
730 fn entry_dated(guid: &str, published: Option<&str>) -> Entry {
731 Entry {
732 guid: guid.to_string(),
733 title: "T".to_string(),
734 published: published.map(str::to_string),
735 url: Some("https://example.com/blog/a".to_string()),
736 summary: None,
737 }
738 }
739
740 fn days_ago(n: i64) -> String {
741 crate::feed::fmt_time(chrono::Utc::now() - chrono::Duration::days(n))
742 }
743
744 const PUB_URL: &str = "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab";
745
746 async fn pool() -> sqlx::SqlitePool {
747 crate::store::init_url("sqlite::memory:").await.unwrap()
748 }
749
750 #[tokio::test]
751 async fn a_complete_read_stores_its_entries_and_reports_them() {
752 let pool = pool().await;
753 let read = read_of(
754 vec![
755 entry_dated("at://d/c/1", Some(&days_ago(1))),
756 entry_dated("at://d/c/2", Some(&days_ago(2))),
757 ],
758 true,
759 );
760 let outcome = store_publication(&pool, PUB_URL, read, 0, 14, 180)
761 .await
762 .unwrap();
763 assert!(
764 matches!(
765 outcome,
766 crate::feed::PollOutcome::Updated { new_entries: 2 }
767 ),
768 "expected two new entries, got {outcome:?}"
769 );
770 let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
771 .fetch_one(&pool)
772 .await
773 .unwrap();
774 assert_eq!(n, 2, "the entries were not stored");
775 }
776
777 /// **Starvation is keyed on what was OFFERED, not on what was stored.**
778 ///
779 /// A truncated read that produced nothing is a failure: the walk gave up
780 /// before the first record. But a truncated read whose entries the retention
781 /// floor dropped is not — the read worked, the entries are simply older than
782 /// the window, and calling that a failure puts a healthy publication into
783 /// backoff forever.
784 #[tokio::test]
785 async fn an_incomplete_read_that_offered_nothing_is_a_failure() {
786 let pool = pool().await;
787 let outcome = store_publication(&pool, PUB_URL, read_of(vec![], false), 0, 14, 180)
788 .await
789 .unwrap();
790 assert!(
791 matches!(outcome, crate::feed::PollOutcome::Failed { .. }),
792 "a truncated read that produced nothing is not a healthy poll: {outcome:?}"
793 );
794 }
795
796 #[tokio::test]
797 async fn an_incomplete_read_whose_entries_the_floor_dropped_is_not_a_failure() {
798 let pool = pool().await;
799 let read = read_of(
800 vec![entry_dated("at://d/c/old", Some(&days_ago(900)))],
801 false,
802 );
803 let outcome = store_publication(&pool, PUB_URL, read, 0, 14, 180)
804 .await
805 .unwrap();
806 assert!(
807 !matches!(outcome, crate::feed::PollOutcome::Failed { .. }),
808 "the read offered an entry; the floor dropping it is not a failed poll: {outcome:?}"
809 );
810 }
811
812 /// A failed poll must not look like a successful one on `/stats`.
813 #[tokio::test]
814 async fn a_failed_read_does_not_stamp_last_polled() {
815 let pool = pool().await;
816 let outcome = store_publication(&pool, PUB_URL, read_of(vec![], false), 0, 14, 180)
817 .await
818 .unwrap();
819 assert!(matches!(outcome, crate::feed::PollOutcome::Failed { .. }));
820 let stamped: Option<String> =
821 sqlx::query_scalar("SELECT last_polled FROM feeds WHERE url = ?1")
822 .bind(PUB_URL)
823 .fetch_optional(&pool)
824 .await
825 .unwrap()
826 .flatten();
827 assert_eq!(
828 stamped, None,
829 "a failed poll stamped last_polled, so the feed reads as freshly polled"
830 );
831 }
832
833 #[tokio::test]
834 async fn an_entry_already_older_than_the_window_is_not_stored() {
835 let pool = pool().await;
836 let read = read_of(
837 vec![
838 entry_dated("at://d/c/fresh", Some(&days_ago(1))),
839 entry_dated("at://d/c/ancient", Some(&days_ago(900))),
840 ],
841 true,
842 );
843 store_publication(&pool, PUB_URL, read, 0, 14, 180)
844 .await
845 .unwrap();
846 let guids: Vec<String> = sqlx::query_scalar("SELECT guid FROM entries ORDER BY guid")
847 .fetch_all(&pool)
848 .await
849 .unwrap();
850 assert_eq!(
851 guids,
852 vec!["at://d/c/fresh".to_string()],
853 "an entry the next sweep would delete was stored anyway"
854 );
855 }
856
857 /// **The floor follows whichever window actually deletes, not the rolling one.**
858 ///
859 /// `retention_days = 0` is a supported configuration meaning "no rolling
860 /// window", and the hard ceiling stays alive independently. Keying the ingest
861 /// floor on the rolling window alone let a 900-day-old entry in, which the
862 /// ceiling then deleted and the next poll re-inserted — and the ceiling spares
863 /// nothing, so a starred entry came back unstarred.
864 #[tokio::test]
865 async fn the_floor_follows_the_hard_ceiling_when_the_window_is_disabled() {
866 let pool = pool().await;
867 let read = read_of(
868 vec![entry_dated("at://d/c/ancient", Some(&days_ago(900)))],
869 true,
870 );
871 store_publication(&pool, PUB_URL, read, 0, 0, 180)
872 .await
873 .unwrap();
874 let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
875 .fetch_one(&pool)
876 .await
877 .unwrap();
878 assert_eq!(
879 n, 0,
880 "an entry the hard ceiling will delete was stored, so it will resurrect"
881 );
882 }
883
884 /// **The WINDOW is the floor when there is one, because it deletes first.**
885 ///
886 /// With a 14-day rolling window and a 180-day ceiling, an entry 100 days old
887 /// is inside the ceiling and outside the window — so the sweep takes it once
888 /// it has been read and the next poll puts it back. Taking the longer of the
889 /// two would store it.
890 #[tokio::test]
891 async fn the_floor_follows_the_window_when_both_are_set() {
892 let pool = pool().await;
893 let read = read_of(
894 vec![entry_dated("at://d/c/hundred", Some(&days_ago(100)))],
895 true,
896 );
897 store_publication(&pool, PUB_URL, read, 0, 14, 180)
898 .await
899 .unwrap();
900 let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
901 .fetch_one(&pool)
902 .await
903 .unwrap();
904 assert_eq!(
905 n, 0,
906 "an entry inside the ceiling but outside the window was stored, so it will cycle"
907 );
908 }
909
910 /// With both windows off there is no floor, because nothing will delete it.
911 /// **The two halves of the retention policy, asserted against each other.**
912 ///
913 /// `Config::retention_for` decides which window a kind gets; `ingest_floor`
914 /// decides what is worth storing; `store::prune_old_entries` decides what is
915 /// deleted. The first two are asserted here against the third's rule, because
916 /// a drift between them is the resurrection cycle — a row the store keeps, the
917 /// sweep deletes, and the next poll re-inserts unread — and nothing else in the
918 /// suite would notice one side changing alone.
919 ///
920 /// The publication case is the one that matters: fed the RSS window a
921 /// publication stores nothing at all, because a real one's newest document is
922 /// months old.
923 #[test]
924 fn a_publications_floor_is_its_archive_ceiling_not_the_rss_window() {
925 let config = crate::config::Config::default();
926 let now = chrono::DateTime::parse_from_rfc3339("2026-09-29T12:00:00Z")
927 .unwrap()
928 .with_timezone(&chrono::Utc);
929
930 let (days, hard) = config.retention_for(crate::feed::FeedKind::Publication);
931 let floor = ingest_floor(days, hard, now).expect("a publication has an ingest floor");
932 let expected = crate::feed::fmt_time(
933 now - chrono::Duration::days(config.publication_retention_days.into()),
934 );
935 assert_eq!(
936 floor, expected,
937 "a publication's floor must be its archive ceiling ({} days), because \
938 that is the only sweep pass that can delete its rows",
939 config.publication_retention_days,
940 );
941
942 // And the number this rules OUT: the RSS window would drop every document
943 // of every real publication measured (newest 109 to 241 days old).
944 let rss_floor = ingest_floor(config.retention_days, config.retention_hard_days, now)
945 .expect("an RSS feed has an ingest floor");
946 assert!(
947 floor < rss_floor,
948 "the publication floor ({floor}) is no older than the RSS one \
949 ({rss_floor}), so a months-old document would still be dropped",
950 );
951 let a_real_publications_newest_document =
952 crate::feed::fmt_time(now - chrono::Duration::days(109));
953 assert!(
954 a_real_publications_newest_document.as_str() >= floor.as_str(),
955 "the newest document a real publication offered would be refused at \
956 ingest: {a_real_publications_newest_document} against a floor of {floor}",
957 );
958 assert!(
959 a_real_publications_newest_document.as_str() < rss_floor.as_str(),
960 "this assertion is only meaningful while the RSS window WOULD have \
961 dropped it, and it no longer does",
962 );
963 }
964
965 /// **The floor, all four configurations, without a database or the clock.**
966 ///
967 /// The end-to-end tests below observe the floor through what gets stored,
968 /// which cannot see the difference between "no floor" and "a floor the
969 /// lexicographic row comparison happens to sort past". This asserts the value
970 /// itself.
971 #[test]
972 fn the_ingest_floor_is_the_window_the_sweep_would_use() {
973 let now = chrono::DateTime::parse_from_rfc3339("2026-09-26T12:00:00Z")
974 .unwrap()
975 .with_timezone(&chrono::Utc);
976
977 assert_eq!(
978 ingest_floor(14, 180, now).as_deref(),
979 Some("2026-09-12T12:00:00Z"),
980 "with both set, the floor is the WINDOW — the thing that deletes first",
981 );
982 assert_eq!(
983 ingest_floor(180, 30, now).as_deref(),
984 Some("2026-03-30T12:00:00Z"),
985 "a ceiling INSIDE the window is one `prune_old_entries` ignores, so it \
986 must not lower the floor — `min` here discarded five months of archive \
987 that nothing would have deleted",
988 );
989 assert_eq!(
990 ingest_floor(0, 30, now).as_deref(),
991 Some("2026-08-27T12:00:00Z"),
992 "with no window the ceiling stands alone, and it still deletes",
993 );
994 assert_eq!(
995 ingest_floor(0, 0, now),
996 None,
997 "with no retention at all there is nothing to floor against",
998 );
999 // `u32::MAX` days is an operator saying "keep everything". Both the
1000 // duration and the subtraction fail there, and the answer is the ABSENCE
1001 // of a floor — not a sentinel instant whose rendering the row comparison
1002 // then has to sort correctly by accident.
1003 assert_eq!(
1004 ingest_floor(u32::MAX, u32::MAX, now),
1005 None,
1006 "an unrepresentable window produced a floor, so the comparison is \
1007 resting on how `fmt_time` renders an out-of-range year",
1008 );
1009 }
1010
1011 /// **A ceiling INSIDE the window is one the sweep ignores, so the floor must
1012 /// ignore it too.**
1013 ///
1014 /// `prune_old_entries` honours `retention_hard_days` only when it is strictly
1015 /// older than `retention_days` (or when there is no window at all): a ceiling
1016 /// inside the window is logged and dropped, because the hard delete spares
1017 /// nothing and would otherwise delete exactly the rows the soft delete exists
1018 /// to spare. So at `days = 180, hard = 30` nothing is deleted before 180 days.
1019 ///
1020 /// An earlier version took the MINIMUM of the two, floored ingest at 30 days,
1021 /// and silently discarded five months of a publisher's archive that nothing
1022 /// would ever have deleted. The older test above passes under both rules —
1023 /// `min(14, 180)` and "the window" are both 14 — which is why this case is
1024 /// the one that had to be written.
1025 #[tokio::test]
1026 async fn a_ceiling_inside_the_window_does_not_lower_the_floor() {
1027 let pool = pool().await;
1028 let read = read_of(
1029 vec![entry_dated("at://d/c/hundred", Some(&days_ago(100)))],
1030 true,
1031 );
1032 store_publication(&pool, PUB_URL, read, 0, 180, 30)
1033 .await
1034 .unwrap();
1035 let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
1036 .fetch_one(&pool)
1037 .await
1038 .unwrap();
1039 assert_eq!(
1040 n, 1,
1041 "an entry inside the 180-day window was dropped because of a 30-day \
1042 ceiling the sweep ignores — five months of archive discarded at ingest \
1043 that nothing would have deleted",
1044 );
1045 }
1046
1047 /// **A complete read of an empty publication is a healthy poll, not a
1048 /// failure.** The failure branch is keyed on `!complete && offered == 0`, and
1049 /// dropping either half of that makes a brand-new or fully-archived
1050 /// publication go into backoff that widens forever.
1051 #[tokio::test]
1052 async fn a_complete_read_of_nothing_is_not_a_failure() {
1053 let pool = pool().await;
1054 let outcome = store_publication(&pool, PUB_URL, read_of(vec![], true), 0, 14, 180)
1055 .await
1056 .unwrap();
1057 assert!(
1058 matches!(
1059 outcome,
1060 crate::feed::PollOutcome::Updated { new_entries: 0 }
1061 ),
1062 "a complete read of an empty publication was not a healthy poll: {outcome:?}"
1063 );
1064 }
1065
1066 /// **The other half of `a_failed_read_does_not_stamp_last_polled`.** That test
1067 /// alone is satisfied by never stamping at all, which would leave every
1068 /// publication permanently due and `/stats` permanently wrong.
1069 #[tokio::test]
1070 async fn a_successful_read_stamps_last_polled() {
1071 let pool = pool().await;
1072 let read = read_of(vec![entry_dated("at://d/c/1", Some(&days_ago(1)))], true);
1073 store_publication(&pool, PUB_URL, read, 0, 14, 180)
1074 .await
1075 .unwrap();
1076 let stamped: Option<String> =
1077 sqlx::query_scalar("SELECT last_polled FROM feeds WHERE url = ?1")
1078 .bind(PUB_URL)
1079 .fetch_one(&pool)
1080 .await
1081 .unwrap();
1082 assert!(
1083 stamped.is_some(),
1084 "a successful read left `last_polled` NULL, so the publication stays \
1085 due forever and `/stats` never shows it as polled",
1086 );
1087 }
1088
1089 /// **The saturating window must saturate in the KEEPING direction.**
1090 ///
1091 /// `an_absurd_retention_window_does_not_panic` only asserts that it returns.
1092 /// A saturation that produced `DateTime::MAX` instead — or a floor of "now" —
1093 /// would satisfy it while silently discarding the publisher's entire archive:
1094 /// `u32::MAX` days is an operator saying "keep everything".
1095 #[tokio::test]
1096 async fn an_absurd_retention_window_keeps_everything_rather_than_nothing() {
1097 let pool = pool().await;
1098 let read = read_of(
1099 vec![entry_dated("at://d/c/ancient", Some(&days_ago(10_000)))],
1100 true,
1101 );
1102 store_publication(&pool, PUB_URL, read, 0, u32::MAX, u32::MAX)
1103 .await
1104 .expect("a huge window is a wide floor, not a crash");
1105 let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
1106 .fetch_one(&pool)
1107 .await
1108 .unwrap();
1109 assert_eq!(
1110 n, 1,
1111 "a window of u32::MAX days dropped a 27-year-old entry, so the \
1112 saturation went the wrong way",
1113 );
1114 }
1115
1116 /// **`max_entries_per_feed` is passed through, and every test above passes
1117 /// `0`.** So a call that dropped the argument, or passed a constant, would
1118 /// have gone unnoticed — and this is the cap that decides which of a
1119 /// publisher's articles a reader keeps.
1120 #[tokio::test]
1121 async fn the_per_feed_cap_is_the_one_the_caller_passed() {
1122 let pool = pool().await;
1123 let read = read_of(
1124 (0..5)
1125 .map(|i| entry_dated(&format!("at://d/c/{i}"), Some(&days_ago(i + 1))))
1126 .collect(),
1127 true,
1128 );
1129 store_publication(&pool, PUB_URL, read, 2, 0, 0)
1130 .await
1131 .unwrap();
1132 let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
1133 .fetch_one(&pool)
1134 .await
1135 .unwrap();
1136 assert_eq!(
1137 n, 2,
1138 "five entries under a cap of two left {n} rows, so the caller's cap is \
1139 not the one being applied",
1140 );
1141 }
1142
1143 #[tokio::test]
1144 async fn no_retention_at_all_means_no_ingest_floor() {
1145 let pool = pool().await;
1146 let read = read_of(
1147 vec![entry_dated("at://d/c/ancient", Some(&days_ago(900)))],
1148 true,
1149 );
1150 store_publication(&pool, PUB_URL, read, 0, 0, 0)
1151 .await
1152 .unwrap();
1153 let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
1154 .fetch_one(&pool)
1155 .await
1156 .unwrap();
1157 assert_eq!(
1158 n, 1,
1159 "with nothing deleting it, an old entry is worth keeping"
1160 );
1161 }
1162
1163 /// A retention window near `u32::MAX` must not panic the poller.
1164 #[tokio::test]
1165 async fn an_absurd_retention_window_does_not_panic() {
1166 let pool = pool().await;
1167 let read = read_of(vec![entry_dated("at://d/c/x", Some(&days_ago(1)))], true);
1168 store_publication(&pool, PUB_URL, read, 0, u32::MAX, u32::MAX)
1169 .await
1170 .expect("a huge window is a wide floor, not a crash");
1171 }
1172
1173 /// The feed row learns the publication's name and homepage — the only reason
1174 /// beyond the timestamp that the upsert is there at all. `upsert_feed`
1175 /// COALESCEs both, so dropping either is silent.
1176 #[tokio::test]
1177 async fn the_feed_row_learns_the_publications_name_and_site() {
1178 let pool = pool().await;
1179 let read = read_of(vec![entry_dated("at://d/c/1", Some(&days_ago(1)))], true);
1180 store_publication(&pool, PUB_URL, read, 0, 14, 180)
1181 .await
1182 .unwrap();
1183 let (title, site): (Option<String>, Option<String>) =
1184 sqlx::query_as("SELECT title, site_url FROM feeds WHERE url = ?1")
1185 .bind(PUB_URL)
1186 .fetch_one(&pool)
1187 .await
1188 .unwrap();
1189 assert_eq!(
1190 title.as_deref(),
1191 Some("Scan's Lab"),
1192 "the name never reached the row"
1193 );
1194 assert_eq!(
1195 site.as_deref(),
1196 Some("https://example.com/blog/"),
1197 "the homepage never reached the row"
1198 );
1199 }
1200
1201 /// **The undated entry is kept, deliberately.** It is dated by `fetched_at`,
1202 /// which holds still, and the alternative is discarding an article the reader
1203 /// can never see. It resurrects once per retention window; that is accepted.
1204 #[tokio::test]
1205 async fn an_undated_entry_is_stored_rather_than_dropped() {
1206 let pool = pool().await;
1207 let read = read_of(vec![entry_dated("at://d/c/undated", None)], true);
1208 store_publication(&pool, PUB_URL, read, 0, 14, 180)
1209 .await
1210 .unwrap();
1211 let n: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM entries")
1212 .fetch_one(&pool)
1213 .await
1214 .unwrap();
1215 assert_eq!(n, 1, "an entry with no date was discarded");
1216 }
1217
1218 /// A record key from a real atproto repo, and the instant it decodes to.
1219 ///
1220 /// **Fixed, not minted.** A test that mints a TID expects "about now", and
1221 /// "about now" is satisfied by any source of the current time — including a
1222 /// fallback that has had the rkey ripped out of it and returns `Utc::now()`
1223 /// instead. Both tests named for the rkey fallback used to survive exactly
1224 /// that mutation. A historical key with a stated answer cannot.
1225 const PAST_TID: &str = "3jzfcijpj2z2a";
1226 const PAST_TID_WRITTEN_AT: &str = "2023-06-30T15:03:01Z";
1227
1228 fn rec(collection: &str, rkey: &str, value: serde_json::Value) -> RecordEntry {
1229 RecordEntry {
1230 uri: format!("at://{DID}/{collection}/{rkey}"),
1231 cid: None,
1232 value,
1233 }
1234 }
1235
1236 fn publication(rkey: &str, url: &str) -> RecordEntry {
1237 rec(
1238 nsid::STANDARD_PUBLICATION,
1239 rkey,
1240 json!({ "name": "Scan's Lab", "url": url }),
1241 )
1242 }
1243
1244 fn document(rkey: &str, site: &str, title: &str, path: &str) -> RecordEntry {
1245 rec(
1246 nsid::STANDARD_DOCUMENT,
1247 rkey,
1248 json!({
1249 "title": title,
1250 "publishedAt": "2026-07-11T00:00:00Z",
1251 "path": path,
1252 "site": site,
1253 "textContent": "body",
1254 }),
1255 )
1256 }
1257
1258 fn canonical(rkey: &str) -> String {
1259 format!("at://{DID}/{}/{rkey}", nsid::STANDARD_PUBLICATION)
1260 }
1261
1262 /// **The `site` filter keys on the URI the PDS minted, not the string the
1263 /// reader subscribed with.** Documents reference their publication by the
1264 /// canonical URI, and every measured document does. An earlier draft
1265 /// compared against the subscribed string; storage was then meant to admit
1266 /// handle-form URIs, and a handle-form subscription found nothing forever
1267 /// while the module declared the feed healthy. #164 made storage DID-only,
1268 /// so the two strings agree today — the canonical one is still the right
1269 /// key, and this pins it.
1270 #[test]
1271 fn the_site_filter_uses_the_uri_the_pds_minted() {
1272 let records = vec![publication("p", "https://scanash.com")];
1273 let (site, pubn) = publication_from_records("p", &records).expect("publication not found");
1274 assert_eq!(site, canonical("p"), "did not take the PDS's canonical URI");
1275
1276 let docs = vec![document("d1", &canonical("p"), "Hello", "/hello")];
1277 let entries = entries_from_records(&site, &pubn, &docs);
1278 assert_eq!(
1279 entries.len(),
1280 1,
1281 "a canonical-site document was not matched"
1282 );
1283 }
1284
1285 /// The mapping into the store's row is total and loses nothing the poller
1286 /// would need — so wiring the reader has nothing to invent.
1287 #[test]
1288 fn an_entry_maps_onto_the_stores_row() {
1289 let records = vec![publication("p", "https://example.com")];
1290 let (site, pubn) = publication_from_records("p", &records).unwrap();
1291 let docs = vec![document("rk1", &site, "Hello", "/hello")];
1292 let row: crate::store::NewEntry = entries_from_records(&site, &pubn, &docs)
1293 .pop()
1294 .unwrap()
1295 .into();
1296 assert_eq!(
1297 row.guid,
1298 format!("at://{DID}/{}/rk1", nsid::STANDARD_DOCUMENT)
1299 );
1300 assert_eq!(row.url.as_deref(), Some("https://example.com/hello"));
1301 assert_eq!(row.title.as_deref(), Some("Hello"));
1302 assert_eq!(row.published.as_deref(), Some("2026-07-11T00:00:00Z"));
1303 assert_eq!(row.content_html.as_deref(), Some("body"));
1304 assert_eq!(row.author, None);
1305 assert_eq!(row.fetched_at, None);
1306 }
1307
1308 /// A repo can hold several publications — measured, some do — and
1309 /// `listRecords` cannot filter server-side.
1310 #[test]
1311 fn documents_are_filtered_by_their_site_field() {
1312 let records = vec![publication("mine", "https://example.com")];
1313 let (site, pubn) = publication_from_records("mine", &records).unwrap();
1314 let docs = vec![
1315 document("a", &site, "Mine", "/a"),
1316 document("b", &canonical("theirs"), "Theirs", "/b"),
1317 document("c", &site, "Mine again", "/c"),
1318 ];
1319 let titles: Vec<String> = entries_from_records(&site, &pubn, &docs)
1320 .into_iter()
1321 .map(|e| e.title)
1322 .collect();
1323 assert_eq!(titles, ["Mine", "Mine again"]);
1324 }
1325
1326 /// **The publication URL is a stranger's string and is vetted as one.**
1327 ///
1328 /// It becomes the base of every `Entry.url`, which is an href. This repo has
1329 /// `safe_link.rs` as a type with a module-private field precisely because
1330 /// the procedural version of this guarantee failed; #143 and #138 are this
1331 /// same bug class on `siteUrl`.
1332 #[test]
1333 fn a_publication_with_a_hostile_url_is_refused() {
1334 for hostile in [
1335 "javascript:alert(1)",
1336 "data:text/html,<script>",
1337 "file:///etc/passwd",
1338 "",
1339 ] {
1340 let records = vec![publication("p", hostile)];
1341 assert!(
1342 publication_from_records("p", &records).is_none(),
1343 "accepted a publication whose url is {hostile:?}",
1344 );
1345 }
1346 }
1347
1348 /// **Entry URLs are joined, not concatenated.**
1349 ///
1350 /// String concatenation produced `https://x.com/https://evil.com/a` for an
1351 /// absolute `path`, and put the path inside the query string for a base
1352 /// carrying one.
1353 #[test]
1354 fn entry_urls_are_joined_against_the_publication_base() {
1355 let records = vec![publication("p", "https://example.com/blog")];
1356 let (site, pubn) = publication_from_records("p", &records).unwrap();
1357 let docs = vec![
1358 document("a", &site, "Relative", "/a"),
1359 document("b", &site, "Absolute-looking", "https://evil.example/x"),
1360 ];
1361 let urls: Vec<Option<String>> = entries_from_records(&site, &pubn, &docs)
1362 .into_iter()
1363 .map(|e| e.url)
1364 .collect();
1365 assert_eq!(urls[0].as_deref(), Some("https://example.com/a"));
1366 // Exactly the publication's base, not merely "not evil": a mutation
1367 // that returned the raw path, or an empty string, passed the weaker
1368 // negative assertion this used to be.
1369 // Off-origin is dropped, not rewritten to the base. This assertion has
1370 // moved twice: it began as "not evil" (a mutant returning the raw path
1371 // passed it), was tightened to the base fallback, and is now `None` —
1372 // the base gave every affected entry the same homepage href.
1373 assert_eq!(
1374 urls[1], None,
1375 "a document path that escapes its publication's origin must yield no URL"
1376 );
1377 }
1378
1379 /// 8% of measured documents (37 of 449) carry neither summary field.
1380 #[test]
1381 fn a_document_with_neither_summary_field_still_yields_an_entry() {
1382 let records = vec![publication("p", "https://example.com/")];
1383 let (site, pubn) = publication_from_records("p", &records).unwrap();
1384 let bare = rec(
1385 nsid::STANDARD_DOCUMENT,
1386 "bare",
1387 json!({
1388 "title": "Bare",
1389 "publishedAt": "2026-07-11T00:00:00Z",
1390 "path": "/bare",
1391 "site": site,
1392 }),
1393 );
1394 let entries = entries_from_records(&site, &pubn, &[bare]);
1395 assert_eq!(entries.len(), 1);
1396 assert_eq!(entries[0].summary, None);
1397 assert_eq!(entries[0].url.as_deref(), Some("https://example.com/bare"));
1398 }
1399
1400 /// `description` is the authored summary; `textContent` is the whole body.
1401 /// An EMPTY description must not shadow a real one.
1402 #[test]
1403 fn an_empty_description_does_not_shadow_the_body() {
1404 let records = vec![publication("p", "https://example.com")];
1405 let (site, pubn) = publication_from_records("p", &records).unwrap();
1406 let doc = rec(
1407 nsid::STANDARD_DOCUMENT,
1408 "d",
1409 json!({
1410 "title": "T",
1411 "publishedAt": "2026-07-11T00:00:00Z",
1412 "path": "/d",
1413 "site": site,
1414 "description": " ",
1415 "textContent": "the real body",
1416 }),
1417 );
1418 let entries = entries_from_records(&site, &pubn, &[doc]);
1419 assert_eq!(entries[0].summary.as_deref(), Some("the real body"));
1420 }
1421
1422 /// **`publishedAt` is parsed, not passed through.** The store's `published`
1423 /// column is the RFC3339 shape `feed::fmt_time` writes, and the reading
1424 /// order sorts on it as a string. A publisher's string went in verbatim —
1425 /// a garbage value would have sorted arbitrarily among real ones, and a
1426 /// valid-but-differently-spelled one (`+00:00`, fractional seconds) would
1427 /// not have matched the RSS path's spelling for the same instant.
1428 #[test]
1429 fn published_at_is_normalised_or_dropped() {
1430 let records = vec![publication("p", "https://example.com")];
1431 let (site, pubn) = publication_from_records("p", &records).unwrap();
1432 let with = |rkey: &str, published_at: serde_json::Value| {
1433 rec(
1434 nsid::STANDARD_DOCUMENT,
1435 rkey,
1436 json!({ "title": "T", "publishedAt": published_at, "path": "/x", "site": site }),
1437 )
1438 };
1439 // Single-character rkeys on purpose: they are not TIDs, so the
1440 // rkey-derived fallback below does not apply and "unparseable" really
1441 // does mean undated here.
1442 let docs = vec![
1443 with("a", json!("2026-07-11T09:30:00.123+02:00")),
1444 with("b", json!("yesterday-ish")),
1445 with("c", json!("2026-07-11T00:00:00Z")),
1446 ];
1447 let published: Vec<Option<String>> = entries_from_records(&site, &pubn, &docs)
1448 .into_iter()
1449 .map(|e| e.published)
1450 .collect();
1451 assert_eq!(
1452 published,
1453 vec![
1454 Some("2026-07-11T07:30:00Z".to_string()),
1455 None,
1456 Some("2026-07-11T00:00:00Z".to_string()),
1457 ],
1458 "publishedAt was not normalised to the store's spelling"
1459 );
1460 }
1461
1462 /// A document with no usable `publishedAt` is dated from its rkey.
1463 ///
1464 /// **An undated entry is not merely untidy, it is immortal-and-mortal at
1465 /// once.** The store's retention sweep and its per-feed cap both order on
1466 /// `COALESCE(published, fetched_at)`, and an entry inserted with no
1467 /// `published` gets `fetched_at` stamped at insertion. So the sweep deletes
1468 /// it once it is `retention_days` old, the next poll re-inserts it with a
1469 /// fresh `fetched_at` and a new `entries.id`, its read state is gone with
1470 /// the cascade, and it arrives unread — again, on the same cycle, forever.
1471 /// The same reset also sorts it newest in the per-feed cap, where it evicts
1472 /// entries that really are newer.
1473 #[test]
1474 fn an_undated_document_is_dated_from_its_tid_rkey() {
1475 let records = vec![publication("p", "https://example.com")];
1476 let (site, pubn) = publication_from_records("p", &records).unwrap();
1477 let docs = vec![rec(
1478 nsid::STANDARD_DOCUMENT,
1479 PAST_TID,
1480 json!({ "title": "T", "path": "/x", "site": site }),
1481 )];
1482 assert_eq!(
1483 entries_from_records(&site, &pubn, &docs)
1484 .into_iter()
1485 .next()
1486 .expect("the document is an entry")
1487 .published,
1488 Some(PAST_TID_WRITTEN_AT.to_string()),
1489 "the date must come from the record key, and be spelled the way the store spells dates"
1490 );
1491 }
1492
1493 #[test]
1494 fn an_unparseable_published_at_falls_back_to_the_tid_rkey() {
1495 let records = vec![publication("p", "https://example.com")];
1496 let (site, pubn) = publication_from_records("p", &records).unwrap();
1497 let docs = vec![rec(
1498 nsid::STANDARD_DOCUMENT,
1499 PAST_TID,
1500 json!({ "title": "T", "publishedAt": "yesterday-ish", "path": "/x", "site": site }),
1501 )];
1502 assert_eq!(
1503 entries_from_records(&site, &pubn, &docs)
1504 .into_iter()
1505 .next()
1506 .expect("the document is an entry")
1507 .published,
1508 Some(PAST_TID_WRITTEN_AT.to_string()),
1509 "a date the parser cannot read is no date at all, so the rkey must stand in"
1510 );
1511 }
1512
1513 #[test]
1514 fn a_stated_date_outranks_the_rkey() {
1515 let records = vec![publication("p", "https://example.com")];
1516 let (site, pubn) = publication_from_records("p", &records).unwrap();
1517 // Both candidate dates are historical, so nothing here depends on
1518 // what the machine's clock reads.
1519 let docs = vec![rec(
1520 nsid::STANDARD_DOCUMENT,
1521 PAST_TID,
1522 json!({ "title": "T", "publishedAt": "2020-01-02T00:00:00Z", "path": "/x", "site": site }),
1523 )];
1524 assert_eq!(
1525 entries_from_records(&site, &pubn, &docs)
1526 .into_iter()
1527 .next()
1528 .expect("the document is an entry")
1529 .published,
1530 Some("2020-01-02T00:00:00Z".to_string()),
1531 "the rkey records when the file was written, which is not when the post was published"
1532 );
1533 }
1534
1535 /// **A stated date in the future is discarded, not clamped.**
1536 ///
1537 /// Clamping it to "now" looks safe and is not. The store refreshes
1538 /// `published` on every poll, so the row would be re-dated to the current
1539 /// hour forever: never older than the retention cutoff, never outranked in
1540 /// the per-feed cap, permanently first in the reading list. The date must
1541 /// not depend on when the mapping ran, which is why both of these name the
1542 /// exact value they expect rather than comparing against the clock.
1543 #[test]
1544 fn a_future_dated_document_falls_back_to_its_rkey() {
1545 let records = vec![publication("p", "https://example.com")];
1546 let (site, pubn) = publication_from_records("p", &records).unwrap();
1547 let docs = vec![rec(
1548 nsid::STANDARD_DOCUMENT,
1549 PAST_TID,
1550 json!({ "title": "T", "publishedAt": "2999-01-01T00:00:00Z", "path": "/x", "site": site }),
1551 )];
1552 assert_eq!(
1553 entries_from_records(&site, &pubn, &docs)
1554 .into_iter()
1555 .next()
1556 .expect("the document is an entry")
1557 .published,
1558 Some(PAST_TID_WRITTEN_AT.to_string()),
1559 "the date must be the record's write time, not the hour the poll happened to run"
1560 );
1561 }
1562
1563 #[test]
1564 fn a_future_dated_document_without_a_tid_rkey_is_undated() {
1565 let records = vec![publication("p", "https://example.com")];
1566 let (site, pubn) = publication_from_records("p", &records).unwrap();
1567 let docs = vec![rec(
1568 nsid::STANDARD_DOCUMENT,
1569 "self",
1570 json!({ "title": "T", "publishedAt": "2999-01-01T00:00:00Z", "path": "/x", "site": site }),
1571 )];
1572 assert_eq!(
1573 entries_from_records(&site, &pubn, &docs)
1574 .into_iter()
1575 .next()
1576 .expect("the document is an entry")
1577 .published,
1578 None,
1579 "with nothing credible to date it by, the row falls to fetched_at, which holds still"
1580 );
1581 }
1582
1583 /// **A little ahead of our clock is skew, not a lie.**
1584 ///
1585 /// The rkey here is deliberately not a TID, so nothing masks a wrongly
1586 /// discarded date: if the stated one is thrown away the entry is undated,
1587 /// and an undated entry sorts to the bottom of a list ordered on a bare
1588 /// `published DESC`. A publisher a few seconds fast would have had their
1589 /// newest post buried.
1590 #[test]
1591 fn a_stated_date_a_little_ahead_of_our_clock_is_still_believed() {
1592 let records = vec![publication("p", "https://example.com")];
1593 let (site, pubn) = publication_from_records("p", &records).unwrap();
1594 let slightly_ahead =
1595 crate::feed::fmt_time(chrono::Utc::now() + chrono::Duration::seconds(10));
1596 let docs = vec![rec(
1597 nsid::STANDARD_DOCUMENT,
1598 "self",
1599 json!({ "title": "T", "publishedAt": slightly_ahead, "path": "/x", "site": site }),
1600 )];
1601 assert_eq!(
1602 entries_from_records(&site, &pubn, &docs)
1603 .into_iter()
1604 .next()
1605 .expect("the document is an entry")
1606 .published,
1607 Some(slightly_ahead),
1608 "a few seconds of clock skew must not cost the entry its date"
1609 );
1610 }
1611
1612 #[test]
1613 fn a_document_with_neither_a_date_nor_a_tid_rkey_stays_undated() {
1614 let records = vec![publication("p", "https://example.com")];
1615 let (site, pubn) = publication_from_records("p", &records).unwrap();
1616 // Two shapes, rejected by two different checks. "my-first-post" is 13
1617 // characters but carries a `-`, so it never reaches the alphabet's
1618 // arithmetic at all. "abcdefghijklm" is 13 valid s32 characters and
1619 // decodes perfectly well — to the year 2192 — which is the case the
1620 // bound in `tid_timestamp` exists for, and the one a publisher naming
1621 // files by slug actually produces.
1622 let docs = vec![
1623 rec(
1624 nsid::STANDARD_DOCUMENT,
1625 "my-first-post",
1626 json!({ "title": "T", "path": "/x", "site": site }),
1627 ),
1628 rec(
1629 nsid::STANDARD_DOCUMENT,
1630 "abcdefghijklm",
1631 json!({ "title": "T", "path": "/y", "site": site }),
1632 ),
1633 ];
1634 assert_eq!(
1635 entries_from_records(&site, &pubn, &docs)
1636 .into_iter()
1637 .map(|e| e.published)
1638 .collect::<Vec<_>>(),
1639 vec![None, None],
1640 "an invented date is worse than no date; the store decides what to do with undated rows"
1641 );
1642 }
1643
1644 /// **Summaries are plain text and are escaped, not sanitised.**
1645 ///
1646 /// The lexicon defines `textContent` and `description` as plain text, and
1647 /// the store's `content_html` is rendered as HTML, so the text must be
1648 /// escaped on the way in. The first version of this ran `ammonia::clean`
1649 /// over them — the RSS body function — which parses its input as markup
1650 /// and deletes everything after a bare `<`. 321 of 449 measured documents
1651 /// use `textContent` as their summary; any post mentioning `Vec<T>` lost
1652 /// the rest of its summary, silently.
1653 #[test]
1654 fn summaries_are_escaped_as_plain_text_not_sanitised_as_markup() {
1655 let records = vec![publication("p", "https://example.com")];
1656 let (site, pubn) = publication_from_records("p", &records).unwrap();
1657 let doc = |rkey: &str, body: &str| {
1658 rec(
1659 nsid::STANDARD_DOCUMENT,
1660 rkey,
1661 json!({
1662 "title": "T",
1663 "publishedAt": "2026-07-11T00:00:00Z",
1664 "path": "/d",
1665 "site": site,
1666 "textContent": body,
1667 }),
1668 )
1669 };
1670 let summaries: Vec<String> = entries_from_records(
1671 &site,
1672 &pubn,
1673 &[
1674 doc("a", "Vec<String> is a type"),
1675 doc("b", "<script>alert(1)</script>"),
1676 ],
1677 )
1678 .into_iter()
1679 .filter_map(|e| e.summary)
1680 .collect();
1681 assert_eq!(
1682 summaries[0], "Vec<String> is a type",
1683 "prose was eaten by an HTML parser"
1684 );
1685 assert!(
1686 !summaries[1].contains("<script"),
1687 "escaping failed: {}",
1688 summaries[1]
1689 );
1690 }
1691
1692 /// **`Entry.url` is a vetted href or nothing.** The scheme guarantee lived
1693 /// only inside `publication_from_records`; `entries_from_records` and
1694 /// `Publication` are both `pub`, so any other constructor — step 3 building
1695 /// one from the stored `feeds` row, say — gave an unparseable base, and the
1696 /// no-base branch then emitted the document's `path` verbatim. A
1697 /// `javascript:` path became the entry link. This is the class `safe_link`
1698 /// exists for.
1699 #[test]
1700 fn an_entry_url_is_never_an_unvetted_path() {
1701 let pubn = Publication {
1702 name: None,
1703 // What a caller that did not go through `publication_from_records`
1704 // can hand this function.
1705 url: "not a url".to_string(),
1706 };
1707 let site = canonical("p");
1708 let docs = vec![
1709 document("a", &site, "Hostile", "javascript:alert(1)"),
1710 document("b", &site, "Fine", "https://example.com/ok"),
1711 ];
1712 let entries = entries_from_records(&site, &pubn, &docs);
1713 assert_eq!(
1714 entries[0].url, None,
1715 "an unvetted path became an entry link"
1716 );
1717 // Contract change: with no parseable base there is no origin to check,
1718 // so a well-formed absolute URL is refused too. `safe_link` alone vets
1719 // the SCHEME; it would have published a publisher-controlled host under
1720 // this publication's name. See `no_parseable_base_means_no_url_not_any_url`.
1721 assert_eq!(
1722 entries[1].url, None,
1723 "an off-origin absolute URL was published under the publication's name"
1724 );
1725 }
1726
1727 /// A document with no `publishedAt` is an entry with no date — the same
1728 /// answer a garbage one gets. The field the module is willing to discard
1729 /// must not be the one whose absence is fatal.
1730 #[test]
1731 fn a_document_without_published_at_is_still_an_entry() {
1732 let records = vec![publication("p", "https://example.com")];
1733 let (site, pubn) = publication_from_records("p", &records).unwrap();
1734 let doc = rec(
1735 nsid::STANDARD_DOCUMENT,
1736 "d",
1737 json!({ "title": "T", "path": "/d", "site": site }),
1738 );
1739 let entries = entries_from_records(&site, &pubn, &[doc]);
1740 assert_eq!(
1741 entries.len(),
1742 1,
1743 "a missing publishedAt dropped the document"
1744 );
1745 assert_eq!(entries[0].published, None);
1746 }
1747
1748 /// **`fetch` refuses a URI naming another collection, before the network.**
1749 /// It lists publications and matches on rkey alone, so without this an
1750 /// `app.bsky.feed.post` URI would be "read as a publication" whenever a
1751 /// publication in that repo shares the rkey. Storage enforces the
1752 /// collection today, but this function is `pub`.
1753 #[tokio::test]
1754 async fn fetch_refuses_a_uri_for_another_collection() {
1755 let uri = AtUri::parse(&format!("at://{DID}/app.bsky.feed.post/3lab")).unwrap();
1756 let err = fetch(&reqwest::Client::new(), "https://plc.example", &uri)
1757 .await
1758 .expect_err("read a feed post as a publication");
1759 assert!(
1760 format!("{err:#}").contains(nsid::STANDARD_PUBLICATION),
1761 "failed for the wrong reason: {err:#}"
1762 );
1763 }
1764
1765 /// **A publication on a subpath keeps it.** `Url::join` is RFC-3986, so a
1766 /// relative `posts/a` against `https://example.com/blog` resolves to
1767 /// `/posts/a` — dropping the subpath every permalink needs, while still
1768 /// passing the origin check. The base is normalised to a directory.
1769 #[test]
1770 fn a_subpath_publication_keeps_its_base_path() {
1771 let records = vec![publication("p", "https://example.com/blog")];
1772 let (site, pubn) = publication_from_records("p", &records).unwrap();
1773 let docs = vec![document("a", &site, "Relative", "posts/a")];
1774 let urls: Vec<Option<String>> = entries_from_records(&site, &pubn, &docs)
1775 .into_iter()
1776 .map(|e| e.url)
1777 .collect();
1778 assert_eq!(urls[0].as_deref(), Some("https://example.com/blog/posts/a"));
1779 }
1780
1781 /// With no parseable base there is no origin to check, so there is no URL
1782 /// — `safe_link` alone vets the scheme and would pass any absolute URL a
1783 /// publisher chose, under this publication's name.
1784 #[test]
1785 fn no_parseable_base_means_no_url_not_any_url() {
1786 let pubn = Publication {
1787 name: None,
1788 url: "not a url".to_string(),
1789 };
1790 let site = canonical("p");
1791 let docs = vec![document("a", &site, "Absolute", "https://evil.example/x")];
1792 let entries = entries_from_records(&site, &pubn, &docs);
1793 assert_eq!(
1794 entries[0].url, None,
1795 "an off-origin absolute URL was published"
1796 );
1797 }
1798
1799 /// **An off-origin path is dropped, not rewritten to the homepage.**
1800 /// Returning the base gave every affected entry the SAME href pointing at
1801 /// the site root — realistic whenever a publication's `url` is the apex
1802 /// and its documents sit on `www.` or a CDN domain. `None` is the honest
1803 /// answer, and the template already has a no-URL branch.
1804 #[test]
1805 fn an_off_origin_path_yields_no_url_rather_than_the_homepage() {
1806 let records = vec![publication("p", "https://example.com/blog")];
1807 let (site, pubn) = publication_from_records("p", &records).unwrap();
1808 let docs = vec![
1809 document("a", &site, "Elsewhere", "https://www.example.com/post"),
1810 document("b", &site, "Home", "/ok"),
1811 ];
1812 let urls: Vec<Option<String>> = entries_from_records(&site, &pubn, &docs)
1813 .into_iter()
1814 .map(|e| e.url)
1815 .collect();
1816 assert_eq!(
1817 urls[0], None,
1818 "an off-origin path was rewritten to the base"
1819 );
1820 assert_eq!(urls[1].as_deref(), Some("https://example.com/ok"));
1821 }
1822
1823 /// **A blank `path` is no URL, not the homepage** — the same answer the
1824 /// off-origin branch now gives, and for the same reason: several
1825 /// documents with an empty `path` otherwise became several entries all
1826 /// linking to the site root.
1827 #[test]
1828 fn a_blank_path_yields_no_url() {
1829 let records = vec![publication("p", "https://example.com/blog")];
1830 let (site, pubn) = publication_from_records("p", &records).unwrap();
1831 let docs = vec![
1832 document("a", &site, "Blank", ""),
1833 document("b", &site, "Spaces", " "),
1834 document("c", &site, "Real", "/real"),
1835 ];
1836 let urls: Vec<Option<String>> = entries_from_records(&site, &pubn, &docs)
1837 .into_iter()
1838 .map(|e| e.url)
1839 .collect();
1840 assert_eq!(urls[0], None, "a blank path became the homepage");
1841 assert_eq!(urls[1], None, "a whitespace path became the homepage");
1842 assert_eq!(urls[2].as_deref(), Some("https://example.com/real"));
1843 }
1844
1845 /// A document without `path` keeps its title, date and summary — the
1846 /// policy `publishedAt` and `Entry.url` already follow.
1847 #[test]
1848 fn a_document_without_a_path_is_still_an_entry() {
1849 let records = vec![publication("p", "https://example.com")];
1850 let (site, pubn) = publication_from_records("p", &records).unwrap();
1851 let doc = rec(
1852 nsid::STANDARD_DOCUMENT,
1853 "d",
1854 json!({ "title": "T", "publishedAt": "2026-07-11T00:00:00Z", "site": site }),
1855 );
1856 let entries = entries_from_records(&site, &pubn, &[doc]);
1857 assert_eq!(
1858 entries.len(),
1859 1,
1860 "a missing path dropped the whole document"
1861 );
1862 assert_eq!(entries[0].title, "T");
1863 assert_eq!(entries[0].url, None);
1864 }
1865
1866 /// **A sibling is not an orphan.** A repo with an empty publication A and
1867 /// a busy publication B is the exact shape the `site` filter exists for;
1868 /// treating B's documents as a signal would warn on every poll of A
1869 /// forever. The signal is a document referencing a publication this repo
1870 /// does not have — a spelling nothing can ever match.
1871 #[test]
1872 fn documents_are_classified_keep_sibling_or_orphan() {
1873 let pubs = [
1874 publication("a", "https://example.com"),
1875 publication("b", "https://b.example"),
1876 ];
1877 let known: std::collections::HashSet<&str> = pubs.iter().map(|p| p.uri.as_str()).collect();
1878 let mine = canonical("a");
1879 let fate = |d: &crate::atproto::RecordEntry| classify_document(d, &mine, &known);
1880
1881 assert_eq!(
1882 fate(&document("d1", &mine, "Mine", "/1")),
1883 DocumentFate::Keep
1884 );
1885 assert_eq!(
1886 fate(&document("d2", &canonical("b"), "B's", "/2")),
1887 DocumentFate::Sibling,
1888 "a sibling publication's document is not an orphan"
1889 );
1890 assert_eq!(
1891 fate(&document(
1892 "d3",
1893 "at://did:plc:other/site.standard.publication/x",
1894 "?",
1895 "/3"
1896 )),
1897 DocumentFate::Orphan
1898 );
1899 assert_eq!(
1900 fate(&rec(
1901 nsid::STANDARD_DOCUMENT,
1902 "d4",
1903 json!({"title": "no rest"})
1904 )),
1905 DocumentFate::Malformed
1906 );
1907 }
1908
1909 /// `AtUri` uses the crate's one spelling of the prefix.
1910 #[test]
1911 fn at_uri_parsing_uses_the_shared_prefix() {
1912 let uri = format!(
1913 "{}{DID}/{}/abc",
1914 crate::atproto::AT_URI_PREFIX,
1915 nsid::STANDARD_PUBLICATION
1916 );
1917 assert!(AtUri::parse(&uri).is_some());
1918 }
1919
1920 /// The guid is the record's own URI. `path` is mutable; dedup is
1921 /// `UNIQUE (feed_id, guid)`.
1922 #[test]
1923 fn the_guid_is_the_record_uri_not_the_path() {
1924 let records = vec![publication("p", "https://example.com")];
1925 let (site, pubn) = publication_from_records("p", &records).unwrap();
1926 let docs = vec![document("rk1", &site, "T", "/moved")];
1927 let entries = entries_from_records(&site, &pubn, &docs);
1928 assert_eq!(
1929 entries[0].guid,
1930 format!("at://{DID}/{}/rk1", nsid::STANDARD_DOCUMENT)
1931 );
1932 }
1933
1934 /// One unreadable record must not cost a publisher the whole feed.
1935 #[test]
1936 fn a_malformed_document_is_skipped_rather_than_fatal() {
1937 let records = vec![publication("p", "https://example.com")];
1938 let (site, pubn) = publication_from_records("p", &records).unwrap();
1939 let docs = vec![
1940 rec(
1941 nsid::STANDARD_DOCUMENT,
1942 "bad",
1943 json!({ "title": "no rest" }),
1944 ),
1945 document("ok", &site, "Good", "/good"),
1946 ];
1947 let entries = entries_from_records(&site, &pubn, &docs);
1948 assert_eq!(entries.len(), 1);
1949 assert_eq!(entries[0].title, "Good");
1950 }
1951
1952 /// An rkey that is not in the repo is absent, not an error.
1953 #[test]
1954 fn a_missing_publication_is_none() {
1955 let records = vec![publication("other", "https://example.com")];
1956 assert!(publication_from_records("p", &records).is_none());
1957 }
1958
1959 #[test]
1960 fn at_uris_parse_in_both_forms_and_reject_malformed_ones() {
1961 let did = AtUri::parse(&format!("at://{DID}/site.standard.publication/abc")).unwrap();
1962 assert_eq!(did.authority, DID);
1963 assert_eq!(did.rkey, "abc");
1964 assert_eq!(
1965 did.to_string(),
1966 format!("at://{DID}/site.standard.publication/abc")
1967 );
1968 assert!(AtUri::parse("at://alice.example.com/site.standard.publication/abc").is_some());
1969 for bad in [
1970 "at://",
1971 "at://only-authority",
1972 "at://authority/collection",
1973 "at://authority/collection/",
1974 "at:///collection/rkey",
1975 "at://authority/collection/rkey/extra",
1976 "https://example.com/feed.xml",
1977 "at:authority/collection/rkey",
1978 ] {
1979 assert!(AtUri::parse(bad).is_none(), "parsed {bad:?}");
1980 }
1981 }
1982}