Skip to main content

feather_reader/
feed.rs

1//! Feed fetch → parse → sanitize → store pipeline.
2//!
3//! This is the module that turns a feed URL into rows in the [`store`]. It is
4//! deliberately conservative on three axes, because a feed reader ingests
5//! **hostile, arbitrary web input**:
6//!
7//! 1. **Politeness** — fetches use a **conditional GET** (`If-None-Match` /
8//!    `If-Modified-Since` from the stored `ETag` / `Last-Modified`), an
9//!    identifiable [`crate::USER_AGENT`], a request timeout, and a simple
10//!    exponential backoff hint on error. A `304 Not Modified` is a no-op:
11//!    the feed is untouched apart from bumping its next-poll time.
12//! 2. **Safety** — every entry's HTML is run through `ammonia` before it is
13//!    ever stored (and therefore before it is ever rendered). Scripts, event
14//!    handlers, `javascript:` URLs, tracking pixels' dangerous attributes, and
15//!    other XSS vectors are stripped. Feeds carrying `<script>` is not
16//!    hypothetical; treat all feed HTML as untrusted.
17//! 3. **Robustness** — a malformed feed is **logged and skipped**, never a
18//!    panic. One bad publisher must not take down the poller. All non-test
19//!    paths use `Result`/`anyhow`; there are no `unwrap`/`expect`s.
20//!
21//! The normalized shape written to the store is the store's own
22//! [`store::NewFeed`] / [`store::NewEntry`]; dedup is by feed-native GUID via
23//! [`store::insert_entries`]'s `ON CONFLICT (feed_id, guid)` upsert.
24
25use std::time::Duration;
26
27use anyhow::{Context, Result};
28use chrono::{DateTime, SecondsFormat, Utc};
29use feed_rs::model::{Entry as RawEntry, Feed as RawFeed, Link as RawLink, Text};
30use reqwest::header::{ETAG, IF_MODIFIED_SINCE, IF_NONE_MATCH, LAST_MODIFIED};
31use reqwest::{Client, StatusCode};
32use sqlx::SqlitePool;
33use url::Url;
34
35use crate::store::{self, Feed, NewEntry, NewFeed};
36
37/// The privacy classification of a feed URL — the output of
38/// [`classify_feed_privacy`].
39///
40/// A **private** feed carries a secret (a token / key / auth credential) *in the
41/// URL itself* — a Substack `…/feed/private/<token>`, a Patreon `?auth=…` feed,
42/// a Ghost members `?uuid=` feed, a private-podcast token feed (Supercast,
43/// Supporting Cast, tokened Megaphone/Acast+), and so on. FeatherReader stores a
44/// user's subscriptions as records in their **public PDS** (unauthenticated
45/// `getRecord` / `listRecords` + the firehose, retained even after delete), so
46/// writing such a URL anywhere — the PDS *or* the server's own store — would risk
47/// leaking paid / members-only access.
48///
49/// **Decision (stopgap until atproto permissioned data ships): FeatherReader
50/// supports PUBLIC feeds only.** A feed classified [`FeedPrivacy::Private`] is
51/// *refused* at the add / import boundary — never fetched, never stored, never
52/// written to the PDS. There is no local-secret fallback and no override: the
53/// server holds NO private secret, ever, which keeps "your data lives in your
54/// public PDS" 100% honest.
55#[derive(Debug, Clone, PartialEq, Eq)]
56pub enum FeedPrivacy {
57    /// No secret detected in the URL; safe to add as a public feed.
58    Public,
59    /// A secret was detected in the URL. The `String` is a short, human-readable
60    /// reason (for logging / the skip report), e.g. `"substack private feed
61    /// path"`. The feed is refused — not fetched, stored, or written anywhere.
62    Private(String),
63}
64
65impl FeedPrivacy {
66    /// Whether this classification is [`FeedPrivacy::Private`].
67    pub fn is_private(&self) -> bool {
68        matches!(self, FeedPrivacy::Private(_))
69    }
70}
71
72/// Query-parameter *keys* that, when present with a long/opaque value, mark a URL
73/// as carrying a secret. Conservative and lowercase-compared; matched as a whole
74/// key (case-insensitive) so a benign `keyword=` does NOT trip `key`. This is the
75/// generic, provider-agnostic credential-in-query defence — it catches paid
76/// feeds from providers we've never heard of. Covers Patreon (`auth`), Ghost
77/// members (`uuid`), token-in-query feeds (`token`/`key`/`k`/`sig`/`hash`), and
78/// the long tail (`access`/`apikey`/`private`/`password`/`u`/`s`/`p`/…).
79const SECRET_QUERY_KEYS: &[&str] = &[
80    "token", "key", "auth", "secret", "k", "sig", "hash", "access", "apikey", "api_key", "uuid",
81    "id", "u", "s", "p", "private", "password", "pw",
82];
83
84/// Path *segments* / fragments that mark a private-feed URL shape. Matched as a
85/// case-insensitive substring of the (lowercased) path so `/feed/private/<tok>`,
86/// `/members/…`, `/subscriber/…` etc. all trip regardless of the token that
87/// follows. Provider-agnostic: many paid providers expose members-only feeds
88/// under one of these path conventions.
89const PRIVATE_PATH_MARKERS: &[&str] = &[
90    "/private/",
91    "/feed/private/",
92    "/rss/private/",
93    "/private-feed/",
94    "/members/",
95    "/member/",
96    "/subscriber/",
97];
98
99/// A KNOWN paid/private feed provider, matched by host substring + (optionally) a
100/// path/query marker specific to that provider. This is the **secondary**,
101/// precision layer on top of the generic heuristic — it names providers so the
102/// skip report can say *why* and so we catch provider-specific shapes that the
103/// generic pass might rate as borderline. Data-driven and easy to extend: add a
104/// row, don't touch the matcher.
105struct KnownProvider {
106    /// Substring that must appear in the URL host (lowercased), e.g.
107    /// `substack.com`.
108    host_contains: &'static str,
109    /// Optional lowercased substring that must appear in the path-or-query for a
110    /// match (a provider's private-feed marker). `None` = the host alone is
111    /// enough (used for hosts that ONLY serve private/tokened feeds).
112    marker: Option<&'static str>,
113    /// Human-readable reason for the skip report.
114    reason: &'static str,
115}
116
117/// The known-provider table. Covers paid NEWSLETTERS and private PODCASTS — an
118/// RSS reader ingests both. Kept intentionally verbose/commented so it's obvious
119/// what each row targets and safe to extend.
120const KNOWN_PROVIDERS: &[KnownProvider] = &[
121    // --- Paid newsletters -------------------------------------------------
122    // Substack private feed: author.substack.com/feed/private/<token>.
123    KnownProvider {
124        host_contains: "substack.com",
125        marker: Some("/feed/private/"),
126        reason: "Substack private feed",
127    },
128    // Patreon RSS carries the member token as ?auth=.
129    KnownProvider {
130        host_contains: "patreon.com",
131        marker: Some("auth="),
132        reason: "Patreon member feed",
133    },
134    // Ghost members feed: ?uuid=<member-uuid> (or a members token path).
135    KnownProvider {
136        host_contains: "ghost.io",
137        marker: Some("uuid="),
138        reason: "Ghost members feed",
139    },
140    // Buttondown paid RSS uses a per-subscriber token in the path/query.
141    KnownProvider {
142        host_contains: "buttondown.email",
143        marker: Some("token"),
144        reason: "Buttondown premium feed",
145    },
146    KnownProvider {
147        host_contains: "buttondown.com",
148        marker: Some("token"),
149        reason: "Buttondown premium feed",
150    },
151    // Beehiiv premium RSS carries a subscriber token.
152    KnownProvider {
153        host_contains: "beehiiv.com",
154        marker: Some("token"),
155        reason: "Beehiiv premium feed",
156    },
157    // Memberful-gated feeds (host or ?auth token).
158    KnownProvider {
159        host_contains: "memberful.com",
160        marker: None,
161        reason: "Memberful members feed",
162    },
163    // Pico / Steady member feeds.
164    KnownProvider {
165        host_contains: "pico.link",
166        marker: None,
167        reason: "Pico member feed",
168    },
169    KnownProvider {
170        host_contains: "steadyhq.com",
171        marker: None,
172        reason: "Steady member feed",
173    },
174    // --- Private podcasts -------------------------------------------------
175    // Supercast private podcast feeds (host serves tokened member feeds only).
176    KnownProvider {
177        host_contains: "supercast.com",
178        marker: None,
179        reason: "Supercast private podcast",
180    },
181    KnownProvider {
182        host_contains: "supercast.tech",
183        marker: None,
184        reason: "Supercast private podcast",
185    },
186    // Supporting Cast private podcast feeds (supportingcast.fm).
187    KnownProvider {
188        host_contains: "supportingcast.fm",
189        marker: None,
190        reason: "Supporting Cast private podcast",
191    },
192    // RedCircle private/exclusive feeds.
193    KnownProvider {
194        host_contains: "redcircle.com",
195        marker: Some("private"),
196        reason: "RedCircle private podcast",
197    },
198    // Private/tokened Megaphone, Acast+, and Omny feeds carry an access token.
199    KnownProvider {
200        host_contains: "megaphone.fm",
201        marker: Some("token"),
202        reason: "Megaphone private podcast",
203    },
204    KnownProvider {
205        host_contains: "acast.com",
206        marker: Some("token"),
207        reason: "Acast+ private podcast",
208    },
209    KnownProvider {
210        host_contains: "omny.fm",
211        marker: Some("token"),
212        reason: "Omny private podcast",
213    },
214    // Apple / Spotify subscriber podcast feeds carry a per-listener token.
215    KnownProvider {
216        host_contains: "podcasts.apple.com",
217        marker: Some("token"),
218        reason: "Apple subscriber podcast",
219    },
220    KnownProvider {
221        host_contains: "spotify.com",
222        marker: Some("token"),
223        reason: "Spotify subscriber podcast",
224    },
225];
226
227/// Whether a URL may be **stored or published** as a feed URL at all.
228///
229/// This is the storage-side twin of the scheme check `net::check_scheme` applies
230/// before fetching. The fetch side has always been safe, because nothing can
231/// reach the network except through `net.rs` — but "safe to fetch" and "safe to
232/// write down" are different questions, and only the first had an answer.
233///
234/// Two paths took a URL from outside and stored it with no validation at all:
235/// `resolve_subscriptions` (any atproto client can write a subscription record
236/// into a user's repo) and the OPML import (`xmlUrl` is whatever the file says).
237/// `classify_feed_privacy` does not cover this — it deliberately returns
238/// `Public` for an unparseable URL, on the stated assumption that "the add path
239/// will reject it as malformed regardless", and those two paths are the ones
240/// that never had an add path to do the rejecting.
241///
242/// Note that `javascript:alert(1)` and `file:///etc/passwd` both *parse* cleanly
243/// as URLs, so parsing is not the check — the scheme is.
244pub fn is_storable_feed_url(url: &str, allow_at_uri: bool) -> bool {
245    // **`at://` is checked BEFORE `Url::parse`, because `Url::parse` cannot read
246    // the form that matters.** `at://did:plc:…/…` fails to parse with *invalid
247    // port number* — the colons in the DID are taken as a port separator — while
248    // the handle form `at://alice.example.com/…` parses fine. So adding `"at"`
249    // to the `matches!` below would appear to work and silently reject every
250    // DID-based at-URI, which is all of them in practice.
251    if let Some(rest) = crate::atproto::strip_at_prefix(url) {
252        // **Recognised case-insensitively, stored canonically.** Schemes are
253        // case-insensitive, so `At://` names the same publication — but
254        // `feeds.url` is UNIQUE, so accepting both spellings is two rows for
255        // one publication, the hazard the canonical-handle rule exists for.
256        // Recognising it here rather than letting it fall through to the
257        // generic checks is what keeps it out of the poller: nothing can fetch
258        // it under any spelling.
259        if !url.starts_with(crate::atproto::AT_URI_PREFIX) {
260            return false;
261        }
262        // **This gates STORING only — polling is handled by exclusion**, by
263        // kind rather than by any re-description of the URL, and
264        // `FeedKind::POLLABLE` is the one place the why is written down.
265        return allow_at_uri && is_storable_publication_uri(rest);
266    }
267    match Url::parse(url) {
268        Ok(u) => {
269            // No host check: for http(s) the `url` crate refuses every hostless
270            // spelling at parse (`http://`, `https://?q`, `http:///` are all
271            // "empty host") and turns `https:///x` into host `x`. A conjunct
272            // requiring a non-empty host was unreachable — a test hunt listed
273            // it as untested, and the honest answer was that no input reaches
274            // it. The `Err` arm below is what refuses a hostless URL.
275            matches!(u.scheme(), "http" | "https")
276        }
277        Err(_) => false,
278    }
279}
280
281/// The body of an `at://` URI — `<did-or-handle>/<collection>/<rkey>` — judged
282/// as a **storable feed**.
283///
284/// An allowlist entry, not a loosening: exactly one foreign collection is
285/// accepted, `site.standard.publication`. The two paths this guard exists for
286/// (`resolve_subscriptions`, the OPML import) take records written by any
287/// atproto client, so "it is an at-URI" is not a reason to store it — only "it
288/// is a publication this reader knows how to poll" is.
289fn is_storable_publication_uri(rest: &str) -> bool {
290    let mut parts = rest.split('/');
291    let (Some(authority), Some(collection), Some(rkey)) =
292        (parts.next(), parts.next(), parts.next())
293    else {
294        return false;
295    };
296    parts.next().is_none()
297        && collection == crate::lexicon::nsid::STANDARD_PUBLICATION
298        // **The rkey is validated against atproto's rules, not a blacklist.**
299        //
300        // A blacklist was the first attempt and it leaked twice: `is_control()`
301        // is Unicode category Cc only, so a bidi override (Cf) passed — and it
302        // reordered both the manage page and `scheduler.rs`'s `%feed.url` log
303        // line. Worse, nothing stopped a query string or fragment living inside
304        // the rkey, which satisfies the three-segment check and is exactly what
305        // `classify_feed_privacy`'s `at://` exemption keys off: a token
306        // smuggled there would have been declared public.
307        //
308        // An allowlist cannot leak the next character class someone finds.
309        // The charset alone still admitted `.`, `..` and a 10 000-byte key;
310        // `is_valid_rkey` carries the length and reserved-name rules too.
311        && crate::atproto::is_valid_rkey(rkey)
312        && is_storable_at_authority(authority)
313}
314
315/// The DID form only. `did:plc:` identifiers are validated by
316/// [`crate::oauth::identity::is_atproto_did`] rather than a `did:` prefix check,
317/// which would accept `did:plc:TOOSHORT`.
318///
319/// **The handle form is not storable, for the reason the canonical-handle rule
320/// already gave:** `feeds.url` is UNIQUE, so `at://alice.example.com/…` beside
321/// `at://did:plc:…/…` is two rows — two sidebar entries, and two polled copies
322/// once the reader is wired — for one publication. A handle is a mutable name
323/// for a DID; the row is keyed on the identity. Resolving a pasted or imported
324/// handle to its DID is the reader's job (#165 already resolves DIDs to their
325/// PDS), and belongs at input, not in storage.
326fn is_storable_at_authority(authority: &str) -> bool {
327    crate::oauth::identity::is_atproto_did(authority)
328}
329
330/// Classify whether a feed URL carries a secret credential in the URL itself.
331///
332/// Returns [`FeedPrivacy::Private`] (with a reason) when the URL looks like it
333/// embeds a token / key / auth credential, else [`FeedPrivacy::Public`].
334///
335/// **Design — provider-agnostic first.** The primary defence is a generic
336/// credential-in-URL heuristic that catches paid feeds from *any* provider, not
337/// just the ones we've named; a secondary known-provider table adds precision
338/// (and a nicer reason) for the common paid newsletters and private podcasts. We
339/// deliberately **bias toward flagging**: a false-positive block of a public feed
340/// is low-harm (the user just can't add that one feed yet), whereas a false
341/// negative would leak a paid secret onto the public network — high-harm.
342///
343/// Detection (any one is sufficient):
344/// 1. **Userinfo** — `https://user:pass@host/…` embeds credentials directly.
345/// 2. **Known private-feed path markers** — `PRIVATE_PATH_MARKERS`
346///    (`/feed/private/`, `/members/`, `/subscriber/`, …).
347/// 3. **Credential query parameters** — a query key in `SECRET_QUERY_KEYS` with
348///    a long/opaque value (Patreon `?auth=`, Ghost `?uuid=`, `?token=`, …).
349/// 4. **High-entropy opaque token segments** — a long opaque blob (hex ≥ 16,
350///    base64url ≥ 16, or a UUID) anywhere in the path or a query value, even
351///    without a telltale name.
352/// 5. **Known providers** — `KNOWN_PROVIDERS` host (+ optional marker) match.
353///
354/// An unparseable URL is treated as [`FeedPrivacy::Public`]: the add path rejects
355/// a malformed URL downstream anyway, and we don't want a parse quirk to
356/// misclassify.
357pub fn classify_feed_privacy(url: &str) -> FeedPrivacy {
358    // **`at://` is classified deliberately, and NOT doing so refused real
359    // subscriptions.** An atproto rkey is a TID — 13 base32-sortable characters
360    // — which is exactly what the generic "high-entropy token in path"
361    // heuristic below is looking for. Measured: without this arm,
362    // `at://did:plc:…/site.standard.publication/3lab2c4d5e6f7g8h` is
363    // classified PRIVATE and the subscription refused, while the DID form slips
364    // through only because it fails to parse as a `Url` at all.
365    //
366    // `Public` is the right answer: a publication is a public record in a
367    // public repo and the rkey is a handle, not a secret, so there is no
368    // private/paid shape for this scheme to carry.
369    if let Some(rest) = crate::atproto::strip_at_prefix(url) {
370        // A non-canonical scheme spelling is an at-URI this reader will not
371        // store, not an unparseable string for the `Err(_) => Public` arm below
372        // to wave through. Fail closed.
373        if !url.starts_with(crate::atproto::AT_URI_PREFIX) {
374            return FeedPrivacy::Private("non-canonical at:// scheme spelling".to_string());
375        }
376        // **Only a WELL-FORMED publication URI is exempt.** The first version
377        // of this was a bare prefix match, which declared any attacker-chosen
378        // string starting `at://` safe to publish — skipping the userinfo
379        // check, the known-provider table, the private-path markers, the
380        // secret-query keys and the entropy heuristics all at once. That is a
381        // regression against every one of them, on a path
382        // (`rename_subscription`) where this function is the only gate and the
383        // value is written to the user's PUBLIC repo.
384        //
385        // Anything else falls through to the generic checks below, which is
386        // where a credential-bearing string belongs.
387        if is_storable_publication_uri(rest) {
388            return FeedPrivacy::Public;
389        }
390        // **Malformed `at://` is REFUSED, not passed through.** Falling through
391        // reaches `Url::parse`, which fails on the DID form and lands on the
392        // `Err(_) => Public` arm below — whose justification is "the add path
393        // will reject it as a malformed URL regardless".
394        //
395        // **Defence in depth, not a live gate.** This comment used to say the
396        // justification is false on the `rename_subscription` path, "where this
397        // function is the only gate". That stopped being true when storability
398        // moved ahead of privacy on the repoint: a review then found no
399        // production caller can reach this arm at all — add pre-checks the
400        // `at://` prefix, rename and OPML and `resolve_subscriptions` all
401        // refuse a non-storable URL first. It stays because a fail-closed
402        // branch is worth its keep for the next caller that arrives without
403        // one, and because deleting a guard on the grounds that nothing
404        // currently reaches it is how the next one gets it wrong. It is not
405        // load-bearing today, and saying so is the honest version.
406        return FeedPrivacy::Private("not a well-formed at:// publication URI".to_string());
407    }
408    let parsed = match Url::parse(url) {
409        Ok(u) => u,
410        // Can't parse => the add path will reject it as a malformed URL regardless.
411        Err(_) => return FeedPrivacy::Public,
412    };
413
414    // (1) Userinfo (`https://user:pass@host/…`) — credentials in the authority.
415    if !parsed.username().is_empty() || parsed.password().is_some() {
416        return FeedPrivacy::Private("credentials in URL userinfo".to_string());
417    }
418
419    let path_lower = parsed.path().to_ascii_lowercase();
420    let query_lower = parsed.query().unwrap_or("").to_ascii_lowercase();
421    let host_lower = parsed.host_str().unwrap_or("").to_ascii_lowercase();
422
423    // (0) Public-feed allowlist. A handful of large, fully-public feed shapes
424    // carry a high-entropy-looking id in the query that would otherwise trip the
425    // generic entropy heuristic. YouTube channel/playlist RSS
426    // (`youtube.com/feeds/videos.xml?channel_id=UC…` / `?playlist_id=PL…`) is the
427    // canonical way any reader subscribes to a channel — the id is a PUBLIC
428    // handle, not a secret. Allowlist it before the generic checks so we don't
429    // false-block it. (Userinfo / known-provider markers are checked below and
430    // still apply, so this can't be used to smuggle a credential.)
431    if is_public_youtube_feed(&host_lower, &path_lower, &parsed) {
432        return FeedPrivacy::Public;
433    }
434
435    // (5) Known-provider precision layer (checked early so its specific reason
436    // wins over a generic one). Host substring + optional path/query marker.
437    for kp in KNOWN_PROVIDERS {
438        if host_lower.contains(kp.host_contains) {
439            let marker_ok = match kp.marker {
440                None => true,
441                Some(m) => {
442                    let m = m.to_ascii_lowercase();
443                    path_lower.contains(&m) || query_lower.contains(&m)
444                }
445            };
446            if marker_ok {
447                return FeedPrivacy::Private(kp.reason.to_string());
448            }
449        }
450    }
451
452    // (2) Known private-feed path markers.
453    for marker in PRIVATE_PATH_MARKERS {
454        if path_lower.contains(marker) {
455            return FeedPrivacy::Private(format!("private feed path `{marker}`"));
456        }
457    }
458
459    // (3) Credential query parameters with a long/opaque value.
460    for (k, v) in parsed.query_pairs() {
461        let key = k.as_ref().to_ascii_lowercase();
462        if SECRET_QUERY_KEYS.iter().any(|sk| *sk == key) && value_is_opaque(v.as_ref()) {
463            return FeedPrivacy::Private(format!("credential query parameter `{key}`"));
464        }
465    }
466
467    // (4) High-entropy opaque token segments (an embedded key/token with no
468    // telltale name): hex ≥ 16, base64url ≥ 16, or a UUID, in path or query.
469    // The dominant real-world private-podcast shape delivers the token as a
470    // *filename* (`<token>.rss` / `<token>.xml`) or affixed inside a larger
471    // segment (`feed-<uuid>`), so [`segment_hides_secret`] strips a trailing feed
472    // extension AND scans dot/underscore/hyphen-delimited sub-parts, not just the
473    // whole segment.
474    for seg in parsed.path().split('/').filter(|s| !s.is_empty()) {
475        if segment_hides_secret(seg) {
476            return FeedPrivacy::Private("high-entropy token in path".to_string());
477        }
478    }
479    for (_, v) in parsed.query_pairs() {
480        if looks_like_embedded_secret(v.as_ref()) {
481            return FeedPrivacy::Private("high-entropy token in query".to_string());
482        }
483    }
484
485    FeedPrivacy::Public
486}
487
488/// Whether a *named* credential query value (`?token=<v>`) is long/opaque enough
489/// to count as a secret. A short value (e.g. an enum like `?token=none`) is not.
490/// We treat a UUID, or anything ≥ 8 chars that isn't an obvious plain word, as
491/// opaque — named credential keys already signal intent, so the length bar is
492/// low.
493fn value_is_opaque(v: &str) -> bool {
494    if v.is_empty() {
495        return false;
496    }
497    if is_uuid(v) {
498        return true;
499    }
500    v.len() >= 8
501}
502
503/// Heuristic: does `s` look like an embedded secret (an opaque high-entropy
504/// token), as opposed to an ordinary slug or word? Matches a UUID, a hex string
505/// ≥ 16 chars, or a base64url-ish blob ≥ 16 chars that mixes letters and digits
506/// and isn't a hyphen/dot slug. Deliberately strict so it only fires on things
507/// that really look like keys — the named-marker and known-provider checks cover
508/// the rest.
509fn looks_like_embedded_secret(s: &str) -> bool {
510    if is_uuid(s) {
511        return true;
512    }
513    // Hex string ≥ 16 chars (e.g. a 32-char MD5-ish token).
514    if s.len() >= 16 && s.chars().all(|c| c.is_ascii_hexdigit()) {
515        return true;
516    }
517    // base64url-ish opaque blob ≥ 16 chars.
518    if s.len() < 16 {
519        return false;
520    }
521    // Hyphen/dot-heavy slugs (`this-is-a-normal-post-title`) are not secrets.
522    let separators = s
523        .bytes()
524        .filter(|b| *b == b'-' || *b == b'.' || *b == b' ')
525        .count();
526    if separators >= 3 {
527        return false;
528    }
529    // Must be plausibly token-charset: base64url alphabet only. `=` is accepted
530    // as base64 padding (it only ever appears trailing on a real blob, so a
531    // padded base64url token like `…dnc=` still counts).
532    let token_chars = s
533        .chars()
534        .filter(|c| c.is_ascii_alphanumeric() || matches!(c, '_' | '-' | '='))
535        .count();
536    if token_chars < s.chars().count() {
537        return false;
538    }
539    let has_alpha = s.chars().any(|c| c.is_ascii_alphabetic());
540    let has_digit = s.chars().any(|c| c.is_ascii_digit());
541    if !(has_alpha && has_digit) {
542        // A token almost always mixes letters and digits; a pure-alpha long
543        // segment is far more likely to be a normal (if long) slug/word.
544        return false;
545    }
546    // Distinct-character ratio: real tokens use most of the alphabet, words
547    // repeat a small set. Require >= 10 distinct chars for a 16+ char blob.
548    let mut seen = std::collections::HashSet::new();
549    for c in s.chars() {
550        seen.insert(c.to_ascii_lowercase());
551    }
552    seen.len() >= 10
553}
554
555/// Known feed/file extensions a token filename may wear (`<token>.rss`,
556/// `<token>.xml`, …). Stripped before the whole-segment secret test so a
557/// tokened *filename* — the dominant private-podcast URL shape — is still caught.
558const FEED_EXTENSIONS: &[&str] = &["rss", "xml", "atom", "json", "rss20"];
559
560/// Whether a single path segment hides an embedded secret. Beyond the plain
561/// whole-segment [`looks_like_embedded_secret`] test, this also catches the two
562/// real-world shapes that wrap a token so the whole segment is no longer a clean
563/// blob:
564///
565/// 1. **Token-as-filename** — `<token>.rss` / `<token>.xml`: strip a trailing
566///    feed extension and re-test the stem.
567/// 2. **Token affixed inside a larger segment** — `feed-<uuid>`, `<token>.xml`,
568///    `pod_<hex32>`: split on `.`/`_`/`-` and test each sub-part, so a
569///    high-entropy blob delimited by an affix is still found.
570fn segment_hides_secret(seg: &str) -> bool {
571    if looks_like_embedded_secret(seg) {
572        return true;
573    }
574    // (1) Strip a trailing known feed extension and re-test the stem.
575    if let Some((stem, ext)) = seg.rsplit_once('.') {
576        if FEED_EXTENSIONS.contains(&ext.to_ascii_lowercase().as_str())
577            && looks_like_embedded_secret(stem)
578        {
579            return true;
580        }
581    }
582    // (2) A UUID embedded with an affix (`feed-<uuid>`, `<uuid>-audio`) — the
583    // `-` delimiters inside the UUID mean a naive split can't see it, so scan for
584    // a canonical UUID substring directly.
585    if contains_uuid(seg) {
586        return true;
587    }
588    // (3) Scan `.`/`_`/`-`-delimited sub-parts for a high-entropy blob affixed to
589    // an ordinary word (`pod_<hex32>`, `<hex32>.mp3`). Only fires on multi-part
590    // segments (a single-part segment was already covered by the whole-segment
591    // test above), so a plain `my-normal-post-slug` — whose parts are short
592    // dictionary words — can't trip it.
593    if seg.contains(['.', '_', '-']) {
594        for part in seg.split(['.', '_', '-']).filter(|p| !p.is_empty()) {
595            if looks_like_embedded_secret(part) {
596                return true;
597            }
598        }
599    }
600    false
601}
602
603/// Whether `s` contains a canonical 8-4-4-4-12 UUID as a substring (allowing an
604/// affix on either side, e.g. `feed-<uuid>` or `<uuid>-audio`). Slides a 36-char
605/// window over the string and tests each with [`is_uuid`].
606fn contains_uuid(s: &str) -> bool {
607    const UUID_LEN: usize = 36; // 8+4+4+4+12 + 4 hyphens.
608    let bytes = s.as_bytes();
609    if bytes.len() < UUID_LEN {
610        return false;
611    }
612    // ASCII-only window: a UUID is pure ASCII hex/hyphen, so byte indexing is
613    // safe here (a multi-byte char in the window just fails is_uuid).
614    (0..=bytes.len() - UUID_LEN).any(|i| s.get(i..i + UUID_LEN).map(is_uuid).unwrap_or(false))
615}
616
617/// Whether this is a PUBLIC YouTube channel/playlist RSS feed
618/// (`www.youtube.com/feeds/videos.xml?channel_id=UC…` or `?playlist_id=PL…`).
619/// The channel/playlist id is a public handle, not a credential, so these feeds
620/// must NOT be flagged by the generic entropy heuristic. We require the exact
621/// public host + feeds path + one of the two public id keys, so this narrow
622/// allowlist can't be abused to smuggle a `?token=` past classification.
623fn is_public_youtube_feed(host_lower: &str, path_lower: &str, parsed: &Url) -> bool {
624    let host_ok = host_lower == "youtube.com"
625        || host_lower == "www.youtube.com"
626        || host_lower.ends_with(".youtube.com");
627    if !host_ok || !path_lower.starts_with("/feeds/videos.xml") {
628        return false;
629    }
630    // Only the public id keys may appear; a `token`/`auth`/… key means treat it
631    // as a normal (potentially private) URL and let the checks below run.
632    parsed.query_pairs().all(|(k, _)| {
633        let k = k.as_ref().to_ascii_lowercase();
634        k == "channel_id" || k == "playlist_id" || k == "user"
635    })
636}
637
638/// Whether `s` is a canonical 8-4-4-4-12 hyphenated UUID (any hex case).
639fn is_uuid(s: &str) -> bool {
640    let groups = [8usize, 4, 4, 4, 12];
641    let parts: Vec<&str> = s.split('-').collect();
642    if parts.len() != groups.len() {
643        return false;
644    }
645    parts
646        .iter()
647        .zip(groups.iter())
648        .all(|(p, &n)| p.len() == n && p.chars().all(|c| c.is_ascii_hexdigit()))
649}
650
651/// How long a single feed fetch may take before we give up.
652const FETCH_TIMEOUT: Duration = Duration::from_secs(30);
653
654/// Per-read idle timeout: cap the wait for the *next* body chunk, so a server
655/// that trickles bytes forever can't tie up a fetch under the total timeout.
656const READ_TIMEOUT: Duration = Duration::from_secs(15);
657
658/// Base backoff applied after a failed poll; the caller multiplies this by the
659/// feed's consecutive-error count (with a ceiling) to space out retries.
660const BACKOFF_BASE: Duration = Duration::from_secs(300);
661
662/// Ceiling on backoff so a persistently broken feed still gets retried daily.
663const BACKOFF_MAX: Duration = Duration::from_secs(24 * 3600);
664
665/// The outcome of polling a single feed. Lets the scheduler decide how to
666/// reschedule (and lets tests assert what happened) without inspecting the DB.
667#[derive(Debug, Clone, PartialEq, Eq)]
668pub enum PollOutcome {
669    /// The feed was fetched, parsed, and stored. `new_entries` is the number of
670    /// entries inserted or updated by this poll.
671    Updated { new_entries: u64 },
672    /// The server returned `304 Not Modified` — nothing changed, nothing stored.
673    NotModified,
674    /// The fetch or parse failed; the feed was left intact and skipped. Carries
675    /// the suggested backoff before the next attempt. Never a panic.
676    ///
677    /// **`kind` and `detail` are the reason, and they exist because their
678    /// absence cost a production investigation.** Until #159 the error was
679    /// logged here and discarded, so `feeds` recorded that a feed was failing
680    /// and never why — which is how sixty feeds broken by our own 304 handling
681    /// looked exactly like sixty dead blogs. `kind` is a small closed
682    /// vocabulary so failures can be counted by cause; `detail` is the message
683    /// for a human reading one row.
684    Failed {
685        backoff: Duration,
686        kind: FailureKind,
687        detail: String,
688    },
689}
690
691/// Why a poll failed, as a closed set.
692///
693/// Closed on purpose: the point is to *count* failures by cause, and a free-text
694/// kind cannot be counted. The detail string carries whatever else matters.
695#[derive(Debug, Clone, Copy, PartialEq, Eq)]
696pub enum FailureKind {
697    /// The request never produced a response — DNS, TLS, timeout, connection
698    /// refused, or a refusal by the SSRF guard.
699    Fetch,
700    /// A response arrived with a non-success status.
701    Status,
702    /// The body was too large, or reading it failed part-way.
703    Body,
704    /// The body arrived and is not a feed this parser can read.
705    Parse,
706}
707
708/// What the poller does with a feed row.
709///
710/// **A column, not a predicate.** "Can this be fetched?" used to be
711/// `substr(url, 1, 5) = 'at://'` spliced into four statements, with a fifth
712/// reader that had already drifted from them. Deciding it once, in Rust, at
713/// insert — and storing the answer — means SQL cannot disagree with the
714/// fetcher, and wiring the standard.site reader becomes a change to this
715/// function plus a dispatch, rather than an edit to every statement that
716/// mentions a URL.
717#[derive(Debug, Clone, Copy, PartialEq, Eq)]
718pub enum FeedKind {
719    /// An RSS/Atom/JSON feed document fetched over http(s).
720    Rss,
721    /// An `at://…/site.standard.publication/…` record pair in somebody's PDS,
722    /// read by `standard_site` and pollable since 0.4.0.
723    Publication,
724    /// Any other `at://` row: another collection, or a spelling the storage
725    /// guard would refuse. Rows like this predate that guard. **Never
726    /// pollable** — handing one to the standard.site reader would fail its
727    /// collection check every tick and publish it as an unreachable publisher —
728    /// and counted as unpollable so the capacity it holds stays visible.
729    Unsupported,
730}
731
732impl FeedKind {
733    /// The kinds the scheduler may select: RSS, and since 0.4.0 standard.site
734    /// publications. [`FeedKind::Unsupported`] is the kind it never selects.
735    ///
736    /// **This is the canonical home of the exclusion; the other sites point
737    /// here.** It used to be a SQL string predicate in `store`, carrying
738    /// its own copy of the rule — which is how one reader (`count_feeds`) came
739    /// to drift from it unnoticed.
740    ///
741    /// Why an unpollable kind is skipped rather than failed: an `Unsupported`
742    /// row names nothing either reader can fetch — another collection, a handle,
743    /// a non-canonical spelling — so polling it could only fail.
744    /// Handing such a row to the poller does not leave the feature dormant — it
745    /// manufactures one permanent failure per row, which the public cause
746    /// histogram then reports as an unreachable publisher. Unsupported is not
747    /// broken, and telling those apart is the entire reason a failure cause is
748    /// recorded. (Rows like this exist: subscriptions written by other clients
749    /// before this reader refused the scheme.)
750    ///
751    /// **`store::count_feeds` is deliberately NOT filtered by this.** The
752    /// global ceiling bounds storage on a small box, and an unpollable row
753    /// occupies a row, so it counts against the cap. An earlier version of this
754    /// note listed the readers without naming the exception, which read as
755    /// completeness it did not have: a review found the ceiling consuming
756    /// capacity that appeared on no surface, since `/stats` measures the poller
757    /// and excludes these rows. `/admin/metrics` renders
758    /// `store::unpollable_feeds` for exactly that reason.
759    /// **Adding a kind here makes a population of rows due all at once.**
760    /// `store::due_feeds` orders `next_poll IS NOT NULL, next_poll ASC`, so a
761    /// row with no scheduled poll sorts ahead of every dated one. Rows that
762    /// were never pollable have no schedule, so the boot that reclassifies
763    /// them hands the poller a block of N rows that outrank every regular
764    /// feed until they drain — `ceil(N / batch)` ticks, measured, during which
765    /// `/stats` shows a climbing backlog and nothing logs why. Bounded and
766    /// harmless at ninety feeds; not at ten thousand. Whoever wires the next
767    /// kind should seed or stagger `next_poll` for the rows it admits.
768    pub const POLLABLE: &'static [FeedKind] = &[FeedKind::Rss, FeedKind::Publication];
769
770    /// The kinds whose entries the retention **window** applies to.
771    ///
772    /// **Age is the wrong retention policy for an archive, and that is a
773    /// measurement, not a preference.** Three real publications, read through
774    /// `standard_site::fetch` on 2026-09-27: Standard.site's newest document was
775    /// **131 days** old, Annotated's **109** (with its oldest at 373), and minus
776    /// listens' **241**. Against the instance default window of 14 days, every
777    /// one of them stored **zero** rows — a green poll, an empty feed, and an
778    /// info log as the only trace. Long-form publishing is not news-paced.
779    ///
780    /// So a publication is bounded by COUNT instead: `max_entries_per_feed` in
781    /// `insert_entries`, which caps a feed at the newest N plus up to N starred.
782    /// That is a real bound — it is what keeps this from being "retention off" —
783    /// and it is the one that suits a source whose value is its archive.
784    ///
785    /// A generous absolute ceiling still applies (`publication_retention_days`),
786    /// because "not aged out" must not mean "immortal": rows belonging to a feed
787    /// nobody polls any more would otherwise never be reaped at all, and the
788    /// per-feed trim only runs when a poll stores something.
789    pub const AGED: &'static [FeedKind] = &[FeedKind::Rss];
790
791    /// The column value. Stable — it is persisted.
792    pub fn as_str(self) -> &'static str {
793        match self {
794            FeedKind::Rss => "rss",
795            FeedKind::Publication => "publication",
796            FeedKind::Unsupported => "unsupported",
797        }
798    }
799
800    /// A closed vocabulary on the way back in: a kind written by a newer build
801    /// is not silently read as one this build knows.
802    pub fn parse(raw: &str) -> Option<Self> {
803        match raw {
804            "rss" => Some(FeedKind::Rss),
805            "publication" => Some(FeedKind::Publication),
806            "unsupported" => Some(FeedKind::Unsupported),
807            _ => None,
808        }
809    }
810
811    /// What a URL will be stored as. The only place the question is asked.
812    ///
813    /// `store::feeds.kind` is a cache of this function, re-derived from the URL
814    /// on every start and on every upsert, so a change here needs no migration
815    /// and SQL cannot hold an opinion of its own about what a row is.
816    ///
817    /// **Case-insensitive on purpose.** An earlier version was case-sensitive,
818    /// with a test pinning that a mixed-case `At://` row IS handed to the
819    /// poller — reasoning that if Rust does not call it an at-URI, neither
820    /// should anything else. That was wrong in the direction that matters: URL
821    /// schemes are case-insensitive, so `Url::parse` folds `At://` to scheme
822    /// `at` and `net::check_scheme` refuses it (the DID form does not parse at
823    /// all). The row could only fail, every tick, forever, and be published in
824    /// the `fetch` bucket as an unreachable publisher. Storing a non-canonical
825    /// spelling is separately refused, because `feeds.url` is UNIQUE.
826    pub fn of(url: &str) -> Self {
827        match crate::atproto::strip_at_prefix(url) {
828            // **Only a URI the storage guard would accept is a publication.**
829            // Since publications became pollable, "it is an at-URI" is no longer
830            // a safe enough reason: another collection would be polled, fail,
831            // and back off forever while reading as an unreachable publisher.
832            // The canonical lowercase scheme too, for the same reason: storage
833            // refuses `At://` (#183), and a legacy row spelled that way would
834            // otherwise be polled and fail its parse every tick.
835            Some(rest) if url.starts_with("at://") && is_storable_publication_uri(rest) => {
836                FeedKind::Publication
837            }
838            Some(_) => FeedKind::Unsupported,
839            None => FeedKind::Rss,
840        }
841    }
842}
843
844/// Apply a [`PollOutcome`] to the feed's row: settle the error columns AND
845/// reschedule it. **Both halves, always, from one place.**
846///
847/// The scheduler did this inline. `web::add_subscription` then copied only the
848/// first half, so a poll taken off the scheduler could clear a stale failure's
849/// COUNT while leaving the feed parked on its stale backoff HORIZON — reported
850/// healthy, not polled for up to 24h. And its failures fed `consecutive_errors`
851/// with no reschedule, so repeated Subscribe clicks drove a shared feed to the
852/// 24h ceiling for every subscriber. Two copies of a sequence drift; this is
853/// the one copy.
854///
855/// `cadence` is the interval to use on success; the scheduler derives it from
856/// the feed's hint, a direct caller passes the configured default. A store
857/// failure is logged and the reschedule still attempted, so a hiccup writing
858/// the count cannot strand the feed at a NULL `next_poll` that `due_feeds`
859/// would then re-poll every tick.
860pub async fn settle_poll(
861    pool: &sqlx::SqlitePool,
862    url: &str,
863    outcome: &PollOutcome,
864    cadence: Duration,
865) {
866    let delay = match outcome {
867        // A 304 is a healthy poll: it proves the fetch worked and nothing changed.
868        PollOutcome::Updated { .. } | PollOutcome::NotModified => {
869            if let Err(err) = crate::store::reset_feed_errors(pool, url).await {
870                tracing::warn!(feed = %url, %err, "failed to reset feed error count");
871            }
872            cadence
873        }
874        PollOutcome::Failed {
875            backoff,
876            kind,
877            detail,
878        } => {
879            // Recompute from the feed's REAL consecutive-error count so a
880            // persistently-broken feed climbs toward the ceiling instead of
881            // retrying at the floor forever; fall back to the outcome's floor.
882            match crate::store::bump_feed_errors(pool, url, *kind, detail).await {
883                Ok(count) => backoff_for(count.max(1) as u32),
884                Err(err) => {
885                    tracing::warn!(feed = %url, %err, "failed to bump feed error count; using floor backoff");
886                    *backoff
887                }
888            }
889        }
890    };
891    if let Err(err) = crate::store::set_next_poll(pool, url, delay).await {
892        tracing::error!(feed = %url, %err, "failed to persist next_poll");
893    }
894}
895
896/// Cap on a failure detail, applied where the string is BUILT.
897///
898/// It was originally applied only inside `store::bump_feed_errors`, which
899/// bounded the database row and nothing else — and widening
900/// [`PollOutcome::Failed`] with this field had quietly opened a second sink:
901/// `web::add_subscription` logs `?outcome` at INFO on a user-facing request
902/// path. Bounding at construction bounds every sink, including ones added
903/// later by someone who never reads this comment.
904pub const MAX_FAILURE_DETAIL_CHARS: usize = 300;
905
906/// Render an error chain into a bounded [`PollOutcome::Failed`] detail.
907///
908/// `{e:#}` — the anyhow CHAIN, not just the outermost context. "fetching
909/// https://…" alone says nothing; the cause is the part that would have named
910/// the 304 bug in #159.
911pub fn failure_detail(err: impl std::fmt::Display) -> String {
912    let s = err.to_string();
913    if s.chars().count() <= MAX_FAILURE_DETAIL_CHARS {
914        return s;
915    }
916    s.chars().take(MAX_FAILURE_DETAIL_CHARS).collect()
917}
918
919impl FailureKind {
920    /// The stable string stored in `feeds.last_error_kind` and aggregated on
921    /// `/stats`. Changing one of these silently rewrites history in the
922    /// aggregate, so they are spelled out rather than derived from the variant.
923    pub fn as_str(self) -> &'static str {
924        match self {
925            Self::Fetch => "fetch",
926            Self::Status => "status",
927            Self::Body => "body",
928            Self::Parse => "parse",
929        }
930    }
931
932    /// Read back a persisted `last_error_kind`. `None` for anything this
933    /// version does not know, so a row written by a newer build is not
934    /// silently attributed to a cause this one recognises — the same contract
935    /// [`crate::metrics::Backend::parse`] keeps for the same reason.
936    pub fn parse(raw: &str) -> Option<Self> {
937        match raw {
938            "fetch" => Some(Self::Fetch),
939            "status" => Some(Self::Status),
940            "body" => Some(Self::Body),
941            "parse" => Some(Self::Parse),
942            _ => None,
943        }
944    }
945
946    /// Every variant, so a test can assert over the whole set rather than a
947    /// list that drifts when a variant is added.
948    pub const ALL: [Self; 4] = [Self::Fetch, Self::Status, Self::Body, Self::Parse];
949}
950
951/// Build a `reqwest::Client` configured for polite **and safe** feed fetching.
952///
953/// Callers should build this **once** and share it (connection pooling), then
954/// hand a reference to [`poll_feed`]. Kept here so the fetch policy (UA,
955/// timeout, redirect behaviour) lives with the code that depends on it.
956///
957/// Auto-redirect is **disabled** on purpose: feed URLs are untrusted, so
958/// redirects are followed manually by [`crate::net::guarded_get`], which
959/// re-validates the scheme + resolved IP of every hop (SSRF defence). A client
960/// that silently followed redirects could be bounced onto `169.254.169.254` or
961/// `127.0.0.1` between the guard's check and the connect.
962pub fn build_client() -> Result<Client> {
963    Client::builder()
964        .user_agent(crate::USER_AGENT)
965        .timeout(FETCH_TIMEOUT)
966        .read_timeout(READ_TIMEOUT)
967        // Ignore ambient proxy configuration, for the same reason the pinned
968        // client does: a proxied request hands the hostname to the proxy to
969        // resolve, so `net`'s IP checks never see the address they are meant to
970        // vet. See `net::build_pinned_client`.
971        .no_proxy()
972        // No auto-redirect: net::guarded_get follows + re-validates each hop.
973        .redirect(reqwest::redirect::Policy::none())
974        .build()
975        .context("failed to build feed HTTP client")
976}
977
978/// Compute the backoff for the `n`th consecutive failure (1-based), clamped to
979/// `BACKOFF_MAX`. Exponential in the error count so transient blips retry soon
980/// while a durably-broken feed backs off toward daily.
981///
982/// The scheduler passes the feed's persisted `consecutive_errors` count (see
983/// [`crate::store::bump_feed_errors`]) so a feed that keeps failing actually
984/// climbs toward `BACKOFF_MAX` instead of retrying at the floor forever.
985pub fn backoff_for(consecutive_errors: u32) -> Duration {
986    let n = consecutive_errors.max(1);
987    // Saturating shift: base * 2^(n-1), capped. Avoids overflow for large n.
988    let factor = 1u64.checked_shl(n.saturating_sub(1)).unwrap_or(u64::MAX);
989    let secs = BACKOFF_BASE
990        .as_secs()
991        .saturating_mul(factor)
992        .min(BACKOFF_MAX.as_secs());
993    Duration::from_secs(secs)
994}
995
996/// Poll one feed by **what it is**: an RSS/Atom/JSON document over HTTP, or a
997/// standard.site publication read from its author's PDS.
998///
999/// The single entry point the scheduler calls, so the choice of reader lives
1000/// with [`FeedKind`] and not in the poll loop. Same contract as [`poll_feed`]:
1001/// `Err` is a broken local store, never a misbehaving source.
1002pub async fn poll_feed_by_kind(
1003    pool: &SqlitePool,
1004    client: &Client,
1005    config: &crate::config::Config,
1006    feed: &Feed,
1007) -> Result<PollOutcome> {
1008    match FeedKind::of(&feed.url) {
1009        FeedKind::Rss => poll_feed(pool, client, feed, config.max_entries_per_feed).await,
1010        FeedKind::Publication => poll_publication(pool, client, config, feed).await,
1011        // Never selected: `due_feeds` reads only POLLABLE kinds. A defensive
1012        // failure rather than a panic if a caller hands one in anyway.
1013        FeedKind::Unsupported => Ok(PollOutcome::Failed {
1014            backoff: backoff_for(1),
1015            kind: FailureKind::Parse,
1016            detail: failure_detail(format!("{} is not a feed this reader can poll", feed.url)),
1017        }),
1018    }
1019}
1020
1021/// What a failed publication read is filed under in the cause histogram.
1022///
1023/// **`Fetch` means the request never produced a response**, so it is the
1024/// fallback, not the default answer: a PLC directory or PDS that answered —
1025/// with a 404 for a tombstoned DID, `RepoNotFound`, `RepoDeactivated` — is
1026/// `Status`, and an answer that was not what it claimed to be is `Parse`. Filed
1027/// as `Fetch`, a deleted account read as its server being down (found in
1028/// review).
1029fn publication_failure_kind(err: &anyhow::Error) -> FailureKind {
1030    use crate::atproto::{AtProtoError, DidResolutionCause};
1031    for cause in err.chain() {
1032        if cause.is::<crate::standard_site::NotAPublication>() || cause.is::<serde_json::Error>() {
1033            return FailureKind::Parse;
1034        }
1035        match cause.downcast_ref::<AtProtoError>() {
1036            Some(AtProtoError::Xrpc { .. }) => return FailureKind::Status,
1037            Some(AtProtoError::DidResolution { cause, .. }) => {
1038                return match cause {
1039                    DidResolutionCause::Status => FailureKind::Status,
1040                    DidResolutionCause::UnsupportedMethod | DidResolutionCause::NoPdsEndpoint => {
1041                        FailureKind::Parse
1042                    }
1043                    DidResolutionCause::NotAPublicTarget => FailureKind::Fetch,
1044                }
1045            }
1046            _ => {}
1047        }
1048    }
1049    FailureKind::Fetch
1050}
1051
1052/// Read a standard.site publication from its author's PDS and store it.
1053///
1054/// A source failure — an unparseable URI, an unreachable PLC directory or
1055/// PDS, a walk that failed — is a [`PollOutcome::Failed`] with backoff, like an
1056/// RSS fetch failure, so `settle_poll` and `/stats` treat both kinds alike.
1057/// Only a broken local store is an `Err`, which `store_publication` decides.
1058async fn poll_publication(
1059    pool: &SqlitePool,
1060    client: &Client,
1061    config: &crate::config::Config,
1062    feed: &Feed,
1063) -> Result<PollOutcome> {
1064    poll_publication_group(pool, client, config, std::slice::from_ref(feed))
1065        .await
1066        .pop()
1067        .unwrap_or_else(|| Err(anyhow::anyhow!("no outcome for {}", feed.url)))
1068}
1069
1070/// Read several publications **from one repo** with one walk of its documents
1071/// (`standard_site::fetch_repo`), and store each. One outcome per feed, in
1072/// order. The caller groups by repo; a feed from another repo, or one whose
1073/// URL is not a publication URI, gets its own failure and does not affect the
1074/// rest.
1075///
1076/// **Why one walk:** cost scales with the repo, not the publication. Nine
1077/// publications in one repo were nine full walks of its documents.
1078pub async fn poll_publication_group(
1079    pool: &SqlitePool,
1080    client: &Client,
1081    config: &crate::config::Config,
1082    feeds: &[Feed],
1083) -> Vec<Result<PollOutcome>> {
1084    let failed = |kind: FailureKind, detail: String| -> Result<PollOutcome> {
1085        Ok(PollOutcome::Failed {
1086            backoff: backoff_for(1),
1087            kind,
1088            detail: failure_detail(detail),
1089        })
1090    };
1091    let uris: Vec<Option<crate::standard_site::AtUri>> = feeds
1092        .iter()
1093        .map(|f| crate::standard_site::AtUri::parse(&f.url))
1094        .collect();
1095    let Some(did) = uris.iter().flatten().next().map(|u| u.authority.clone()) else {
1096        return feeds
1097            .iter()
1098            .map(|f| {
1099                failed(
1100                    FailureKind::Parse,
1101                    format!("{} is not a readable at:// URI", f.url),
1102                )
1103            })
1104            .collect();
1105    };
1106    // Only the feeds of that repo with a publication URI are read together.
1107    let readable: Vec<usize> = (0..feeds.len())
1108        .filter(|&i| {
1109            uris[i].as_ref().is_some_and(|u| {
1110                u.authority == did && u.collection == crate::lexicon::nsid::STANDARD_PUBLICATION
1111            })
1112        })
1113        .collect();
1114    let rkeys: Vec<String> = readable
1115        .iter()
1116        .map(|&i| uris[i].as_ref().unwrap().rkey.clone())
1117        .collect();
1118
1119    // **One deadline for the whole read.** It is otherwise bounded only per
1120    // request (FETCH_TIMEOUT x MAX_LIST_PAGES): hours, against a repo that
1121    // pages slowly, all of it holding up the publication loop.
1122    let fetched = tokio::time::timeout(
1123        config.publication_read_deadline,
1124        crate::standard_site::fetch_repo(client, &config.oauth.plc_directory, &did, &rkeys),
1125    )
1126    .await;
1127    let mut reads: Vec<Option<anyhow::Result<crate::standard_site::PublicationRead>>> =
1128        feeds.iter().map(|_| None).collect();
1129    let repo_failure = match fetched {
1130        Ok(Ok(per)) => {
1131            for (slot, read) in readable.iter().zip(per) {
1132                reads[*slot] = Some(read);
1133            }
1134            None
1135        }
1136        Ok(Err(err)) => Some((publication_failure_kind(&err), format!("{err:#}"))),
1137        Err(_) => Some((
1138            FailureKind::Fetch,
1139            format!(
1140                "the read did not finish within {:?}",
1141                config.publication_read_deadline
1142            ),
1143        )),
1144    };
1145
1146    let (retention_days, retention_hard_days) = config.retention_for(FeedKind::Publication);
1147    let mut out = Vec::with_capacity(feeds.len());
1148    for (i, feed) in feeds.iter().enumerate() {
1149        let outcome = match (reads[i].take(), &repo_failure) {
1150            (Some(Ok(read)), _) => {
1151                crate::standard_site::store_publication(
1152                    pool,
1153                    &feed.url,
1154                    read,
1155                    config.max_entries_per_feed,
1156                    retention_days,
1157                    retention_hard_days,
1158                )
1159                .await
1160            }
1161            (Some(Err(err)), _) => failed(publication_failure_kind(&err), format!("{err:#}")),
1162            (None, Some((kind, detail))) if readable.contains(&i) => failed(*kind, detail.clone()),
1163            (None, _) => failed(
1164                FailureKind::Parse,
1165                format!("{} is not a publication in {did}", feed.url),
1166            ),
1167        };
1168        out.push(outcome);
1169    }
1170    out
1171}
1172
1173/// Fetch, parse, sanitize, normalize, and store a single feed.
1174///
1175/// Performs a conditional GET using the feed's stored `ETag` / `Last-Modified`.
1176/// On `304` it returns [`PollOutcome::NotModified`] without touching entries. On
1177/// `200` it parses with `feed-rs`, sanitizes every entry's HTML with `ammonia`,
1178/// upserts the feed row (carrying the fresh validators) and inserts new entries
1179/// (deduped by GUID). Any fetch/parse error is logged and returned as
1180/// [`PollOutcome::Failed`] — it never panics and never propagates as `Err` for
1181/// a merely-broken feed, so one bad publisher can't stall the scheduler.
1182///
1183/// `Err` is reserved for *store* failures (a broken local DB is a real error the
1184/// caller should see), not for feed misbehaviour.
1185///
1186/// `max_entries_per_feed` caps how many entries this feed retains after insert
1187/// (newest N by published date); `<= 0` disables the per-feed trim.
1188pub async fn poll_feed(
1189    pool: &SqlitePool,
1190    client: &Client,
1191    feed: &Feed,
1192    max_entries_per_feed: i64,
1193) -> Result<PollOutcome> {
1194    // --- conditional GET (through the SSRF guard) ----------------------------
1195    // The guard re-validates the scheme + resolved IP of the target and of every
1196    // redirect hop, so a subscribed feed can't bounce the poller onto an
1197    // internal address (cloud metadata / loopback). Conditional-GET validators
1198    // ride along as extra headers.
1199    let mut extra: Vec<(reqwest::header::HeaderName, reqwest::header::HeaderValue)> = Vec::new();
1200    if let Some(etag) = feed.etag.as_deref() {
1201        if let Ok(v) = reqwest::header::HeaderValue::from_str(etag) {
1202            extra.push((IF_NONE_MATCH, v));
1203        }
1204    }
1205    if let Some(lm) = feed.last_modified.as_deref() {
1206        if let Ok(v) = reqwest::header::HeaderValue::from_str(lm) {
1207            extra.push((IF_MODIFIED_SINCE, v));
1208        }
1209    }
1210
1211    let resp = match crate::net::guarded_get(client, &feed.url, &extra).await {
1212        Ok(r) => r,
1213        Err(e) => {
1214            tracing::warn!(feed = %feed.url, error = %e, "feed fetch failed (or blocked by SSRF guard)");
1215            return Ok(PollOutcome::Failed {
1216                backoff: backoff_for(1),
1217                kind: FailureKind::Fetch,
1218                detail: failure_detail(format!("{e:#}")),
1219            });
1220        }
1221    };
1222
1223    let status = resp.status();
1224    if status == StatusCode::NOT_MODIFIED {
1225        tracing::debug!(feed = %feed.url, "feed not modified (304)");
1226        // Bump last_polled/next_poll only; leave validators + entries untouched.
1227        touch_polled(pool, &feed.url, None, None)
1228            .await
1229            .with_context(|| format!("touch_polled after 304 for {}", feed.url))?;
1230        return Ok(PollOutcome::NotModified);
1231    }
1232    if !status.is_success() {
1233        tracing::warn!(feed = %feed.url, %status, "feed returned non-success status");
1234        return Ok(PollOutcome::Failed {
1235            backoff: backoff_for(1),
1236            kind: FailureKind::Status,
1237            detail: failure_detail(status),
1238        });
1239    }
1240
1241    // Capture validators for the *next* conditional GET before consuming body.
1242    let new_etag = header_str(resp.headers().get(ETAG));
1243    let new_last_modified = header_str(resp.headers().get(LAST_MODIFIED));
1244
1245    // Stream the body with a hard byte cap, aborting mid-stream if it exceeds
1246    // it. We never trust Content-Length: reqwest's gzip layer strips it, so a
1247    // small gzip bomb could otherwise inflate to GBs before any size check.
1248    let body = match crate::net::read_capped(resp).await {
1249        Ok(b) => b,
1250        Err(e) => {
1251            tracing::warn!(feed = %feed.url, error = %e, "feed body rejected (too large / read error)");
1252            return Ok(PollOutcome::Failed {
1253                backoff: backoff_for(1),
1254                kind: FailureKind::Body,
1255                detail: failure_detail(format!("{e:#}")),
1256            });
1257        }
1258    };
1259
1260    // --- parse (malformed feed => log + skip, never panic) -------------------
1261    let parsed = match parse_feed(&body[..]) {
1262        Ok(f) => f,
1263        Err(e) => {
1264            tracing::warn!(feed = %feed.url, error = %e, "malformed feed; skipping");
1265            return Ok(PollOutcome::Failed {
1266                backoff: backoff_for(1),
1267                kind: FailureKind::Parse,
1268                detail: failure_detail(format!("{e:#}")),
1269            });
1270        }
1271    };
1272
1273    // --- normalize + sanitize ------------------------------------------------
1274    let (title, site_url) = feed_metadata(&parsed);
1275    let new_feed = NewFeed {
1276        url: feed.url.clone(),
1277        title,
1278        site_url,
1279        etag: new_etag,
1280        last_modified: new_last_modified,
1281        last_polled: Some(now_rfc3339()),
1282        next_poll: None, // the scheduler owns cadence; leave it to set next_poll.
1283    };
1284
1285    let entries: Vec<NewEntry> = parsed.entries.iter().map(normalize_entry).collect();
1286
1287    // --- store (a store failure IS a real error) -----------------------------
1288    let feed_id = store::upsert_feed(pool, &new_feed)
1289        .await
1290        .with_context(|| format!("upsert_feed for {}", feed.url))?;
1291    let n = store::insert_entries(pool, feed_id, &entries, max_entries_per_feed)
1292        .await
1293        .with_context(|| format!("insert_entries for {}", feed.url))?;
1294
1295    tracing::info!(feed = %feed.url, entries = n, "feed polled");
1296    Ok(PollOutcome::Updated { new_entries: n })
1297}
1298
1299/// Bump `last_polled` (and optionally validators) without changing entries —
1300/// used on the `304 Not Modified` path.
1301async fn touch_polled(
1302    pool: &SqlitePool,
1303    url: &str,
1304    etag: Option<String>,
1305    last_modified: Option<String>,
1306) -> Result<()> {
1307    // `None` means "keep current" — upsert_feed COALESCEs the validators, so a
1308    // 304 that repeats no headers leaves the stored ones untouched. This used to
1309    // re-read the row and re-supply them by hand because the upsert clobbered
1310    // unconditionally; the read-modify-write is gone now that the upsert is
1311    // honest, and with it a race where a concurrent poll's validators could be
1312    // read here and written back stale.
1313    let nf = NewFeed {
1314        url: url.to_string(),
1315        etag,
1316        last_modified,
1317        last_polled: Some(now_rfc3339()),
1318        ..Default::default()
1319    };
1320    store::upsert_feed(pool, &nf).await?;
1321    Ok(())
1322}
1323
1324/// Extract `(title, site_url)` from a parsed feed. `site_url` prefers an
1325/// `alternate`/no-rel HTML link over the feed's self link.
1326fn feed_metadata(parsed: &RawFeed) -> (Option<String>, Option<String>) {
1327    let title = parsed
1328        .title
1329        .as_ref()
1330        .map(|t| bound_text(text_plain(t), MAX_TITLE_BYTES));
1331    let site_url = parsed
1332        .links
1333        .iter()
1334        // Prefer an explicit human-facing page: rel="alternate" or no rel at all.
1335        .find(|l| {
1336            l.rel.as_deref() == Some("alternate")
1337                || (l.rel.is_none()
1338                    && l.media_type.as_deref() != Some("application/rss+xml")
1339                    && l.media_type.as_deref() != Some("application/atom+xml"))
1340        })
1341        .or_else(|| {
1342            parsed
1343                .links
1344                .iter()
1345                .find(|l| l.rel.as_deref() != Some("self"))
1346        })
1347        .or_else(|| parsed.links.first())
1348        .map(|l| bound_text(l.href.clone(), MAX_URL_BYTES));
1349    (title, site_url)
1350}
1351
1352/// Parse a feed body with feed-rs, generating ids for id-less entries the way
1353/// feed-rs 2.4 did.
1354///
1355/// An entry's id is its dedup key, so how feed-rs fills a missing one is part
1356/// of FeatherReader's storage contract. feed-rs 3.0 also puts `<comments>` and
1357/// `wfw:commentRss` URLs in `entry.links`, and its default generator hashes the
1358/// FIRST link with the title — so an id-less item listing its comments link
1359/// before `<link>` got a new id on upgrade, and was stored a second time. This
1360/// generator hashes the first link that is not a comments link, which is the
1361/// link 2.4 hashed, through the same public 2.4/3.0 function.
1362///
1363/// With no such link, it returns an empty id rather than feed-rs's fallback (a
1364/// random UUID, which made the item a new row on every poll); `normalize_entry`
1365/// then derives [`stable_guid`]. `parse` has no base URI, so feed-rs's
1366/// uri+title branch was never reached and nothing else is lost.
1367///
1368/// **A panic inside feed-rs is returned as an error.** feed-rs 3.0 panics on
1369/// some hostile-but-plausible input — an author address with a multi-byte
1370/// character beside it, e.g. `jose@example.com(José)` (its name/address
1371/// splitter slices on a byte that is not a character boundary). Feed bodies are
1372/// arbitrary web input, so any such panic is a malformed feed: it takes the
1373/// ordinary parse-failure path (logged, `FailureKind::Parse`, backed off) rather
1374/// than unwinding out of the poll task, which recorded nothing and left the
1375/// feed silently un-polled. Unwinding is safe here: the parser is built and
1376/// dropped inside the closure and touches no state of ours.
1377fn parse_feed(body: &[u8]) -> Result<RawFeed> {
1378    let parsed = std::panic::catch_unwind(|| {
1379        feed_rs::parser::Builder::new()
1380            .id_generator(entry_id)
1381            .build()
1382            .parse(body)
1383    });
1384    match parsed {
1385        Ok(result) => Ok(result?),
1386        Err(panic) => {
1387            let why = panic
1388                .downcast_ref::<String>()
1389                .map(String::as_str)
1390                .or_else(|| panic.downcast_ref::<&str>().copied())
1391                .unwrap_or("non-string panic payload");
1392            anyhow::bail!("feed parser panicked: {why}")
1393        }
1394    }
1395}
1396
1397/// The id generator [`parse_feed`] installs; see there.
1398fn entry_id(links: &[RawLink], title: &Option<Text>, _uri: Option<&str>) -> String {
1399    match links.iter().find(|l| is_primary_link(l)) {
1400        Some(link) => feed_rs::parser::generate_id_from_link_and_title(link, title),
1401        None => String::new(),
1402    }
1403}
1404
1405/// Whether a link is one of the entry's own links rather than a pointer to its
1406/// comments (feed-rs 3.0's `<comments>` / `wfw:commentRss`, marked by
1407/// `target`). Only these are candidates for the permalink and the id, which is
1408/// all feed-rs 2.4 ever put in `entry.links`.
1409fn is_primary_link(l: &RawLink) -> bool {
1410    l.target.is_none()
1411}
1412
1413/// Turn a parsed [`RawEntry`] into the store's [`NewEntry`], sanitizing HTML.
1414///
1415/// Content preference: full `content` body, else `summary`. Whichever is chosen
1416/// is **always** passed through [`sanitize_html`] before storage. GUID falls
1417/// back to the entry link, then to a stable hash of title+link, so an entry
1418/// missing an `id` still deduplicates instead of being re-inserted forever.
1419fn normalize_entry(e: &RawEntry) -> NewEntry {
1420    let url = entry_link(e);
1421    let content_html = e
1422        .content
1423        .as_ref()
1424        .and_then(|c| c.body.as_deref())
1425        .or_else(|| e.summary.as_ref().map(|t| t.content.as_str()))
1426        .map(|raw| sanitize_html_bounded(raw, MAX_CONTENT_HTML_BYTES));
1427
1428    // GUID may use the raw link (dedup key only, never rendered), so prefer the
1429    // entry's first raw link for identity even when it's not a safe href.
1430    let guid = if !e.id.trim().is_empty() {
1431        e.id.trim().to_string()
1432    } else if let Some(link) = raw_entry_link(e) {
1433        link
1434    } else {
1435        // Last resort: derive a stable id so re-fetches dedup rather than dupe.
1436        stable_guid(e)
1437    };
1438
1439    NewEntry {
1440        guid: bound_guid(guid),
1441        url: url.map(|u| bound_text(u, MAX_URL_BYTES)),
1442        title: e
1443            .title
1444            .as_ref()
1445            .map(|t| bound_text(text_plain(t), MAX_TITLE_BYTES)),
1446        author: entry_author(e).map(|a| bound_text(a, MAX_AUTHOR_BYTES)),
1447        published: entry_time(e),
1448        content_html,
1449        fetched_at: None, // store defaults to "now".
1450    }
1451}
1452
1453/// The raw best-permalink URL for an entry (no scheme filtering) — used only as
1454/// a dedup GUID, never rendered as an href.
1455///
1456/// Comments links are never candidates (see [`is_primary_link`]).
1457fn raw_entry_link(e: &RawEntry) -> Option<String> {
1458    let mut links = e.links.iter().filter(|l| is_primary_link(l));
1459    links
1460        .clone()
1461        .find(|l| l.rel.as_deref() == Some("alternate") || l.rel.is_none())
1462        .or_else(|| links.next())
1463        .map(|l| l.href.clone())
1464}
1465
1466/// The best display/permalink URL for an entry, **scheme-allow-listed** so it is
1467/// safe to render as an `href`: prefer `rel="alternate"` or a no-rel link, else
1468/// the first link — but only if it is an `http`/`https` URL. A `javascript:` or
1469/// `data:` permalink (a stored-XSS vector that survives HTML escaping, since it
1470/// carries no HTML-special characters) is dropped here at ingest, before it can
1471/// ever reach the store or a template.
1472fn entry_link(e: &RawEntry) -> Option<String> {
1473    raw_entry_link(e).and_then(|href| crate::net::safe_link(&href))
1474}
1475
1476/// The first author's name, if it has one.
1477///
1478/// feed-rs 3.0 splits RSS's `address (Name)` into an email and a name, but
1479/// keeps the parentheses: `(Name)`. They are stripped here. An author given
1480/// only as an address has no name and no byline — the address is not shown.
1481/// (feed-rs 2.4 named every RSS `<author>` "author", and an Atom author with no
1482/// name "unknown"; those placeholders were stored as bylines.)
1483///
1484/// The byline is the first author that has a name once stripped: an item may
1485/// give `<author>` as a bare address and the name in `<dc:creator>`.
1486fn entry_author(e: &RawEntry) -> Option<String> {
1487    e.authors.iter().find_map(|p| {
1488        let name = p.name.as_deref()?.trim();
1489        let name = [('(', ')'), ('<', '>'), ('[', ']')]
1490            .iter()
1491            .find_map(|&(open, close)| name.strip_prefix(open)?.strip_suffix(close))
1492            .unwrap_or(name)
1493            .trim();
1494        (!name.is_empty()).then(|| name.to_string())
1495    })
1496}
1497
1498/// Best CREDIBLE publication time (published, else updated) as an RFC3339
1499/// string, or `None` when neither is credible.
1500///
1501/// **A date in the future is discarded, not clamped, and not stored.** There was
1502/// no upper bound here, so an item dated in the year 2999 was stored verbatim
1503/// and became permanent: both retention sweeps test
1504/// `COALESCE(published, fetched_at) < cutoff` and a future date is never less
1505/// than either, the per-feed keep-set orders on the same expression `DESC` where
1506/// it is rank one forever, and every list view puts it at the top. A publisher
1507/// with a broken clock does that by accident; anyone wanting a permanent slot at
1508/// the top of a reader's list does it on purpose.
1509///
1510/// Discarded rather than clamped to now because the entry upsert refreshes
1511/// `published` on every poll while stamping `fetched_at` once — so a value
1512/// derived from the current clock is rewritten every cycle and the row can never
1513/// age at all. Clamping relocates the defect. Undated is the honest answer, and
1514/// `fetched_at` then dates the row and holds still. That is the rule the
1515/// publication path already follows; see `standard_site::entries_from_records`.
1516///
1517/// **Each candidate is judged separately**, so a bogus `<published>` beside a
1518/// credible `<updated>` keeps the good date. That helps Atom, and RSS 2 only
1519/// when the item carries an `<atom:updated>` (read since feed-rs 3.0):
1520/// otherwise `feed-rs` copies `published` into `updated` when `updated` is
1521/// absent (`parser/rss2/mod.rs`), so the second candidate holds the same value
1522/// and the fall-through is a no-op. Worth keeping where two independent dates
1523/// exist; worth not overstating where they do not.
1524///
1525/// **The ceiling is [`MAX_FUTURE_PUBLISHED_DAYS`], NOT the publication path's
1526/// clock-skew grace, and the asymmetry is deliberate.** There, a refused
1527/// `publishedAt` falls back to the record key's TID — the real write time, a
1528/// credible date — so a five-minute bound costs almost nothing. Here there is no
1529/// such fallback: refusing leaves the entry undated and the reader sees no date
1530/// at all. Five minutes is sized for skew between two clocks, while the ordinary
1531/// cause of a future `pubDate` is a local time stamped `+0000` (up to 14 hours
1532/// out, the widest real UTC offset) or a post scheduled a little ahead. Those are
1533/// dates worth keeping, and they stop being future on their own.
1534///
1535/// What the bound must prevent is a date that can never become past, because that
1536/// is what makes a row permanently unsweepable, un-evictable and first in the
1537/// list.
1538fn entry_time(e: &RawEntry) -> Option<String> {
1539    let ceiling = Utc::now() + chrono::Duration::days(MAX_FUTURE_PUBLISHED_DAYS);
1540    e.published
1541        .filter(|d| *d <= ceiling)
1542        .or_else(|| e.updated.filter(|d| *d <= ceiling))
1543        .map(fmt_time)
1544}
1545
1546/// How far ahead of now a feed may date an entry before [`entry_time`] refuses
1547/// the date and lets `fetched_at` stand in.
1548///
1549/// Two days: the widest real UTC offset is +14:00, so a local time mislabelled as
1550/// UTC lands inside this, as does a post scheduled slightly ahead. Both are dates
1551/// worth keeping, and both stop being future without help. Anything further is
1552/// refused, because a date that never becomes past is what makes a row
1553/// permanently unsweepable and permanently first in the reading list.
1554pub(crate) const MAX_FUTURE_PUBLISHED_DAYS: i64 = 2;
1555
1556/// Extract the plain string content of a feed [`Text`] node.
1557fn text_plain(t: &Text) -> String {
1558    t.content.trim().to_string()
1559}
1560
1561/// Sanitize hostile feed **HTML** with ammonia's whitelist cleaner. Applied to
1562/// every RSS/Atom entry body unconditionally, because every one of them is
1563/// markup.
1564///
1565/// **Not for plain text.** An earlier version of this comment claimed it was
1566/// "safe on plain text too (it will simply escape/strip as needed)". It is
1567/// not: `clean` PARSES its input, so a bare `<` in prose swallows the rest —
1568/// `"if x<y then z"` comes back as `"if x"`. That sentence is how a
1569/// plain-text field got run through here once already. Use
1570/// [`plain_text_to_html`].
1571pub(crate) fn sanitize_html(raw: &str) -> String {
1572    ammonia::clean(raw)
1573}
1574
1575/// Render **plain text** into the HTML the `content_html` column holds.
1576///
1577/// **Not [`sanitize_html`].** `ammonia::clean` parses its input as markup, so a
1578/// bare `<` in prose swallows the rest: measured here, `"if x<y then z"` comes
1579/// back as `"if x"`. That is correct for an RSS body, which IS markup, and
1580/// silent data loss for a field a lexicon defines as text. Escape first, then
1581/// add the only markup this needs — line breaks, which the column's consumer
1582/// renders as HTML and would otherwise collapse.
1583pub(crate) fn plain_text_to_html(raw: &str) -> String {
1584    let escaped = raw
1585        .replace('&', "&amp;")
1586        .replace('<', "&lt;")
1587        .replace('>', "&gt;");
1588    // Safe by order: every `<` from the input is already `&lt;` before this
1589    // adds a real tag.
1590    escaped.replace('\n', "<br>")
1591}
1592
1593/// **What one entry, or one feed row, may store per field (#205).**
1594///
1595/// Every field below arrives from someone else's server — an RSS document or a
1596/// publisher's `site.standard.document` record — and nothing between the wire
1597/// and SQLite used to shorten it. A title could be megabytes, then indexed, read
1598/// back and rendered into every list view of that feed.
1599///
1600/// Chosen from measurement, not guessed. Production's 4,389 entries on
1601/// 2026-10-03: title max 253 bytes (p99 148), url max 235 (p99 178), author max
1602/// 26, content_html max 86,969 (p99 17,535). Each bound is at least 8x the
1603/// largest real value, so no ordinary article is touched; `MAX_TITLE_BYTES` is
1604/// `site.standard.document`'s own `title.maxLength`.
1605///
1606/// **Truncated, never refused.** Dropping an article because one field is long
1607/// is the failure mode the retention work was careful to avoid.
1608pub(crate) const MAX_TITLE_BYTES: usize = 5_000;
1609/// See [`MAX_TITLE_BYTES`].
1610pub(crate) const MAX_AUTHOR_BYTES: usize = 1_000;
1611/// See [`MAX_TITLE_BYTES`]. A truncated URL is a broken link, which was judged
1612/// better than no link; at 35x the longest real one it should never happen.
1613pub(crate) const MAX_URL_BYTES: usize = 8_192;
1614/// See [`MAX_TITLE_BYTES`]. Applies to the STORED HTML, after sanitizing or
1615/// escaping — see [`sanitize_html_bounded`] for why that is the bound that matters.
1616pub(crate) const MAX_CONTENT_HTML_BYTES: usize = 2 * 1024 * 1024;
1617/// An entry id longer than this is replaced by a stable hash of the whole id
1618/// (see [`bound_guid`]): it is the dedup key, under a UNIQUE index.
1619pub(crate) const MAX_GUID_BYTES: usize = 2_048;
1620
1621/// The largest index `<= at` that is a character boundary of `s`.
1622fn floor_char_boundary(s: &str, at: usize) -> usize {
1623    if at >= s.len() {
1624        return s.len();
1625    }
1626    (0..=at).rev().find(|&i| s.is_char_boundary(i)).unwrap_or(0)
1627}
1628
1629/// `s` cut to at most `max` bytes, on a character boundary. Plain-text fields
1630/// only: cutting markup or an escaped string here could split a tag or an
1631/// entity, which is what [`sanitize_html_bounded`] and [`plain_text_to_html_bounded`] exist for.
1632pub(crate) fn bound_text(mut s: String, max: usize) -> String {
1633    let cut = floor_char_boundary(&s, max);
1634    s.truncate(cut);
1635    s
1636}
1637
1638/// Plain text escaped into `content_html`, cut so the **output** fits `max`.
1639///
1640/// **Exact, in one pass.** Escaping is a fixed size per character (`&` is
1641/// five bytes, `<` and `>` four, a newline `<br>` four, anything else its UTF-8
1642/// length), so the longest prefix whose escaped form fits is found by adding
1643/// those up — no rendering, no search. Three rounds of review found bugs in a
1644/// generic re-render search that this replaces (#224).
1645pub(crate) fn plain_text_to_html_bounded(raw: &str, max: usize) -> String {
1646    let mut size = 0usize;
1647    let mut cut = raw.len();
1648    for (i, c) in raw.char_indices() {
1649        let escaped = match c {
1650            '&' => 5,
1651            '<' | '>' | '\n' => 4,
1652            c => c.len_utf8(),
1653        };
1654        if size + escaped > max {
1655            cut = i;
1656            break;
1657        }
1658        size += escaped;
1659    }
1660    plain_text_to_html(&raw[..cut])
1661}
1662
1663/// Feed HTML sanitized into `content_html`, cut so the **output** fits `max`.
1664///
1665/// **Sanitize once, then cut the sanitized output, not the input.** Cutting
1666/// the input made the result depend on how much of it the sanitizer would
1667/// strip — a large `data:` image, `<style>` or unterminated comment — and the
1668/// searches that tried to account for that kept nothing, or ran for hours, in
1669/// review (#224). Sanitized HTML is already clean: cutting it on a character
1670/// boundary and sanitizing that prefix again only closes the tags the cut left
1671/// open, so the second pass usually grows it by little. The first cut leaves a
1672/// margin for that growth; deeply nested markup, whose closers can outgrow any
1673/// margin, falls through to a bounded bisection.
1674///
1675/// Cost: the first sanitize is the one every body always had; the bounded
1676/// passes — one, or at most 1 + [`SANITIZE_BOUND_ATTEMPTS`] — run on at most
1677/// `max` bytes of already-clean HTML, and only for a body over the bound.
1678pub(crate) fn sanitize_html_bounded(raw: &str, max: usize) -> String {
1679    let clean = sanitize_html(raw);
1680    if clean.len() <= max {
1681        return clean;
1682    }
1683    // First, the cut that almost always works: just under the bound, with a
1684    // margin for the closers the cut leaves open. When it fits, that is the
1685    // answer — within 1/64 of the bound, in one extra pass.
1686    let margin = (max / 64).max(64);
1687    let first = floor_char_boundary(&clean, max.saturating_sub(margin));
1688    let again = sanitize_html(&clean[..first]);
1689    if again.len() <= max {
1690        return again;
1691    }
1692    // **Then a bounded bisection, not a widening margin.** A closing tag is
1693    // longer than the tag it opens, so a cut through deeply nested markup can
1694    // grow past the bound by more than any fixed margin; widening the margin
1695    // 4x a round reached a cut of 0 and stored nothing where nearly all of it
1696    // fit (found in review). `lo` always fits (the empty prefix does), `hi`
1697    // never does, and the next probe is taken AFTER they move.
1698    let (mut lo, mut hi) = (0usize, first);
1699    let mut best = String::new();
1700    let resolution = (max / 1024).max(1);
1701    for _ in 0..SANITIZE_BOUND_ATTEMPTS {
1702        if hi - lo <= resolution {
1703            break;
1704        }
1705        let mut cut = floor_char_boundary(&clean, lo + (hi - lo) / 2);
1706        if cut <= lo {
1707            // A multi-byte character straddles the midpoint: step past it
1708            // rather than give up.
1709            cut = ceil_char_boundary(&clean, lo + 1);
1710            if cut >= hi {
1711                break;
1712            }
1713        }
1714        let out = sanitize_html(&clean[..cut]);
1715        if out.len() <= max {
1716            lo = cut;
1717            best = out;
1718        } else {
1719            hi = cut;
1720        }
1721    }
1722    best
1723}
1724
1725/// The smallest index `>= at` that is a character boundary of `s`.
1726fn ceil_char_boundary(s: &str, at: usize) -> usize {
1727    (at..=s.len())
1728        .find(|&i| s.is_char_boundary(i))
1729        .unwrap_or(s.len())
1730}
1731
1732/// The most bisection passes [`sanitize_html_bounded`] spends after its first
1733/// cut: enough to resolve a 2 MiB bound to about 1/1024 of it.
1734const SANITIZE_BOUND_ATTEMPTS: usize = 14;
1735
1736/// An entry id, or a stable stand-in for one too long to index.
1737///
1738/// The id is the dedup key under `UNIQUE (feed_id, guid)`, so truncating it
1739/// would merge distinct entries that share a long prefix. A hash of the WHOLE
1740/// id keeps them apart and keeps the same entry deduplicating across polls —
1741/// the same construction as [`stable_guid`].
1742pub(crate) fn bound_guid(guid: String) -> String {
1743    if guid.len() <= MAX_GUID_BYTES {
1744        return guid;
1745    }
1746    use std::hash::{Hash, Hasher};
1747    let mut h = dedup_hasher();
1748    guid.hash(&mut h);
1749    format!("featherreader:long-guid:{:016x}", h.finish())
1750}
1751
1752/// The hasher behind the stored dedup keys of [`bound_guid`] and
1753/// [`stable_guid`]: SipHash-1-3 with zero keys, from the `siphasher` crate.
1754///
1755/// **Not `std`'s `DefaultHasher`**, whose algorithm the standard library
1756/// documents as unspecified and free to change between Rust releases: a
1757/// toolchain bump could re-key every such entry, and each would be stored a
1758/// second time. `DefaultHasher` is SipHash-1-3 with zero keys today, so this
1759/// produces the values already stored; tests pin them.
1760fn dedup_hasher() -> siphasher::sip::SipHasher13 {
1761    siphasher::sip::SipHasher13::new_with_keys(0, 0)
1762}
1763
1764/// Format a chrono timestamp as RFC3339 (UTC, seconds precision) to match the
1765/// store's string columns.
1766pub(crate) fn fmt_time(dt: DateTime<Utc>) -> String {
1767    dt.to_rfc3339_opts(SecondsFormat::Secs, true)
1768}
1769
1770/// "Now" in the store's RFC3339 shape.
1771fn now_rfc3339() -> String {
1772    Utc::now().to_rfc3339_opts(SecondsFormat::Secs, true)
1773}
1774
1775/// A stable GUID derived from an entry's title + first link, for feeds that
1776/// supply neither an id nor a usable link id. Deterministic so re-fetches dedup.
1777fn stable_guid(e: &RawEntry) -> String {
1778    use std::hash::{Hash, Hasher};
1779    let mut h = dedup_hasher();
1780    e.title.as_ref().map(|t| t.content.as_str()).hash(&mut h);
1781    e.links
1782        .iter()
1783        .find(|l| is_primary_link(l))
1784        .map(|l| l.href.as_str())
1785        .hash(&mut h);
1786    e.summary.as_ref().map(|s| s.content.as_str()).hash(&mut h);
1787    format!("featherreader:synthetic:{:016x}", h.finish())
1788}
1789
1790/// Decode an HTTP header value to an owned `String`, dropping non-UTF-8 values.
1791fn header_str(v: Option<&reqwest::header::HeaderValue>) -> Option<String> {
1792    v.and_then(|h| h.to_str().ok()).map(str::to_string)
1793}
1794
1795/// Discover a feed URL from a site's HTML via
1796/// `<link rel="alternate" type="application/rss+xml|atom+xml" href="…">`.
1797///
1798/// Returns the first RSS/Atom autodiscovery link found, resolved against the
1799/// page URL if the `href` is relative. This is what lets a user paste a *site*
1800/// URL and have FeatherReader find the actual feed ("subscribe by URL").
1801/// Returns `None` if the HTML carries no autodiscovery link.
1802///
1803/// The `base` is the URL the HTML was fetched from, used to resolve relative
1804/// `href`s. Pass `None` to only accept absolute hrefs.
1805pub fn discover_feed(site_html: &str, base: Option<&Url>) -> Option<Url> {
1806    // Parse the HTML with html5ever (via ammonia's dependency graph is separate;
1807    // use a light hand-rolled scan over <link> tags to avoid a new dependency).
1808    // We look for <link ...> elements whose rel contains "alternate" and whose
1809    // type is an RSS/Atom feed media type, and take the href.
1810    for tag in link_tags(site_html) {
1811        let rel = attr(&tag, "rel").unwrap_or_default().to_ascii_lowercase();
1812        let typ = attr(&tag, "type").unwrap_or_default().to_ascii_lowercase();
1813        let is_feed_type = typ.contains("application/rss+xml")
1814            || typ.contains("application/atom+xml")
1815            || typ.contains("application/feed+json")
1816            || typ.contains("application/json");
1817        // rel="alternate" is the standard; be lenient and also accept a bare
1818        // feed type with any rel, but require the feed media type either way.
1819        let rel_ok = rel.split_whitespace().any(|r| r == "alternate") || rel.is_empty();
1820        if is_feed_type && rel_ok {
1821            if let Some(href) = attr(&tag, "href") {
1822                let href = href.trim();
1823                if href.is_empty() {
1824                    continue;
1825                }
1826                // Absolute URL wins directly; otherwise resolve against `base`.
1827                // Either way, only http(s): the href is publisher-controlled and
1828                // `Url::parse` accepts any scheme, so this is where an `at://`
1829                // (or `file:`, `javascript:`) alternate would otherwise become
1830                // the URL the add path stores — after its input gate has run.
1831                // Skip, don't stop: a later real feed link still wins.
1832                let resolved = match Url::parse(href) {
1833                    Ok(u) => Some(u),
1834                    Err(_) => base.and_then(|b| b.join(href).ok()),
1835                };
1836                match resolved {
1837                    Some(u) if matches!(u.scheme(), "http" | "https") => return Some(u),
1838                    _ => continue,
1839                }
1840            }
1841        }
1842    }
1843    None
1844}
1845
1846/// Extract the raw text of every `<link ...>` tag (self-closing or not) from an
1847/// HTML string. A deliberately small, allocation-light scan — feed
1848/// autodiscovery does not need a full DOM, and avoiding one keeps the dependency
1849/// surface minimal (design bias: boring, small-dependency).
1850fn link_tags(html: &str) -> Vec<String> {
1851    let mut out = Vec::new();
1852    let bytes = html.as_bytes();
1853    let lower = html.to_ascii_lowercase();
1854    let mut search_from = 0usize;
1855    while let Some(rel_idx) = lower[search_from..].find("<link") {
1856        let start = search_from + rel_idx;
1857        // Ensure it's a tag boundary ("<link" followed by whitespace, '>' or '/').
1858        let after = bytes.get(start + 5).copied();
1859        let boundary = matches!(after, Some(b) if b == b' ' || b == b'\t' || b == b'\n' || b == b'\r' || b == b'>' || b == b'/');
1860        if !boundary {
1861            search_from = start + 5;
1862            continue;
1863        }
1864        // Find the closing '>' for this tag.
1865        if let Some(end_rel) = html[start..].find('>') {
1866            let end = start + end_rel;
1867            out.push(html[start..=end].to_string());
1868            search_from = end + 1;
1869        } else {
1870            break;
1871        }
1872    }
1873    out
1874}
1875
1876/// Read an attribute value from a single tag string, handling both single- and
1877/// double-quoted values. Case-insensitive attribute name match.
1878fn attr(tag: &str, name: &str) -> Option<String> {
1879    let lower = tag.to_ascii_lowercase();
1880    let needle = format!("{name}=");
1881    let mut from = 0usize;
1882    while let Some(rel) = lower[from..].find(&needle) {
1883        let name_start = from + rel;
1884        // Guard against matching a suffix of a longer attribute name
1885        // (e.g. matching "type=" inside "mytype="): the char before must be a
1886        // tag/whitespace boundary.
1887        let ok_prefix = name_start == 0
1888            || matches!(
1889                tag.as_bytes().get(name_start - 1),
1890                Some(b' ') | Some(b'\t') | Some(b'\n') | Some(b'\r') | Some(b'<')
1891            );
1892        let val_start = name_start + needle.len();
1893        if !ok_prefix {
1894            from = val_start;
1895            continue;
1896        }
1897        let rest = &tag[val_start..];
1898        let quote = rest.chars().next();
1899        let value = match quote {
1900            Some('"') => rest[1..].split('"').next(),
1901            Some('\'') => rest[1..].split('\'').next(),
1902            // Unquoted: read up to whitespace, '>' or '/'.
1903            _ => rest
1904                .split(|c: char| c.is_whitespace() || c == '>' || c == '/')
1905                .next(),
1906        };
1907        return value.map(str::to_string);
1908    }
1909    None
1910}
1911
1912#[cfg(test)]
1913mod tests {
1914    use super::*;
1915
1916    const RSS_SAMPLE: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
1917<rss version="2.0">
1918  <channel>
1919    <title>Example RSS Feed</title>
1920    <link>https://example.com/</link>
1921    <description>An example feed for tests</description>
1922    <item>
1923      <title>First post</title>
1924      <link>https://example.com/first</link>
1925      <guid>https://example.com/first</guid>
1926      <author>alice@example.com (Alice)</author>
1927      <pubDate>Fri, 10 Jul 2026 08:00:00 GMT</pubDate>
1928      <description><![CDATA[<p>Hello <b>world</b>.</p><script>alert('xss')</script><img src="x" onerror="alert(1)">]]></description>
1929    </item>
1930    <item>
1931      <title>Second post</title>
1932      <link>https://example.com/second</link>
1933      <guid>guid-second</guid>
1934      <pubDate>Sat, 11 Jul 2026 08:00:00 GMT</pubDate>
1935      <description><![CDATA[<a href="javascript:alert(1)">click</a><a href="https://ok.example/">ok</a>]]></description>
1936    </item>
1937  </channel>
1938</rss>"#;
1939
1940    const ATOM_SAMPLE: &str = r#"<?xml version="1.0" encoding="utf-8"?>
1941<feed xmlns="http://www.w3.org/2005/Atom">
1942  <title>Example Atom Feed</title>
1943  <link rel="alternate" href="https://atom.example.com/"/>
1944  <link rel="self" href="https://atom.example.com/feed.xml"/>
1945  <id>urn:uuid:feed-1</id>
1946  <updated>2026-07-11T08:00:00Z</updated>
1947  <entry>
1948    <title>Atom entry</title>
1949    <id>urn:uuid:entry-1</id>
1950    <link rel="alternate" href="https://atom.example.com/a"/>
1951    <author><name>Bob</name></author>
1952    <updated>2026-07-11T08:00:00Z</updated>
1953    <content type="html"><![CDATA[<p>Safe <em>text</em>.</p><script>steal()</script><iframe src="evil"></iframe>]]></content>
1954  </entry>
1955</feed>"#;
1956
1957    /// [`ATOM_SAMPLE`] with the `self` link before the `alternate` one.
1958    const ATOM_SELF_FIRST: &str = r#"<?xml version="1.0" encoding="utf-8"?>
1959<feed xmlns="http://www.w3.org/2005/Atom">
1960  <title>Example Atom Feed</title>
1961  <link rel="self" href="https://atom.example.com/feed.xml"/>
1962  <link rel="alternate" href="https://atom.example.com/"/>
1963  <id>urn:uuid:feed-1</id>
1964  <updated>2026-07-11T08:00:00Z</updated>
1965  <entry>
1966    <title>Atom entry</title>
1967    <id>urn:uuid:entry-1</id>
1968    <link rel="alternate" href="https://atom.example.com/a"/>
1969    <author><name>Bob</name></author>
1970    <updated>2026-07-11T08:00:00Z</updated>
1971    <content type="html"><![CDATA[<p>Safe <em>text</em>.</p><script>steal()</script><iframe src="evil"></iframe>]]></content>
1972  </entry>
1973</feed>"#;
1974
1975    // ---- #205: what one entry may store, per field ----------------------
1976
1977    /// An RSS document carrying one item with exactly these fields.
1978    fn rss_with_fields(title: &str, link: &str, author: &str, body: &str, guid: &str) -> String {
1979        format!(
1980            r#"<?xml version="1.0"?><rss version="2.0" xmlns:dc="http://purl.org/dc/elements/1.1/"><channel>
1981<title>{title}</title><link>https://example.com/{link}</link>
1982<item><title>{title}</title><link>https://example.com/{link}</link><guid>{guid}</guid>
1983<dc:creator>{author}</dc:creator><description><![CDATA[{body}]]></description></item>
1984</channel></rss>"#
1985        )
1986    }
1987
1988    #[test]
1989    fn bound_text_cuts_on_a_character_boundary() {
1990        // 'é' is two bytes, so an odd limit lands mid-character.
1991        let cut = bound_text("é".repeat(100), 51);
1992        assert!(cut.len() <= 51, "not bounded: {} bytes", cut.len());
1993        assert_eq!(cut, "é".repeat(25), "cut too short or mid-character");
1994        assert_eq!(
1995            bound_text("short".into(), 51),
1996            "short",
1997            "a short value changed"
1998        );
1999    }
2000
2001    #[test]
2002    fn rendered_content_fits_even_when_rendering_grows_it() {
2003        // Escaping turns each `&` into `&amp;` — five bytes from one.
2004        let out = plain_text_to_html_bounded(&"&".repeat(1_000), 100);
2005        assert!(
2006            out.len() <= 100,
2007            "escaped output not bounded: {} bytes",
2008            out.len()
2009        );
2010        assert!(!out.is_empty(), "bounded to nothing");
2011        assert_eq!(
2012            out.matches("&amp;").count() * 5,
2013            out.len(),
2014            "cut mid-entity: {out}"
2015        );
2016    }
2017
2018    #[test]
2019    fn an_rss_items_text_fields_are_bounded() {
2020        let big = "x".repeat(100_000);
2021        let xml = rss_with_fields(&big, &big, &big, "body", "id-1");
2022        let parsed = parse_feed(xml.as_bytes()).unwrap();
2023        let e = normalize_entry(&parsed.entries[0]);
2024        assert!(e.title.as_ref().unwrap().len() <= MAX_TITLE_BYTES, "title");
2025        assert!(e.url.as_ref().unwrap().len() <= MAX_URL_BYTES, "url");
2026        assert!(
2027            e.author.as_ref().unwrap().len() <= MAX_AUTHOR_BYTES,
2028            "author"
2029        );
2030        let (title, site) = feed_metadata(&parsed);
2031        assert!(title.unwrap().len() <= MAX_TITLE_BYTES, "feed title");
2032        assert!(site.unwrap().len() <= MAX_URL_BYTES, "feed site url");
2033    }
2034
2035    #[test]
2036    fn an_rss_body_is_bounded_and_still_well_formed() {
2037        // Long enough to need cutting, with markup straddling the cut.
2038        let body = format!("<p>{}<b>tail</b></p>", "a".repeat(MAX_CONTENT_HTML_BYTES));
2039        let xml = rss_with_fields("t", "l", "a", &body, "id-2");
2040        let parsed = parse_feed(xml.as_bytes()).unwrap();
2041        let html = normalize_entry(&parsed.entries[0]).content_html.unwrap();
2042        assert!(
2043            html.len() <= MAX_CONTENT_HTML_BYTES,
2044            "body not bounded: {}",
2045            html.len()
2046        );
2047        assert_eq!(
2048            sanitize_html(&html),
2049            html,
2050            "the stored body is not well-formed sanitized HTML"
2051        );
2052    }
2053
2054    /// Review of #224: cutting the INPUT first threw away content that would
2055    /// have fit. A large inline `data:` image, which ammonia strips anyway,
2056    /// ahead of the article left the cut ending inside the image tag, and the
2057    /// article was stored as an empty body.
2058    #[test]
2059    fn content_the_sanitizer_strips_does_not_count_against_the_bound() {
2060        const MAX: usize = 64 * 1024;
2061        let body = format!(
2062            r#"<p><img src="data:image/png;base64,{}"></p><p>the article</p>"#,
2063            "A".repeat(MAX + 16 * 1024)
2064        );
2065        let html = sanitize_html_bounded(&body, MAX);
2066        assert!(
2067            html.contains("the article"),
2068            "the article was cut away: {} bytes kept",
2069            html.len()
2070        );
2071        assert!(html.len() <= MAX);
2072    }
2073
2074    /// Review of #224 (second round): the re-cut loop shrank its input by a
2075    /// few bytes a round when the bytes beyond the cut were ones the sanitizer
2076    /// strips anyway — measured at ~32 bytes/round, hours for one entry, on the
2077    /// poller's async task. The number of renders must be bounded.
2078    /// Review of #224: a cut into bytes the sanitizer strips anyway (here an
2079    /// unterminated comment) crept a few bytes a round, for hours. Bounding
2080    /// works on the SANITIZED output now, so stripped input costs nothing.
2081    #[test]
2082    fn content_cut_beside_stripped_bytes_still_keeps_what_fits() {
2083        const MAX: usize = 64 * 1024;
2084        let body = format!("<p>{}</p><!--{}", "a".repeat(MAX + 8), "x".repeat(16 * MAX));
2085        let html = sanitize_html_bounded(&body, MAX);
2086        assert!(html.len() <= MAX);
2087        assert!(
2088            html.len() > MAX - MAX / 16,
2089            "kept far less than fits: {}",
2090            html.len()
2091        );
2092    }
2093
2094    /// Also from that review: a cut landing inside a stripped prefix stored an
2095    /// empty body when an article that fits followed it.
2096    /// Also from review: a cut landing inside a stripped prefix stored an
2097    /// empty body when an article that fits followed it.
2098    #[test]
2099    fn a_stripped_prefix_does_not_leave_an_empty_body() {
2100        const MAX: usize = 64 * 1024;
2101        let body = format!(
2102            r#"<p><img src="data:image/png;base64,{}"></p><p>{}</p>"#,
2103            "A".repeat(3 * MAX),
2104            "&".repeat(MAX)
2105        );
2106        let html = sanitize_html_bounded(&body, MAX);
2107        assert!(html.len() <= MAX);
2108        assert!(
2109            html.len() > MAX - MAX / 16,
2110            "nothing like what fits was kept: {} bytes",
2111            html.len()
2112        );
2113    }
2114
2115    /// Third review: a multi-byte character where the search's stale probe
2116    /// landed ended it early, keeping 0 bytes where ~2 MiB fit.
2117    #[test]
2118    fn a_stripped_multibyte_prefix_does_not_leave_an_empty_body() {
2119        const MAX: usize = 64 * 1024;
2120        let body = format!(
2121            "<!--{}--><p>{}</p>",
2122            "漢".repeat(MAX),
2123            "&".repeat(MAX * 3 / 10)
2124        );
2125        let html = sanitize_html_bounded(&body, MAX);
2126        assert!(html.len() <= MAX);
2127        assert!(
2128            html.len() > MAX - MAX / 16,
2129            "kept {} of ~{MAX} that fits",
2130            html.len()
2131        );
2132    }
2133
2134    /// Fourth review of #224: a closing tag is longer than the tag it closes,
2135    /// so a cut through nested markup grew past the bound on re-sanitizing,
2136    /// and the widening margin jumped straight to a cut of 0 — an empty body
2137    /// where nearly all of it fit.
2138    #[test]
2139    fn nested_markup_is_cut_not_emptied() {
2140        const MAX: usize = 64 * 1024;
2141        let html = sanitize_html_bounded(&"<span>".repeat(16_384), MAX);
2142        assert!(html.len() <= MAX);
2143        assert!(
2144            html.len() > MAX / 3,
2145            "kept {} of ~{MAX} that fits",
2146            html.len()
2147        );
2148
2149        let mixed = format!("{}{}", "t".repeat(MAX * 3 / 4), "<span>".repeat(MAX / 8));
2150        let html = sanitize_html_bounded(&mixed, MAX);
2151        assert!(html.len() <= MAX);
2152        assert!(
2153            html.len() > MAX - MAX / 16,
2154            "kept {} of ~{MAX} that fits",
2155            html.len()
2156        );
2157    }
2158
2159    /// The plain-text bound is exact: escaping is linear, so the longest
2160    /// fitting prefix is found in one pass, never by search.
2161    #[test]
2162    fn the_plain_text_bound_is_exact() {
2163        const MAX: usize = 64 * 1024;
2164        let raw = format!("{}{}", "漢".repeat(MAX / 4), "&".repeat(MAX / 10));
2165        let out = plain_text_to_html_bounded(&raw, MAX);
2166        assert!(out.len() <= MAX);
2167        // The next character would not have fitted: '&' escapes to 5 bytes.
2168        assert!(out.len() > MAX - 5, "kept {} of {MAX}", out.len());
2169        assert!(
2170            raw.starts_with(&out.replace("&amp;", "&")),
2171            "not a prefix of the input"
2172        );
2173    }
2174
2175    #[test]
2176    fn an_overlong_rss_guid_becomes_a_stable_short_one() {
2177        let long = "g".repeat(10_000);
2178        let xml = rss_with_fields("t", "l", "a", "b", &long);
2179        let parsed = parse_feed(xml.as_bytes()).unwrap();
2180        let a = normalize_entry(&parsed.entries[0]).guid;
2181        let b = normalize_entry(&parsed.entries[0]).guid;
2182        assert!(a.len() <= MAX_GUID_BYTES, "guid not bounded: {}", a.len());
2183        assert_eq!(
2184            a, b,
2185            "the stand-in is not stable, so the entry would duplicate"
2186        );
2187        let other = rss_with_fields("t", "l", "a", "b", &format!("{long}h"));
2188        let other = parse_feed(other.as_bytes()).unwrap();
2189        assert_ne!(
2190            a,
2191            normalize_entry(&other.entries[0]).guid,
2192            "two ids collapsed into one"
2193        );
2194    }
2195
2196    #[test]
2197    fn an_ordinary_long_article_is_untouched() {
2198        // The largest body production held on 2026-10-03 was 86,969 bytes.
2199        let body = format!("<p>{}</p>", "word ".repeat(18_000));
2200        let xml = rss_with_fields(
2201            "A normal title",
2202            "post",
2203            "Author",
2204            &body,
2205            "https://example.com/post",
2206        );
2207        let parsed = parse_feed(xml.as_bytes()).unwrap();
2208        let e = normalize_entry(&parsed.entries[0]);
2209        assert_eq!(e.content_html.unwrap(), sanitize_html(&body));
2210        assert_eq!(e.title.as_deref(), Some("A normal title"));
2211        assert_eq!(e.guid, "https://example.com/post");
2212    }
2213
2214    /// **A stated date in the future is discarded, not stored and not clamped.**
2215    ///
2216    /// `entry_time` was `e.published.or(e.updated)` with no ceiling, so an item
2217    /// dated in the year 2999 was stored verbatim and then became permanent:
2218    /// both retention sweeps test `COALESCE(published, fetched_at) < cutoff` and
2219    /// a future date is never less than either; the per-feed keep-set orders on
2220    /// the same expression `DESC`, where it is rank one forever; and every list
2221    /// view orders on `published DESC`, where it sits at the top. One item in one
2222    /// feed, there for good. A publisher with a broken clock does this by
2223    /// accident.
2224    ///
2225    /// Discarded rather than clamped to now, which is the rule `#186`
2226    /// established on the atproto side and the reasoning transfers exactly: the
2227    /// entry upsert refreshes `published` on every poll but stamps `fetched_at`
2228    /// once, so a value derived from the current clock is rewritten every cycle
2229    /// and the row can never age at all. Clamping moves the defect. Falling back
2230    /// to undated lets `fetched_at` date it, and that holds still.
2231    ///
2232    /// The ceiling is [`MAX_FUTURE_PUBLISHED_DAYS`] (two days), deliberately
2233    /// looser than the publication path's five-minute grace: a publication can
2234    /// fall back to its record key's TID, and a feed has no such fallback.
2235    #[test]
2236    fn a_future_dated_rss_item_is_stored_undated_rather_than_dated_in_2999() {
2237        let future = r#"<?xml version="1.0"?>
2238<rss version="2.0"><channel><title>Clock</title><link>https://clock.example/</link>
2239<item><title>From the future</title><link>https://clock.example/1</link>
2240<guid>https://clock.example/1</guid>
2241<pubDate>Sat, 01 Jan 2999 00:00:00 GMT</pubDate></item>
2242</channel></rss>"#;
2243        let parsed = parse_feed(future.as_bytes()).expect("should parse");
2244        // The fixture is only meaningful if feed-rs actually read the date.
2245        assert!(
2246            parsed.entries[0].published.is_some(),
2247            "the fixture's pubDate did not parse, so this test proves nothing",
2248        );
2249
2250        let e = normalize_entry(&parsed.entries[0]);
2251        assert_eq!(
2252            e.published, None,
2253            "a year-2999 date was stored, which makes the row unsweepable, \
2254             un-evictable and permanently first in the reading list",
2255        );
2256
2257        // And the other direction: an ordinary past date must survive, or this
2258        // would be satisfied by discarding every date.
2259        let past = future.replace("01 Jan 2999", "01 Jan 2020");
2260        let parsed = parse_feed(past.as_bytes()).expect("should parse");
2261        let e = normalize_entry(&parsed.entries[0]);
2262        assert!(
2263            e.published
2264                .as_deref()
2265                .is_some_and(|p| p.starts_with("2020")),
2266            "an ordinary past date was discarded: {:?}",
2267            e.published,
2268        );
2269
2270        // **A merely MISLABELLED date must survive.** The ordinary cause of a
2271        // future `pubDate` is a local time stamped `+0000` — up to 14 hours out,
2272        // not a clock a few minutes fast. Refusing those would leave real
2273        // articles undated and dateless on screen, which is why the bound is two
2274        // days rather than the publication path's five-minute skew grace.
2275        let soon = (Utc::now() + chrono::Duration::hours(14)).to_rfc2822();
2276        let near = future.replace("Sat, 01 Jan 2999 00:00:00 GMT", &soon);
2277        let parsed = parse_feed(near.as_bytes()).expect("should parse");
2278        assert!(
2279            parsed.entries[0].published.is_some(),
2280            "the mislabelled-date fixture did not parse",
2281        );
2282        let e = normalize_entry(&parsed.entries[0]);
2283        assert!(
2284            e.published.is_some(),
2285            "a date 14 hours ahead — the widest real UTC offset — was refused, \
2286             so a timezone-mislabelled article loses its date entirely",
2287        );
2288
2289        // And the bound still bounds: a month out is refused.
2290        let far = (Utc::now() + chrono::Duration::days(30)).to_rfc2822();
2291        let month = future.replace("Sat, 01 Jan 2999 00:00:00 GMT", &far);
2292        let parsed = parse_feed(month.as_bytes()).expect("should parse");
2293        let e = normalize_entry(&parsed.entries[0]);
2294        assert_eq!(
2295            e.published, None,
2296            "a date a month ahead was kept, so the row leads the list for a month",
2297        );
2298
2299        // **Each candidate is judged separately, not the winner of `or`.**
2300        //
2301        // Atom specifically: for RSS 2 `feed-rs` copies `published` into
2302        // `updated` when `updated` is absent, so the second candidate holds the
2303        // same bogus value and the fall-through cannot help. Only a format
2304        // carrying two independent dates exercises this.
2305        let both = r#"<?xml version="1.0"?>
2306<feed xmlns="http://www.w3.org/2005/Atom"><title>Clock</title>
2307<entry><title>Mixed</title><id>https://clock.example/2</id>
2308<link href="https://clock.example/2"/>
2309<published>2999-01-01T00:00:00Z</published>
2310<updated>2020-06-01T00:00:00Z</updated></entry></feed>"#;
2311        let parsed = parse_feed(both.as_bytes()).expect("should parse");
2312        assert!(
2313            parsed.entries[0].published.is_some() && parsed.entries[0].updated.is_some(),
2314            "the fixture needs BOTH dates parsed for this case to mean anything",
2315        );
2316        let e = normalize_entry(&parsed.entries[0]);
2317        assert!(
2318            e.published
2319                .as_deref()
2320                .is_some_and(|p| p.starts_with("2020")),
2321            "a credible `updated` was discarded along with a bogus `published`, \
2322             leaving the entry undated: {:?}",
2323            e.published,
2324        );
2325    }
2326
2327    /// Parse a static RSS sample through feed-rs + our normalize/sanitize path
2328    /// (no network) and assert the entries come out sanitized and well-shaped.
2329    #[test]
2330    fn rss_parses_and_sanitizes() {
2331        let parsed = parse_feed(RSS_SAMPLE.as_bytes()).expect("RSS should parse");
2332        assert_eq!(
2333            parsed.title.as_ref().map(text_plain).as_deref(),
2334            Some("Example RSS Feed")
2335        );
2336        assert_eq!(parsed.entries.len(), 2);
2337
2338        let (title, site) = feed_metadata(&parsed);
2339        assert_eq!(title.as_deref(), Some("Example RSS Feed"));
2340        assert_eq!(site.as_deref(), Some("https://example.com/"));
2341
2342        let e0 = normalize_entry(&parsed.entries[0]);
2343        assert_eq!(e0.guid, "https://example.com/first");
2344        assert_eq!(e0.title.as_deref(), Some("First post"));
2345        assert_eq!(e0.url.as_deref(), Some("https://example.com/first"));
2346        assert!(e0.published.is_some());
2347        let html0 = e0.content_html.expect("content present");
2348        // Sanitized: benign markup kept, script + onerror stripped.
2349        assert!(html0.contains("Hello"));
2350        assert!(html0.contains("<b>world</b>") || html0.contains("<b>"));
2351        assert!(!html0.to_ascii_lowercase().contains("<script"));
2352        assert!(!html0.to_ascii_lowercase().contains("onerror"));
2353        assert!(!html0.to_ascii_lowercase().contains("alert"));
2354
2355        // Second entry: javascript: URL scrubbed, safe link kept.
2356        let e1 = normalize_entry(&parsed.entries[1]);
2357        assert_eq!(e1.guid, "guid-second");
2358        let html1 = e1.content_html.expect("content present");
2359        assert!(!html1.to_ascii_lowercase().contains("javascript:"));
2360        assert!(html1.contains("https://ok.example/"));
2361    }
2362
2363    /// Same, for an Atom sample: alternate link is the site URL, dangerous
2364    /// elements are stripped from entry content.
2365    /// **`rel="alternate"` is preferred over a `rel="self"` listed FIRST.** In
2366    /// `ATOM_SAMPLE` the alternate link is already first, so "prefer alternate"
2367    /// and "take the first link" were indistinguishable; the selection could
2368    /// be replaced by `links.first()` with the suite green. A feed listing
2369    /// `self` first — very common in Atom — would store the feed XML URL as
2370    /// the subscription's `siteUrl`, published to the reader's PDS.
2371    #[test]
2372    fn atom_prefers_alternate_over_a_self_link_listed_first() {
2373        let parsed = parse_feed(ATOM_SELF_FIRST.as_bytes()).expect("Atom should parse");
2374        let (title, site) = feed_metadata(&parsed);
2375        assert_eq!(title.as_deref(), Some("Example Atom Feed"));
2376        // alternate link preferred over rel="self".
2377        assert_eq!(site.as_deref(), Some("https://atom.example.com/"));
2378    }
2379
2380    #[test]
2381    fn atom_parses_and_sanitizes() {
2382        let parsed = parse_feed(ATOM_SAMPLE.as_bytes()).expect("Atom should parse");
2383        let (title, site) = feed_metadata(&parsed);
2384        assert_eq!(title.as_deref(), Some("Example Atom Feed"));
2385        // alternate link preferred over rel="self".
2386        assert_eq!(site.as_deref(), Some("https://atom.example.com/"));
2387
2388        assert_eq!(parsed.entries.len(), 1);
2389        let e = normalize_entry(&parsed.entries[0]);
2390        assert_eq!(e.guid, "urn:uuid:entry-1");
2391        assert_eq!(e.title.as_deref(), Some("Atom entry"));
2392        assert_eq!(e.author.as_deref(), Some("Bob"));
2393        assert_eq!(e.url.as_deref(), Some("https://atom.example.com/a"));
2394        let html = e.content_html.expect("content present");
2395        assert!(html.contains("Safe"));
2396        assert!(!html.to_ascii_lowercase().contains("<script"));
2397        assert!(!html.to_ascii_lowercase().contains("<iframe"));
2398    }
2399
2400    /// **The rkey obeys all of atproto's record-key rules, not just the
2401    /// charset.** `.` and `..` are reserved and the length is 1..=512; the
2402    /// charset alone admitted both and a 10 000-character key, into a UNIQUE
2403    /// column and the user's public PDS. The rule is the one the repo's own
2404    /// TID tests already state.
2405    #[test]
2406    fn an_rkey_must_obey_atprotos_length_and_dot_rules() {
2407        let uri = |rkey: &str| {
2408            format!("at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/{rkey}")
2409        };
2410        assert!(
2411            !is_storable_feed_url(&uri("."), true),
2412            "`.` is a reserved rkey"
2413        );
2414        assert!(
2415            !is_storable_feed_url(&uri(".."), true),
2416            "`..` is a reserved rkey"
2417        );
2418        assert!(
2419            !is_storable_feed_url(&uri(&"a".repeat(513)), true),
2420            "an rkey over 512 bytes was accepted"
2421        );
2422        assert!(
2423            is_storable_feed_url(&uri(&"a".repeat(512)), true),
2424            "an rkey of exactly 512 bytes is valid"
2425        );
2426        assert!(is_storable_feed_url(&uri("3lab2c4d5e6f7g8h"), true));
2427    }
2428
2429    /// **Every `KNOWN_PROVIDERS` row is pinned by its own reason.**
2430    ///
2431    /// The provider test used URLs that the generic heuristics catch on their
2432    /// own — Substack's `/feed/private/` is also a path marker, Patreon's
2433    /// `?auth=…` an opaque secret key — and never asserted the reason. With
2434    /// 17 of 18 rows deleted, the suite stayed green. The provider layer runs
2435    /// FIRST so its specific reason wins; asserting the reason pins each row
2436    /// even where a generic rule would still refuse the URL. Values are kept
2437    /// short and plain so the generic query rule (`value_is_opaque`) does not
2438    /// fire — most of these are refused by the provider row alone.
2439    #[test]
2440    fn each_known_provider_is_caught_by_its_own_row() {
2441        for (url, reason) in [
2442            (
2443                "https://author.substack.com/feed/private/x",
2444                "Substack private feed",
2445            ),
2446            (
2447                "https://www.patreon.com/rss/creator?auth=ab",
2448                "Patreon member feed",
2449            ),
2450            ("https://blog.ghost.io/rss/?uuid=x", "Ghost members feed"),
2451            (
2452                "https://buttondown.email/me/rss?token=x",
2453                "Buttondown premium feed",
2454            ),
2455            (
2456                "https://buttondown.com/me/rss?token=x",
2457                "Buttondown premium feed",
2458            ),
2459            (
2460                "https://rss.beehiiv.com/feeds/x.xml?token=x",
2461                "Beehiiv premium feed",
2462            ),
2463            (
2464                "https://example.memberful.com/feed",
2465                "Memberful members feed",
2466            ),
2467            ("https://example.pico.link/feed", "Pico member feed"),
2468            ("https://steadyhq.com/rss/example", "Steady member feed"),
2469            (
2470                "https://example.supercast.com/feed",
2471                "Supercast private podcast",
2472            ),
2473            (
2474                "https://example.supercast.tech/feed",
2475                "Supercast private podcast",
2476            ),
2477            (
2478                "https://example.supportingcast.fm/feed",
2479                "Supporting Cast private podcast",
2480            ),
2481            (
2482                "https://feeds.redcircle.com/x?private=1",
2483                "RedCircle private podcast",
2484            ),
2485            (
2486                "https://feeds.megaphone.fm/x?token=x",
2487                "Megaphone private podcast",
2488            ),
2489            (
2490                "https://feeds.acast.com/public/shows/x?token=x",
2491                "Acast+ private podcast",
2492            ),
2493            (
2494                "https://omny.fm/shows/x/playlists/podcast.rss?token=x",
2495                "Omny private podcast",
2496            ),
2497            (
2498                "https://podcasts.apple.com/feed/x?token=x",
2499                "Apple subscriber podcast",
2500            ),
2501            (
2502                "https://anchor.spotify.com/s/x/podcast/rss?token=x",
2503                "Spotify subscriber podcast",
2504            ),
2505        ] {
2506            match classify_feed_privacy(url) {
2507                FeedPrivacy::Private(r) => {
2508                    assert_eq!(r, reason, "{url} was refused by another rule")
2509                }
2510                FeedPrivacy::Public => panic!("{url} was not refused at all"),
2511            }
2512        }
2513    }
2514
2515    /// **Plain text is escaped, not sanitised.** `ammonia::clean` parses its
2516    /// input as markup, so a `<` in prose swallows everything after it:
2517    /// measured in this tree, `"if x<y then z"` becomes `"if x"`. That is the
2518    /// right function for an RSS body (which IS markup) and exactly the wrong
2519    /// one for a field the lexicon defines as plain text — it silently deletes
2520    /// the reader's content.
2521    #[test]
2522    fn plain_text_is_escaped_rather_than_swallowed() {
2523        assert_eq!(
2524            plain_text_to_html("Vec<String> is a type"),
2525            "Vec&lt;String&gt; is a type"
2526        );
2527        assert_eq!(plain_text_to_html("if x<y then z"), "if x&lt;y then z");
2528        assert_eq!(plain_text_to_html("a & b"), "a &amp; b");
2529        // Line structure survives into a field rendered as HTML.
2530        assert_eq!(plain_text_to_html("one\ntwo"), "one<br>two");
2531        // And it is still safe: the escaping happens before any markup is added.
2532        let hostile = plain_text_to_html("<script>alert(1)</script>");
2533        assert!(!hostile.contains("<script"), "{hostile}");
2534    }
2535
2536    #[test]
2537    fn discover_finds_rss_link() {
2538        let html = r#"<!doctype html><html><head>
2539            <title>Blog</title>
2540            <link rel="stylesheet" href="/style.css">
2541            <link rel="alternate" type="application/rss+xml" title="RSS" href="/feed.xml">
2542        </head><body>hi</body></html>"#;
2543        let base = Url::parse("https://blog.example.com/").unwrap();
2544        let found = discover_feed(html, Some(&base)).expect("should discover feed");
2545        assert_eq!(found.as_str(), "https://blog.example.com/feed.xml");
2546    }
2547
2548    #[test]
2549    fn discover_finds_atom_absolute_link() {
2550        let html = r#"<head><link rel="alternate" type="application/atom+xml" href="https://x.example/atom"></head>"#;
2551        let found = discover_feed(html, None).expect("should discover absolute feed");
2552        assert_eq!(found.as_str(), "https://x.example/atom");
2553    }
2554
2555    /// **Autodiscovery only ever yields an http(s) URL.**
2556    ///
2557    /// The href is publisher-controlled and `Url::parse` accepts any scheme, so
2558    /// a page could hand the add path an `at://` publication URI (or anything
2559    /// else) that the user never typed — and the add path's input gate has
2560    /// already run by then. A non-http(s) alternate is skipped, not returned,
2561    /// so a later real feed link still wins.
2562    #[test]
2563    fn discover_skips_a_non_http_alternate() {
2564        let at_link = r#"<link rel="alternate" type="application/rss+xml" href="at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h">"#;
2565        assert!(
2566            discover_feed(&format!("<head>{at_link}</head>"), None).is_none(),
2567            "an at:// alternate was handed back as a feed URL"
2568        );
2569        let ftp_link =
2570            r#"<link rel="alternate" type="application/atom+xml" href="ftp://x.example/atom">"#;
2571        assert!(discover_feed(&format!("<head>{ftp_link}</head>"), None).is_none());
2572
2573        let real =
2574            r#"<link rel="alternate" type="application/atom+xml" href="https://x.example/atom">"#;
2575        let found = discover_feed(&format!("<head>{at_link}{real}</head>"), None)
2576            .expect("the http(s) link after a skipped one must still be found");
2577        assert_eq!(found.as_str(), "https://x.example/atom");
2578    }
2579
2580    #[test]
2581    fn discover_returns_none_without_feed_link() {
2582        let html =
2583            r#"<head><link rel="stylesheet" href="/s.css"><link rel="icon" href="/f.ico"></head>"#;
2584        assert!(discover_feed(html, None).is_none());
2585    }
2586
2587    #[test]
2588    fn synthetic_guid_is_stable_and_dedups() {
2589        // An item with neither guid nor link: `parse_feed` leaves its id empty
2590        // (feed-rs's own fallback is a random UUID). Clearing id and links
2591        // anyway keeps this a test of *our* synthetic fallback alone.
2592        let xml = r#"<?xml version="1.0"?><rss version="2.0"><channel>
2593            <title>t</title>
2594            <item><title>only a title</title><description>body</description></item>
2595        </channel></rss>"#;
2596        let mut parsed = parse_feed(xml.as_bytes()).expect("parse");
2597        parsed.entries[0].id.clear();
2598        parsed.entries[0].links.clear();
2599        let g1 = normalize_entry(&parsed.entries[0]).guid;
2600        let g2 = normalize_entry(&parsed.entries[0]).guid;
2601        assert_eq!(g1, g2);
2602        assert!(g1.starts_with("featherreader:synthetic:"));
2603    }
2604
2605    // ---- feed-rs 3.0: entry identity and links held to their 2.4 values ----
2606    //
2607    // The guid is the dedup key under `UNIQUE (feed_id, guid)`. If a parser
2608    // upgrade changes it, every reader gets every live entry of that feed a
2609    // second time. The pinned ids below are what feed-rs 2.4.0 generated for
2610    // these exact bytes.
2611
2612    /// An id-less item with an ordinary link keeps the id 2.4 generated.
2613    #[test]
2614    fn a_generated_entry_id_is_the_one_feed_rs_2_4_produced() {
2615        let xml = r#"<?xml version="1.0"?><rss version="2.0"><channel><title>t</title><item><title>No guid here</title><link>https://n.example/post</link></item></channel></rss>"#;
2616        let e = normalize_entry(&parse_feed(xml.as_bytes()).expect("parse").entries[0]);
2617        assert_eq!(e.guid, "5813b43a0512aaef2750311bf4d978a");
2618        assert_eq!(e.url.as_deref(), Some("https://n.example/post"));
2619    }
2620
2621    /// **feed-rs 3.0 adds `<comments>` and `wfw:commentRss` to `entry.links`.**
2622    /// Listed before `<link>`, the comments URL became the entry's permalink,
2623    /// and for an id-less item it was hashed into the generated id, so the
2624    /// same item got a new guid after the upgrade: a duplicate in every
2625    /// reader's list.
2626    #[test]
2627    fn a_comments_link_listed_first_is_neither_the_permalink_nor_the_id() {
2628        let xml = r#"<?xml version="1.0"?>
2629<rss version="2.0" xmlns:wfw="http://wellformedweb.org/CommentAPI/"><channel><title>c</title><link>https://c.example/</link>
2630<item><title>Comments listed first, no guid</title><comments>https://c.example/1#comments</comments><link>https://c.example/1</link><wfw:commentRss>https://c.example/1/feed</wfw:commentRss></item>
2631<item><title>Comments first, guid present</title><comments>https://c.example/6#comments</comments><link>https://c.example/6</link><guid>c6</guid></item>
2632</channel></rss>"#;
2633        let parsed = parse_feed(xml.as_bytes()).expect("parse");
2634        let e0 = normalize_entry(&parsed.entries[0]);
2635        assert_eq!(
2636            e0.guid, "cd0017f2746ee934cf45ca0125100796",
2637            "the generated id moved, so this item would be stored twice"
2638        );
2639        assert_eq!(e0.url.as_deref(), Some("https://c.example/1"));
2640        let e1 = normalize_entry(&parsed.entries[1]);
2641        assert_eq!(e1.guid, "c6");
2642        assert_eq!(e1.url.as_deref(), Some("https://c.example/6"));
2643    }
2644
2645    /// The same for Atom, where 3.0 also reads `wfw:commentRss`.
2646    #[test]
2647    fn an_atom_comment_feed_link_is_neither_the_permalink_nor_the_id() {
2648        let xml = r#"<?xml version="1.0" encoding="utf-8"?>
2649<feed xmlns="http://www.w3.org/2005/Atom" xmlns:wfw="http://wellformedweb.org/CommentAPI/"><title>a</title>
2650<entry><title>commentRss before link, no id</title><wfw:commentRss>https://a.example/1/feed</wfw:commentRss><link href="https://a.example/1"/><updated>2020-06-01T00:00:00Z</updated></entry>
2651</feed>"#;
2652        let e = normalize_entry(&parse_feed(xml.as_bytes()).expect("parse").entries[0]);
2653        assert_eq!(e.guid, "5148a3d11836efc42f82a7b25a38383d");
2654        assert_eq!(e.url.as_deref(), Some("https://a.example/1"));
2655    }
2656
2657    /// **An item with no guid and no permalink gets a guid that holds still.**
2658    /// feed-rs's default generator falls back to a random UUID when an entry
2659    /// has no link, so such an item was a new row on every poll (in 2.4 as in
2660    /// 3.0). It now falls through to [`stable_guid`]. A comments link is not a
2661    /// permalink, so an item carrying only one is in the same position — and
2662    /// is not given the comments page as its URL.
2663    #[test]
2664    fn an_item_without_guid_or_permalink_dedups_across_polls() {
2665        let xml = r#"<?xml version="1.0"?><rss version="2.0"><channel><title>t</title>
2666<item><title>only a title</title><description>body</description></item>
2667<item><title>Only a comments link</title><comments>https://c.example/2#comments</comments></item>
2668</channel></rss>"#;
2669        let first = parse_feed(xml.as_bytes()).expect("parse");
2670        let second = parse_feed(xml.as_bytes()).expect("parse");
2671        for i in 0..2 {
2672            let a = normalize_entry(&first.entries[i]);
2673            let b = normalize_entry(&second.entries[i]);
2674            assert_eq!(
2675                a.guid, b.guid,
2676                "entry {i}'s guid changed between two parses"
2677            );
2678            assert!(a.guid.starts_with("featherreader:synthetic:"), "{}", a.guid);
2679            assert_eq!(a.url, None, "entry {i}");
2680        }
2681    }
2682
2683    /// **The author is the person's name.** feed-rs 2.4 named every RSS
2684    /// `<author>` "author" (the element name; the text went to `email`) and an
2685    /// Atom author with an empty `<name>` "unknown", and FeatherReader stored
2686    /// those words as the byline. 3.0 splits name from address but leaves the
2687    /// `(Name)` of the RSS `address (Name)` form in its parentheses.
2688    #[test]
2689    fn the_author_is_the_name_not_the_element_or_the_address() {
2690        let xml = r#"<?xml version="1.0"?>
2691<rss version="2.0" xmlns:dc="http://purl.org/dc/elements/1.1/"><channel><title>p</title>
2692<item><title>a1</title><guid>a1</guid><author>alice@example.com (Alice Example)</author></item>
2693<item><title>a2</title><guid>a2</guid><author>bob@example.com</author></item>
2694<item><title>a3</title><guid>a3</guid><author>Carol</author></item>
2695<item><title>a4</title><guid>a4</guid><dc:creator>Dave &lt;dave@example.com&gt;</dc:creator></item>
2696<item><title>a5</title><guid>a5</guid><author>erin@example.com ()</author></item>
2697</channel></rss>"#;
2698        let parsed = parse_feed(xml.as_bytes()).expect("parse");
2699        let authors: Vec<Option<String>> = parsed
2700            .entries
2701            .iter()
2702            .map(|e| normalize_entry(e).author)
2703            .collect();
2704        assert_eq!(
2705            authors,
2706            vec![
2707                Some("Alice Example".to_string()),
2708                None,
2709                Some("Carol".to_string()),
2710                Some("Dave".to_string()),
2711                None,
2712            ]
2713        );
2714
2715        let atom = r#"<?xml version="1.0" encoding="utf-8"?>
2716<feed xmlns="http://www.w3.org/2005/Atom"><title>a</title>
2717<entry><id>x1</id><title>t</title><author><name></name></author></entry>
2718<entry><id>x2</id><title>t</title><author><name>Bob</name></author></entry>
2719</feed>"#;
2720        let parsed = parse_feed(atom.as_bytes()).expect("parse");
2721        assert_eq!(normalize_entry(&parsed.entries[0]).author, None);
2722        assert_eq!(
2723            normalize_entry(&parsed.entries[1]).author.as_deref(),
2724            Some("Bob")
2725        );
2726    }
2727
2728    /// **The byline is the first author that HAS a name.** An item giving
2729    /// `<author>` as a bare address and the name in `<dc:creator>` has two
2730    /// people, and the first has no name; taking only the first lost the byline.
2731    #[test]
2732    fn the_byline_is_the_first_author_with_a_name() {
2733        let xml = r#"<?xml version="1.0"?>
2734<rss version="2.0" xmlns:dc="http://purl.org/dc/elements/1.1/"><channel><title>p</title>
2735<item><title>b1</title><guid>b1</guid><author>bob@example.com</author><dc:creator>Bob Smith</dc:creator></item>
2736<item><title>b2</title><guid>b2</guid><author>erin@example.com ()</author><dc:creator>Erin</dc:creator></item>
2737<item><title>b3</title><guid>b3</guid><author>only@example.com</author></item>
2738</channel></rss>"#;
2739        let parsed = parse_feed(xml.as_bytes()).expect("parse");
2740        let authors: Vec<Option<String>> = parsed
2741            .entries
2742            .iter()
2743            .map(|e| normalize_entry(e).author)
2744            .collect();
2745        assert_eq!(
2746            authors,
2747            vec![
2748                Some("Bob Smith".to_string()),
2749                Some("Erin".to_string()),
2750                None
2751            ]
2752        );
2753    }
2754
2755    /// Author strings that make feed-rs 3.0 panic: its name/address splitter
2756    /// (`parser/util/mod.rs`, `parse_person_name_email`) slices one byte either
2757    /// side of the address, which is not a character boundary when a multi-byte
2758    /// character touches it.
2759    const PANICKING_AUTHORS: [&str; 3] = [
2760        "jose@example.com(José)",
2761        "«zoe@example.com» Zoë Long Name Here",
2762        "Zoë Long Name «zoe@example.com»",
2763    ];
2764
2765    /// Every place feed-rs runs that splitter, as a whole document around `v`.
2766    fn documents_with_author(v: &str) -> Vec<(&'static str, String)> {
2767        vec![
2768            (
2769                "rss author",
2770                format!(
2771                    r#"<?xml version="1.0"?><rss version="2.0"><channel><title>t</title><item><title>x</title><guid>g</guid><author>{v}</author></item></channel></rss>"#
2772                ),
2773            ),
2774            (
2775                "rss dc:creator",
2776                format!(
2777                    r#"<?xml version="1.0"?><rss version="2.0" xmlns:dc="http://purl.org/dc/elements/1.1/"><channel><title>t</title><item><title>x</title><guid>g</guid><dc:creator>{v}</dc:creator></item></channel></rss>"#
2778                ),
2779            ),
2780            (
2781                "rss managingEditor",
2782                format!(
2783                    r#"<?xml version="1.0"?><rss version="2.0"><channel><title>t</title><managingEditor>{v}</managingEditor><item><title>x</title><guid>g</guid></item></channel></rss>"#
2784                ),
2785            ),
2786            (
2787                "rss webMaster",
2788                format!(
2789                    r#"<?xml version="1.0"?><rss version="2.0"><channel><title>t</title><webMaster>{v}</webMaster><item><title>x</title><guid>g</guid></item></channel></rss>"#
2790                ),
2791            ),
2792            (
2793                "rss1 dc:creator",
2794                format!(
2795                    r#"<?xml version="1.0"?><rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" xmlns="http://purl.org/rss/1.0/" xmlns:dc="http://purl.org/dc/elements/1.1/"><channel><title>t</title></channel><item><title>x</title><link>https://e.example/1</link><dc:creator>{v}</dc:creator></item></rdf:RDF>"#
2796                ),
2797            ),
2798            (
2799                "json author",
2800                format!(
2801                    r#"{{"version":"https://jsonfeed.org/version/1.1","title":"t","items":[{{"id":"1","content_text":"x","authors":[{{"name":"{v}"}}]}}]}}"#
2802                ),
2803            ),
2804        ]
2805    }
2806
2807    /// **A parser panic is a parse failure, not a crash.** feed-rs 3.0 panics
2808    /// on [`PANICKING_AUTHORS`]. Uncaught, the panic escaped `poll_feed`, killed
2809    /// the poll task after the scheduler had already moved `next_poll`, and
2810    /// recorded no failure — the feed stopped updating with nothing to say why.
2811    #[test]
2812    fn a_feed_rs_panic_is_returned_as_an_error() {
2813        for v in PANICKING_AUTHORS {
2814            for (slot, doc) in documents_with_author(v) {
2815                let err = match parse_feed(doc.as_bytes()) {
2816                    Ok(_) => {
2817                        panic!("{slot} {v:?}: parsed; the fixture no longer reaches the panic")
2818                    }
2819                    Err(e) => format!("{e:#}"),
2820                };
2821                assert!(err.contains("panicked"), "{slot} {v:?}: {err}");
2822            }
2823        }
2824        // The same slots with an ASCII neighbour parse normally.
2825        for (slot, doc) in documents_with_author("jose@example.com (Jose)") {
2826            assert!(parse_feed(doc.as_bytes()).is_ok(), "{slot}");
2827        }
2828    }
2829
2830    /// End to end: a poll of a feed that panics the parser reports
2831    /// `FailureKind::Parse`, so it is backed off and its `last_error` says why.
2832    #[tokio::test]
2833    async fn a_poll_of_a_feed_that_panics_the_parser_is_a_parse_failure() {
2834        let doc = &documents_with_author(PANICKING_AUTHORS[0])[0].1;
2835        let base = crate::net::tests::serve_body(doc.as_bytes().to_vec()).await;
2836        let port: u16 = base
2837            .trim_end_matches('/')
2838            .rsplit(':')
2839            .next()
2840            .unwrap()
2841            .parse()
2842            .unwrap();
2843        crate::net::test_host_override(
2844            "panicking-author.test",
2845            std::net::SocketAddr::from(([127, 0, 0, 1], port)),
2846        );
2847        let url = format!("http://panicking-author.test:{port}/feed.xml");
2848        let pool = crate::store::init_url("sqlite::memory:").await.unwrap();
2849        crate::store::upsert_feed(
2850            &pool,
2851            &crate::store::NewFeed {
2852                url: url.clone(),
2853                ..Default::default()
2854            },
2855        )
2856        .await
2857        .unwrap();
2858        let feed = crate::store::get_feed_by_url(&pool, &url)
2859            .await
2860            .unwrap()
2861            .unwrap();
2862        let client = build_client().unwrap();
2863        let outcome = poll_feed(&pool, &client, &feed, 0)
2864            .await
2865            .expect("a parser panic surfaced as a store error");
2866        match outcome {
2867            PollOutcome::Failed {
2868                kind: FailureKind::Parse,
2869                detail,
2870                ..
2871            } => assert!(detail.contains("panicked"), "{detail}"),
2872            other => panic!("expected a parse failure, got {other:?}"),
2873        }
2874    }
2875
2876    // ---- dedup-key hashes are fixed across toolchains -------------------
2877    //
2878    // `stable_guid` and `bound_guid` are stored dedup keys. They were built on
2879    // `std`'s `DefaultHasher`, whose algorithm the standard library documents
2880    // as unspecified and subject to change between releases; a toolchain bump
2881    // could have re-keyed every such entry and duplicated it. The values below
2882    // were produced by that `DefaultHasher` code, so they also prove the
2883    // replacement hashes identically today.
2884
2885    #[test]
2886    fn a_synthetic_guid_is_the_value_already_stored() {
2887        let xml = r#"<?xml version="1.0"?><rss version="2.0"><channel><title>t</title>
2888<item><title>only a title</title><description>body</description></item>
2889<item><description>no title either</description></item>
2890<item><title>Tïtle wíth ünïcode</title></item>
2891</channel></rss>"#;
2892        let parsed = parse_feed(xml.as_bytes()).expect("parse");
2893        let guids: Vec<String> = parsed
2894            .entries
2895            .iter()
2896            .map(|e| normalize_entry(e).guid)
2897            .collect();
2898        assert_eq!(
2899            guids,
2900            vec![
2901                "featherreader:synthetic:964034cf9a24b551".to_string(),
2902                "featherreader:synthetic:baa7c19198c76b30".into(),
2903                "featherreader:synthetic:34b72cc7fa0a59ac".into(),
2904            ]
2905        );
2906    }
2907
2908    #[test]
2909    fn a_long_guid_is_the_value_already_stored() {
2910        let a = bound_guid("g".repeat(MAX_GUID_BYTES + 1));
2911        let b = bound_guid(format!(
2912            "https://long.example/{}",
2913            "ü".repeat(MAX_GUID_BYTES)
2914        ));
2915        assert_eq!(
2916            (a.as_str(), b.as_str()),
2917            (
2918                "featherreader:long-guid:966404db20ef8f05",
2919                "featherreader:long-guid:a1d631602455ace4",
2920            )
2921        );
2922    }
2923
2924    #[test]
2925    fn entry_link_scheme_allowlist_neutralizes_javascript() {
2926        // An entry whose only link is a javascript: URL must yield no href.
2927        let xml = r#"<?xml version="1.0"?><rss version="2.0"><channel>
2928            <title>t</title>
2929            <item>
2930              <title>evil</title>
2931              <link>javascript:alert(document.domain)</link>
2932              <guid>evil-1</guid>
2933            </item>
2934        </channel></rss>"#;
2935        let parsed = parse_feed(xml.as_bytes()).expect("parse");
2936        let e = normalize_entry(&parsed.entries[0]);
2937        // url is dropped (not a safe http(s) link)…
2938        assert_eq!(e.url, None);
2939        // …but the entry still dedups (guid preserved from <guid>).
2940        assert_eq!(e.guid, "evil-1");
2941
2942        // A data: URL is likewise dropped.
2943        let xml2 = r#"<?xml version="1.0"?><rss version="2.0"><channel>
2944            <title>t</title>
2945            <item><title>d</title><link>data:text/html,<script>1</script></link><guid>d1</guid></item>
2946        </channel></rss>"#;
2947        let parsed2 = parse_feed(xml2.as_bytes()).expect("parse");
2948        let e2 = normalize_entry(&parsed2.entries[0]);
2949        assert_eq!(e2.url, None);
2950
2951        // A normal https link survives.
2952        let xml3 = r#"<?xml version="1.0"?><rss version="2.0"><channel>
2953            <title>t</title>
2954            <item><title>ok</title><link>https://ok.example/post</link><guid>ok1</guid></item>
2955        </channel></rss>"#;
2956        let parsed3 = parse_feed(xml3.as_bytes()).expect("parse");
2957        let e3 = normalize_entry(&parsed3.entries[0]);
2958        assert_eq!(e3.url.as_deref(), Some("https://ok.example/post"));
2959    }
2960
2961    #[test]
2962    fn classify_privacy_flags_secret_urls_across_providers() {
2963        // --- Known providers: newsletters ---
2964        // Substack private feed path.
2965        assert!(
2966            classify_feed_privacy("https://author.substack.com/feed/private/deadbeefcafe1234")
2967                .is_private()
2968        );
2969        // Patreon ?auth= member feed.
2970        assert!(classify_feed_privacy(
2971            "https://www.patreon.com/rss/author?auth=Zm9vYmFyc2VjcmV0dG9rZW4"
2972        )
2973        .is_private());
2974        // Ghost members feed via ?uuid=.
2975        assert!(classify_feed_privacy(
2976            "https://blog.ghost.io/rss/?uuid=1f2e3d4c-5b6a-7089-90ab-cdef01234567"
2977        )
2978        .is_private());
2979
2980        // --- Known providers: private podcasts ---
2981        // Supporting Cast tokened podcast feed.
2982        assert!(classify_feed_privacy(
2983            "https://feeds.supportingcast.fm/show/abcdef0123456789abcdef01"
2984        )
2985        .is_private());
2986        // Supercast private podcast (host alone is enough).
2987        assert!(classify_feed_privacy("https://feeds.supercast.com/12345/rss").is_private());
2988
2989        // --- Generic, provider-agnostic heuristic ---
2990        // Named credential query params with an opaque value.
2991        assert!(
2992            classify_feed_privacy("https://example.com/feed?token=Zm9vYmFyc2VjcmV0").is_private()
2993        );
2994        assert!(
2995            classify_feed_privacy("https://example.com/feed?key=Zm9vYmFyc2VjcmV0").is_private()
2996        );
2997        assert!(
2998            classify_feed_privacy("https://example.com/feed?secret=Zm9vYmFyc2VjcmV0").is_private()
2999        );
3000        // Userinfo credentials in the authority.
3001        assert!(classify_feed_privacy("https://user:pass@example.com/feed").is_private());
3002        // A `/private/` path segment on an unknown host.
3003        assert!(classify_feed_privacy("https://blog.example.com/private/rss").is_private());
3004        // `/members/` path convention.
3005        assert!(classify_feed_privacy("https://news.example.com/members/feed.xml").is_private());
3006        // A high-entropy opaque token embedded in the path with no telltale name.
3007        assert!(
3008            classify_feed_privacy("https://feeds.example.com/aB3xK9zQ7mP2rT5wL8nD4vF6")
3009                .is_private()
3010        );
3011        // A bare UUID path segment (many tokened feeds).
3012        assert!(classify_feed_privacy(
3013            "https://feeds.example.com/1f2e3d4c-5b6a-7089-90ab-cdef01234567"
3014        )
3015        .is_private());
3016    }
3017
3018    /// The dominant real-world private-podcast shape delivers the token as a
3019    /// FILENAME (`<token>.rss` / `<token>.xml`) or affixed inside a larger
3020    /// segment (`feed-<uuid>`). Named providers are caught by their host rule;
3021    /// these are UNKNOWN-provider CDNs that must still be caught by the generic
3022    /// backstop, so the secret is never fetched or stored.
3023    #[test]
3024    fn classify_privacy_catches_tokened_filenames_on_unknown_hosts() {
3025        // **A stem only the extension-strip branch can see.** Every other case
3026        // here is also caught by the sub-part scan (branch 3) or the UUID scan,
3027        // so deleting the strip left the suite green — `FEED_EXTENSIONS` was
3028        // effectively dead. This stem's `-`-separated parts are each too short
3029        // to look like a secret on their own, and the whole segment fails on
3030        // the `.` — only stripping `.rss` and re-testing the 26-char stem sees
3031        // it. That is exactly the shape a hyphen-bearing base64url token
3032        // filename takes.
3033        assert!(
3034            classify_feed_privacy("https://cdn.example/feeds/aB3xK9pQ-7mZ2vN8w-Qr5tYuW.rss")
3035                .is_private(),
3036            "a token stem visible only after stripping the extension was not caught"
3037        );
3038        // hex-32 token as an .xml filename.
3039        assert!(classify_feed_privacy(
3040            "https://cdn.somepod.io/f/a1b2c3d4e5f60718293a4b5c6d7e8f90.xml"
3041        )
3042        .is_private());
3043        // hex-32 token as a .rss filename on an unknown CDN.
3044        assert!(classify_feed_privacy(
3045            "https://dcs.megaphone.example/network/a1b2c3d4e5f60718293a4b5c6d7e8f90.rss"
3046        )
3047        .is_private());
3048        // UUID + .xml filename.
3049        assert!(classify_feed_privacy(
3050            "https://brandnew.example/feed/1f2e3d4c-5b6a-7089-90ab-cdef01234567.xml"
3051        )
3052        .is_private());
3053        // UUID affixed with a prefix (`feed-<uuid>`) — split can't see it, the
3054        // UUID-substring scan must.
3055        assert!(classify_feed_privacy(
3056            "https://x.example/feed-1f2e3d4c-5b6a-7089-90ab-cdef01234567"
3057        )
3058        .is_private());
3059        // UUID + .rss suffix.
3060        assert!(classify_feed_privacy(
3061            "https://x.example/1f2e3d4c-5b6a-7089-90ab-cdef01234567.rss"
3062        )
3063        .is_private());
3064        // hex-16 token as an .xml filename.
3065        assert!(classify_feed_privacy("https://x.example/feed/9f8e7d6c5b4a3928.xml").is_private());
3066        // A base64url token with `=` padding as a clean path segment.
3067        assert!(
3068            classify_feed_privacy("https://cdn.pod.io/f/YWJjZGVmZ2hpamtsbW5vcHFyc3R1dnc=")
3069                .is_private()
3070        );
3071    }
3072
3073    /// YouTube channel/playlist RSS feeds are FULLY PUBLIC (the id is a public
3074    /// handle, not a secret) and are the standard way to subscribe to a channel —
3075    /// they must NOT be false-blocked by the generic entropy heuristic.
3076    #[test]
3077    fn classify_privacy_allows_public_youtube_feeds() {
3078        assert_eq!(
3079            classify_feed_privacy(
3080                "https://www.youtube.com/feeds/videos.xml?channel_id=UC-lHJZR3Gqxm24_Vd_AJ5Yw"
3081            ),
3082            FeedPrivacy::Public
3083        );
3084        assert_eq!(
3085            classify_feed_privacy(
3086                "https://www.youtube.com/feeds/videos.xml?playlist_id=PLFgquLnL59alCl_2TQvOiD5Vgm1hCaGSI"
3087            ),
3088            FeedPrivacy::Public
3089        );
3090        // Bare host form too.
3091        assert_eq!(
3092            classify_feed_privacy(
3093                "https://youtube.com/feeds/videos.xml?channel_id=UC-lHJZR3Gqxm24_Vd_AJ5Yw"
3094            ),
3095            FeedPrivacy::Public
3096        );
3097        // The allowlist is narrow: a `token=` on the YouTube feeds path still
3098        // classifies private (can't smuggle a credential through the allowlist).
3099        assert!(classify_feed_privacy(
3100            "https://www.youtube.com/feeds/videos.xml?token=Zm9vYmFyc2VjcmV0dG9rZW4"
3101        )
3102        .is_private());
3103    }
3104
3105    #[test]
3106    fn classify_privacy_leaves_normal_public_feeds_public() {
3107        // Plain feed documents.
3108        assert_eq!(
3109            classify_feed_privacy("https://example.com/feed.xml"),
3110            FeedPrivacy::Public
3111        );
3112        assert_eq!(
3113            classify_feed_privacy("https://blog.example.com/rss"),
3114            FeedPrivacy::Public
3115        );
3116        assert_eq!(
3117            classify_feed_privacy("https://blog.example.com/rss.xml"),
3118            FeedPrivacy::Public
3119        );
3120        // A Substack PUBLIC feed (`/feed`, not `/feed/private/`) stays public.
3121        assert_eq!(
3122            classify_feed_privacy("https://author.substack.com/feed"),
3123            FeedPrivacy::Public
3124        );
3125        // A WordPress `/feed` endpoint.
3126        assert_eq!(
3127            classify_feed_privacy("https://wordpress.example.com/feed/"),
3128            FeedPrivacy::Public
3129        );
3130        // A plain Atom feed.
3131        assert_eq!(
3132            classify_feed_privacy("https://example.org/atom.xml"),
3133            FeedPrivacy::Public
3134        );
3135        // A long, hyphenated slug must NOT be mistaken for an embedded secret.
3136        assert_eq!(
3137            classify_feed_privacy("https://example.com/2026/07/my-first-long-blog-post-title/feed"),
3138            FeedPrivacy::Public
3139        );
3140        // A benign query key that merely contains "key" as a substring is fine.
3141        assert_eq!(
3142            classify_feed_privacy("https://example.com/feed?keyword=rust"),
3143            FeedPrivacy::Public
3144        );
3145        // A short, non-opaque value on a named key (e.g. an enum) is not a secret.
3146        assert_eq!(
3147            classify_feed_privacy("https://example.com/feed?p=2"),
3148            FeedPrivacy::Public
3149        );
3150        // An empty credential value is not a secret.
3151        assert_eq!(
3152            classify_feed_privacy("https://example.com/feed?token="),
3153            FeedPrivacy::Public
3154        );
3155        // A hyphenated slug ending in a feed extension must NOT be seen as a
3156        // tokened filename (the stem is short dictionary words, not a blob).
3157        assert_eq!(
3158            classify_feed_privacy("https://example.com/my-first-long-blog-post.xml"),
3159            FeedPrivacy::Public
3160        );
3161        // A short hex episode id in an .xml filename (< 16 chars) is not a secret.
3162        assert_eq!(
3163            classify_feed_privacy("https://example.com/episodes/ab12cd.xml"),
3164            FeedPrivacy::Public
3165        );
3166        // A dotted host-style filename slug stays public.
3167        assert_eq!(
3168            classify_feed_privacy("https://example.com/category/tech-news/feed.xml"),
3169            FeedPrivacy::Public
3170        );
3171        // Unparseable URL: treated as Public (add path rejects it downstream).
3172        assert_eq!(classify_feed_privacy("not a url"), FeedPrivacy::Public);
3173    }
3174
3175    /// **The detail is bounded where it is CONSTRUCTED, not only where it is
3176    /// stored.**
3177    ///
3178    /// Review found that widening `PollOutcome::Failed` with this field opened a
3179    /// second sink nobody looked at: `web.rs`'s `add_subscription` logs
3180    /// `?outcome` at INFO on a user-facing request path, so the whole
3181    /// untruncated anyhow chain — redirect-hop URLs, the SSRF guard's refusal
3182    /// text naming a resolved internal address — went to the access log.
3183    ///
3184    /// Bounding inside `bump_feed_errors` protected the database and nothing
3185    /// else. Bounding at construction protects every sink, including the ones
3186    /// added later.
3187    #[test]
3188    fn a_failure_detail_is_bounded_at_construction() {
3189        let huge = "x".repeat(10_000);
3190        let outcome = PollOutcome::Failed {
3191            backoff: BACKOFF_BASE,
3192            kind: FailureKind::Fetch,
3193            detail: failure_detail(&huge),
3194        };
3195        let PollOutcome::Failed { detail, .. } = &outcome else {
3196            panic!("wrong variant");
3197        };
3198        assert!(
3199            detail.chars().count() <= MAX_FAILURE_DETAIL_CHARS,
3200            "detail was {} chars",
3201            detail.chars().count(),
3202        );
3203        // And the Debug rendering — which is what actually reached the log — is
3204        // bounded with it.
3205        assert!(format!("{outcome:?}").len() < 1_000);
3206    }
3207
3208    /// **Every failure kind has its own label, and they round-trip.**
3209    ///
3210    /// Review found that collapsing all four `as_str` arms to `"fetch"` left
3211    /// the whole suite green: every test of these columns passed string
3212    /// literals, so nothing tied a variant to its label. A histogram whose
3213    /// buckets all say the same thing is worse than no histogram — it reports a
3214    /// single confident cause for four different failures.
3215    ///
3216    /// Asserted over `ALL` rather than a hand-written list, so adding a variant
3217    /// without a label fails here instead of silently sharing one.
3218    #[test]
3219    fn every_failure_kind_has_a_distinct_round_tripping_label() {
3220        let mut seen = std::collections::BTreeSet::new();
3221        for kind in FailureKind::ALL {
3222            let label = kind.as_str();
3223            assert!(
3224                seen.insert(label),
3225                "{label:?} is used by more than one FailureKind",
3226            );
3227            assert_eq!(
3228                FailureKind::parse(label),
3229                Some(kind),
3230                "{label:?} does not read back as the kind that wrote it",
3231            );
3232        }
3233        assert_eq!(seen.len(), FailureKind::ALL.len());
3234        // A label from a newer build is not attributed to a cause this one
3235        // knows — the `metrics::Backend::parse` contract.
3236        assert_eq!(FailureKind::parse("quota"), None);
3237    }
3238
3239    /// **Escalation reaches `settle_poll`.** `backoff_for` grows with the
3240    /// count and is tested alone; nothing asserted that the poll path passes
3241    /// the COUNT in. `backoff_for(1)` in its place left the whole suite green
3242    /// — a permanently dead feed retrying forever at the first-failure floor,
3243    /// which the comment on that line says must not happen.
3244    #[tokio::test]
3245    async fn backoff_escalates_with_consecutive_failures() -> anyhow::Result<()> {
3246        let pool = crate::store::init_url("sqlite::memory:").await?;
3247        let url = "https://dead.example/feed.xml";
3248        crate::store::upsert_feed(
3249            &pool,
3250            &crate::store::NewFeed {
3251                url: url.to_string(),
3252                ..Default::default()
3253            },
3254        )
3255        .await?;
3256        for _ in 0..5 {
3257            crate::store::bump_feed_errors(&pool, url, FailureKind::Fetch, "down").await?;
3258        }
3259        let before = chrono::Utc::now();
3260        settle_poll(
3261            &pool,
3262            url,
3263            &PollOutcome::Failed {
3264                backoff: Duration::from_secs(300),
3265                kind: FailureKind::Fetch,
3266                detail: "still down".to_string(),
3267            },
3268            Duration::from_secs(3600),
3269        )
3270        .await;
3271        let next: String = sqlx::query_scalar("SELECT next_poll FROM feeds WHERE url = ?1")
3272            .bind(url)
3273            .fetch_one(&pool)
3274            .await?;
3275        let next = chrono::DateTime::parse_from_rfc3339(&next)?.with_timezone(&chrono::Utc);
3276        let delay = (next - before).num_seconds();
3277        let expected = backoff_for(6).as_secs() as i64;
3278        assert!(
3279            (delay - expected).abs() <= 60,
3280            "sixth failure scheduled {delay}s out; escalation says {expected}s"
3281        );
3282        assert!(
3283            delay > backoff_for(1).as_secs() as i64 + 60,
3284            "the sixth failure landed on the first-failure floor"
3285        );
3286        Ok(())
3287    }
3288
3289    #[test]
3290    fn backoff_grows_and_is_capped() {
3291        assert_eq!(backoff_for(1), BACKOFF_BASE);
3292        assert!(backoff_for(2) > backoff_for(1));
3293        assert_eq!(backoff_for(100), BACKOFF_MAX);
3294    }
3295
3296    /// **Storable and pollable are ONE decision.**
3297    ///
3298    /// Review found the sequencing error this closes: making `at://` storable
3299    /// while nothing can poll it does not leave the feature dormant, it creates
3300    /// permanent failures that the cause histogram then publishes as
3301    /// unreachable publishers — the exact conflation it exists to end.
3302    #[test]
3303    fn an_at_uri_is_not_storable_while_standard_site_is_off() {
3304        let uri = "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab";
3305        assert!(
3306            !is_storable_feed_url(uri, false),
3307            "stored a feed nothing can poll"
3308        );
3309        assert!(is_storable_feed_url(uri, true));
3310        assert!(is_storable_feed_url("https://example.com/feed.xml", false));
3311        assert!(is_storable_feed_url("https://example.com/feed.xml", true));
3312    }
3313
3314    /// **A non-canonical scheme spelling is recognised and refused.** URL
3315    /// schemes are case-insensitive, so `At://` names the same thing as
3316    /// `at://` — but `feeds.url` is UNIQUE, so accepting both is two rows for
3317    /// one publication. Recognised (not passed through to the generic checks
3318    /// as if it were an ordinary URL), then refused for the spelling.
3319    /// Since publications are polled, only a URI the storage guard accepts is
3320    /// a publication. Any other at-URI is Unsupported, which no poller reads.
3321    #[test]
3322    fn only_a_storable_publication_uri_is_a_publication() {
3323        let did = "did:plc:ohutz6x5acjmpuulp3x7wxxc";
3324        let pubn = crate::lexicon::nsid::STANDARD_PUBLICATION;
3325        assert_eq!(
3326            FeedKind::of(&format!("at://{did}/{pubn}/3lab")),
3327            FeedKind::Publication
3328        );
3329        for unsupported in [
3330            format!("at://{did}/app.bsky.feed.post/3lab"),
3331            format!("At://{did}/{pubn}/3lab"),
3332            format!("at://alice.example.com/{pubn}/3lab"),
3333            format!("at://did:plc:short/{pubn}/3lab"),
3334        ] {
3335            assert_eq!(
3336                FeedKind::of(&unsupported),
3337                FeedKind::Unsupported,
3338                "{unsupported}"
3339            );
3340        }
3341        assert_eq!(FeedKind::of("https://example.com/feed.xml"), FeedKind::Rss);
3342        assert!(!FeedKind::POLLABLE.contains(&FeedKind::Unsupported));
3343        assert_eq!(FeedKind::parse("unsupported"), Some(FeedKind::Unsupported));
3344    }
3345
3346    #[test]
3347    fn a_non_canonical_at_uri_spelling_is_recognised_and_refused() {
3348        for odd in [
3349            "At://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
3350            "AT://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
3351        ] {
3352            assert!(!is_storable_feed_url(odd, true), "stored {odd:?}");
3353            // Fails CLOSED: it is an at-URI this reader will not store, not an
3354            // unparseable string that the `Err(_) => Public` arm waves through.
3355            assert!(
3356                classify_feed_privacy(odd).is_private(),
3357                "{odd:?} was declared publishable"
3358            );
3359        }
3360    }
3361
3362    /// **The handle form is not storable — the DID form is the identity.**
3363    ///
3364    /// `feeds.url` is UNIQUE; a handle and its DID would be two rows for one
3365    /// publication, and a handle can change hands. Every spelling is refused,
3366    /// canonical or not; resolving one to a DID is the input path's job.
3367    #[test]
3368    fn a_handle_form_publication_uri_is_not_storable() {
3369        for authority in [
3370            "alice.example.com",
3371            "EXAMPLE.COM",
3372            "169.254.169.254",
3373            "pds.internal",
3374            "printer.local",
3375            "host:8080",
3376            "-.-",
3377            "a b.c",
3378        ] {
3379            let uri = format!("at://{authority}/site.standard.publication/3lab");
3380            assert!(
3381                !is_storable_feed_url(&uri, true),
3382                "accepted authority {authority:?}"
3383            );
3384        }
3385    }
3386
3387    /// A control character or space in the at-URI is refused: `scheduler.rs`
3388    /// logs `%feed.url` with Display, and `feeds.url` is UNIQUE.
3389    #[test]
3390    fn an_at_uri_with_control_characters_is_not_storable() {
3391        for bad in [
3392            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab\n",
3393            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab ",
3394            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3l\tab",
3395        ] {
3396            assert!(!is_storable_feed_url(bad, true), "accepted {bad:?}");
3397        }
3398    }
3399
3400    /// **The `at://` exemption is a REGRESSION unless it is narrow.**
3401    ///
3402    /// Fan-out review found the arm I added was a bare prefix match, so *any*
3403    /// attacker-chosen string starting `at://` was declared safe to publish —
3404    /// skipping the userinfo check, the known-provider table, the private-path
3405    /// markers, the secret-query keys and the entropy heuristics. Measured
3406    /// against `main`, these three went from `Private` to `Public`.
3407    ///
3408    /// That matters because `rename_subscription` caches the URL AND rewrites
3409    /// the user's PUBLIC PDS record, with `classify_feed_privacy` as its only
3410    /// gate.
3411    #[test]
3412    fn a_credential_bearing_at_uri_is_still_private() {
3413        for hostile in [
3414            "at://user:pass@private.example.com/feed/private/TOKEN?apikey=deadbeefdeadbeef",
3415            "at://patreon.com/rss/12345?auth=deadbeefdeadbeefdeadbeef",
3416            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab?apikey=sekrit",
3417        ] {
3418            assert!(
3419                matches!(classify_feed_privacy(hostile), FeedPrivacy::Private(_)),
3420                "declared public: {hostile}"
3421            );
3422        }
3423    }
3424
3425    /// An rkey is `[A-Za-z0-9._:~-]` per atproto. Without that, a query string
3426    /// or path fragment smuggled into the rkey satisfies the three-segment
3427    /// check — which is what the exemption above keys off.
3428    #[test]
3429    fn an_rkey_outside_the_atproto_charset_is_not_storable() {
3430        for bad in [
3431            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab?apikey=sekrit",
3432            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab#frag",
3433            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab%2Fevil",
3434            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/caf\u{e9}",
3435            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab\u{202e}x",
3436        ] {
3437            assert!(!is_storable_feed_url(bad, true), "accepted rkey in {bad:?}");
3438        }
3439        // The legitimate charset still passes.
3440        for good in [
3441            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h",
3442            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/a.b_c~d-e",
3443        ] {
3444            assert!(is_storable_feed_url(good, true), "refused {good:?}");
3445        }
3446    }
3447
3448    /// **A real rkey is a TID, and a TID looks exactly like a secret.**
3449    ///
3450    /// Without an explicit `at://` arm, `classify_feed_privacy` runs the generic
3451    /// "high-entropy token in path" heuristic over the rkey. Measured: a
3452    /// realistic 16-char rkey on a handle-form at-URI is classified PRIVATE and
3453    /// the subscription REFUSED. The DID form escaped only because it fails to
3454    /// parse as a `Url` at all — so the bug was invisible from that side.
3455    ///
3456    /// The first version of this test used the rkey `3lab`, which is too short
3457    /// to trip the heuristic, so it passed with and without the fix.
3458    #[test]
3459    fn a_realistic_at_uri_rkey_is_not_mistaken_for_a_secret() {
3460        for uri in [
3461            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h",
3462            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/aB3xK9pQ7mZ2vN8w",
3463            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h",
3464        ] {
3465            assert_eq!(
3466                classify_feed_privacy(uri),
3467                FeedPrivacy::Public,
3468                "a publication rkey was mistaken for a credential: {uri}"
3469            );
3470        }
3471    }
3472
3473    /// **The DID form is the one that matters, and the one `Url::parse` cannot
3474    /// read.**
3475    ///
3476    /// `Url::parse("at://did:plc:…/…")` fails with *invalid port number* — the
3477    /// colons in the DID are taken as a port separator. So the obvious
3478    /// implementation, adding `"at"` to the `matches!` on `u.scheme()`, silently
3479    /// rejects every DID-based at-URI while appearing to work: the handle form
3480    /// (`at://alice.example.com/…`) parses fine and would pass such a test.
3481    ///
3482    /// All 19 at-URI rows in production are the DID form. A test written with a
3483    /// handle would have passed against an implementation that cannot store a
3484    /// single one of them.
3485    #[test]
3486    fn a_did_form_publication_uri_is_storable() {
3487        assert!(is_storable_feed_url(
3488            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
3489            true
3490        ));
3491    }
3492
3493    /// **An allowlist entry, not a loosening.** `at://` is accepted for exactly
3494    /// one foreign collection. Any other collection is somebody else's lexicon
3495    /// arriving through a path (`resolve_subscriptions`, OPML import) that takes
3496    /// records from outside with no add-path to reject them.
3497    #[test]
3498    fn an_at_uri_for_another_collection_is_not_storable() {
3499        assert!(!is_storable_feed_url(
3500            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/community.lexicon.rss.subscription/3lab",
3501            true
3502        ));
3503        assert!(!is_storable_feed_url(
3504            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/app.bsky.feed.post/3lab",
3505            true
3506        ));
3507    }
3508
3509    /// The malformed shapes, each of which a naive `split('/')` would accept.
3510    #[test]
3511    fn a_malformed_at_uri_is_not_storable() {
3512        for bad in [
3513            "at://",
3514            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc",
3515            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication",
3516            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/",
3517            "at:///site.standard.publication/3lab",
3518            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab/extra",
3519            "at://not-a-did-or-handle/site.standard.publication/3lab",
3520            "at://did:plc:TOOSHORT/site.standard.publication/3lab",
3521        ] {
3522            assert!(!is_storable_feed_url(bad, true), "accepted {bad:?}");
3523        }
3524    }
3525
3526    /// **The reason the function exists, unchanged.** Mutating the new branch to
3527    /// accept any scheme makes this fail while the at-URI tests keep passing —
3528    /// that asymmetry is what says the change was an allowlist entry.
3529    #[test]
3530    fn the_refused_schemes_are_still_refused() {
3531        for bad in [
3532            "javascript:alert(1)",
3533            "file:///etc/passwd",
3534            "data:text/html,<script>",
3535            "ftp://example.com/feed.xml",
3536            "at:did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
3537        ] {
3538            assert!(!is_storable_feed_url(bad, true), "accepted {bad:?}");
3539        }
3540        // A hostless http(s) URL is a PARSE error, not a parsed URL with no
3541        // host — `https:///feed.xml` even parses as host `feed.xml`. What
3542        // refuses these is the `Err` arm, so that is what this pins.
3543        for hostless in ["http://", "https://?q=1", "http:///"] {
3544            assert!(
3545                !is_storable_feed_url(hostless, true),
3546                "a hostless URL {hostless:?} was storable"
3547            );
3548        }
3549        assert!(is_storable_feed_url("https://example.com/feed.xml", true));
3550        assert!(is_storable_feed_url("http://example.com/feed.xml", true));
3551    }
3552}