Skip to main content

feather_reader/
feed.rs

1//! Feed fetch → parse → sanitize → store pipeline.
2//!
3//! This is the module that turns a feed URL into rows in the [`store`]. It is
4//! deliberately conservative on three axes, because a feed reader ingests
5//! **hostile, arbitrary web input**:
6//!
7//! 1. **Politeness** — fetches use a **conditional GET** (`If-None-Match` /
8//!    `If-Modified-Since` from the stored `ETag` / `Last-Modified`), an
9//!    identifiable [`crate::USER_AGENT`], a request timeout, and a simple
10//!    exponential backoff hint on error. A `304 Not Modified` is a no-op:
11//!    the feed is untouched apart from bumping its next-poll time.
12//! 2. **Safety** — every entry's HTML is run through `ammonia` before it is
13//!    ever stored (and therefore before it is ever rendered). Scripts, event
14//!    handlers, `javascript:` URLs, tracking pixels' dangerous attributes, and
15//!    other XSS vectors are stripped. Feeds carrying `<script>` is not
16//!    hypothetical; treat all feed HTML as untrusted.
17//! 3. **Robustness** — a malformed feed is **logged and skipped**, never a
18//!    panic. One bad publisher must not take down the poller. All non-test
19//!    paths use `Result`/`anyhow`; there are no `unwrap`/`expect`s.
20//!
21//! The normalized shape written to the store is the store's own
22//! [`store::NewFeed`] / [`store::NewEntry`]; dedup is by feed-native GUID via
23//! [`store::insert_entries`]'s `ON CONFLICT (feed_id, guid)` upsert.
24
25use std::time::Duration;
26
27use anyhow::{Context, Result};
28use chrono::{DateTime, SecondsFormat, Utc};
29use feed_rs::model::{Entry as RawEntry, Feed as RawFeed, Text};
30use reqwest::header::{ETAG, IF_MODIFIED_SINCE, IF_NONE_MATCH, LAST_MODIFIED};
31use reqwest::{Client, StatusCode};
32use sqlx::SqlitePool;
33use url::Url;
34
35use crate::store::{self, Feed, NewEntry, NewFeed};
36
37/// The privacy classification of a feed URL — the output of
38/// [`classify_feed_privacy`].
39///
40/// A **private** feed carries a secret (a token / key / auth credential) *in the
41/// URL itself* — a Substack `…/feed/private/<token>`, a Patreon `?auth=…` feed,
42/// a Ghost members `?uuid=` feed, a private-podcast token feed (Supercast,
43/// Supporting Cast, tokened Megaphone/Acast+), and so on. FeatherReader stores a
44/// user's subscriptions as records in their **public PDS** (unauthenticated
45/// `getRecord` / `listRecords` + the firehose, retained even after delete), so
46/// writing such a URL anywhere — the PDS *or* the server's own store — would risk
47/// leaking paid / members-only access.
48///
49/// **Decision (stopgap until atproto permissioned data ships): FeatherReader
50/// supports PUBLIC feeds only.** A feed classified [`FeedPrivacy::Private`] is
51/// *refused* at the add / import boundary — never fetched, never stored, never
52/// written to the PDS. There is no local-secret fallback and no override: the
53/// server holds NO private secret, ever, which keeps "your data lives in your
54/// public PDS" 100% honest.
55#[derive(Debug, Clone, PartialEq, Eq)]
56pub enum FeedPrivacy {
57    /// No secret detected in the URL; safe to add as a public feed.
58    Public,
59    /// A secret was detected in the URL. The `String` is a short, human-readable
60    /// reason (for logging / the skip report), e.g. `"substack private feed
61    /// path"`. The feed is refused — not fetched, stored, or written anywhere.
62    Private(String),
63}
64
65impl FeedPrivacy {
66    /// Whether this classification is [`FeedPrivacy::Private`].
67    pub fn is_private(&self) -> bool {
68        matches!(self, FeedPrivacy::Private(_))
69    }
70}
71
72/// Query-parameter *keys* that, when present with a long/opaque value, mark a URL
73/// as carrying a secret. Conservative and lowercase-compared; matched as a whole
74/// key (case-insensitive) so a benign `keyword=` does NOT trip `key`. This is the
75/// generic, provider-agnostic credential-in-query defence — it catches paid
76/// feeds from providers we've never heard of. Covers Patreon (`auth`), Ghost
77/// members (`uuid`), token-in-query feeds (`token`/`key`/`k`/`sig`/`hash`), and
78/// the long tail (`access`/`apikey`/`private`/`password`/`u`/`s`/`p`/…).
79const SECRET_QUERY_KEYS: &[&str] = &[
80    "token", "key", "auth", "secret", "k", "sig", "hash", "access", "apikey", "api_key", "uuid",
81    "id", "u", "s", "p", "private", "password", "pw",
82];
83
84/// Path *segments* / fragments that mark a private-feed URL shape. Matched as a
85/// case-insensitive substring of the (lowercased) path so `/feed/private/<tok>`,
86/// `/members/…`, `/subscriber/…` etc. all trip regardless of the token that
87/// follows. Provider-agnostic: many paid providers expose members-only feeds
88/// under one of these path conventions.
89const PRIVATE_PATH_MARKERS: &[&str] = &[
90    "/private/",
91    "/feed/private/",
92    "/rss/private/",
93    "/private-feed/",
94    "/members/",
95    "/member/",
96    "/subscriber/",
97];
98
99/// A KNOWN paid/private feed provider, matched by host substring + (optionally) a
100/// path/query marker specific to that provider. This is the **secondary**,
101/// precision layer on top of the generic heuristic — it names providers so the
102/// skip report can say *why* and so we catch provider-specific shapes that the
103/// generic pass might rate as borderline. Data-driven and easy to extend: add a
104/// row, don't touch the matcher.
105struct KnownProvider {
106    /// Substring that must appear in the URL host (lowercased), e.g.
107    /// `substack.com`.
108    host_contains: &'static str,
109    /// Optional lowercased substring that must appear in the path-or-query for a
110    /// match (a provider's private-feed marker). `None` = the host alone is
111    /// enough (used for hosts that ONLY serve private/tokened feeds).
112    marker: Option<&'static str>,
113    /// Human-readable reason for the skip report.
114    reason: &'static str,
115}
116
117/// The known-provider table. Covers paid NEWSLETTERS and private PODCASTS — an
118/// RSS reader ingests both. Kept intentionally verbose/commented so it's obvious
119/// what each row targets and safe to extend.
120const KNOWN_PROVIDERS: &[KnownProvider] = &[
121    // --- Paid newsletters -------------------------------------------------
122    // Substack private feed: author.substack.com/feed/private/<token>.
123    KnownProvider {
124        host_contains: "substack.com",
125        marker: Some("/feed/private/"),
126        reason: "Substack private feed",
127    },
128    // Patreon RSS carries the member token as ?auth=.
129    KnownProvider {
130        host_contains: "patreon.com",
131        marker: Some("auth="),
132        reason: "Patreon member feed",
133    },
134    // Ghost members feed: ?uuid=<member-uuid> (or a members token path).
135    KnownProvider {
136        host_contains: "ghost.io",
137        marker: Some("uuid="),
138        reason: "Ghost members feed",
139    },
140    // Buttondown paid RSS uses a per-subscriber token in the path/query.
141    KnownProvider {
142        host_contains: "buttondown.email",
143        marker: Some("token"),
144        reason: "Buttondown premium feed",
145    },
146    KnownProvider {
147        host_contains: "buttondown.com",
148        marker: Some("token"),
149        reason: "Buttondown premium feed",
150    },
151    // Beehiiv premium RSS carries a subscriber token.
152    KnownProvider {
153        host_contains: "beehiiv.com",
154        marker: Some("token"),
155        reason: "Beehiiv premium feed",
156    },
157    // Memberful-gated feeds (host or ?auth token).
158    KnownProvider {
159        host_contains: "memberful.com",
160        marker: None,
161        reason: "Memberful members feed",
162    },
163    // Pico / Steady member feeds.
164    KnownProvider {
165        host_contains: "pico.link",
166        marker: None,
167        reason: "Pico member feed",
168    },
169    KnownProvider {
170        host_contains: "steadyhq.com",
171        marker: None,
172        reason: "Steady member feed",
173    },
174    // --- Private podcasts -------------------------------------------------
175    // Supercast private podcast feeds (host serves tokened member feeds only).
176    KnownProvider {
177        host_contains: "supercast.com",
178        marker: None,
179        reason: "Supercast private podcast",
180    },
181    KnownProvider {
182        host_contains: "supercast.tech",
183        marker: None,
184        reason: "Supercast private podcast",
185    },
186    // Supporting Cast private podcast feeds (supportingcast.fm).
187    KnownProvider {
188        host_contains: "supportingcast.fm",
189        marker: None,
190        reason: "Supporting Cast private podcast",
191    },
192    // RedCircle private/exclusive feeds.
193    KnownProvider {
194        host_contains: "redcircle.com",
195        marker: Some("private"),
196        reason: "RedCircle private podcast",
197    },
198    // Private/tokened Megaphone, Acast+, and Omny feeds carry an access token.
199    KnownProvider {
200        host_contains: "megaphone.fm",
201        marker: Some("token"),
202        reason: "Megaphone private podcast",
203    },
204    KnownProvider {
205        host_contains: "acast.com",
206        marker: Some("token"),
207        reason: "Acast+ private podcast",
208    },
209    KnownProvider {
210        host_contains: "omny.fm",
211        marker: Some("token"),
212        reason: "Omny private podcast",
213    },
214    // Apple / Spotify subscriber podcast feeds carry a per-listener token.
215    KnownProvider {
216        host_contains: "podcasts.apple.com",
217        marker: Some("token"),
218        reason: "Apple subscriber podcast",
219    },
220    KnownProvider {
221        host_contains: "spotify.com",
222        marker: Some("token"),
223        reason: "Spotify subscriber podcast",
224    },
225];
226
227/// Whether a URL may be **stored or published** as a feed URL at all.
228///
229/// This is the storage-side twin of the scheme check `net::check_scheme` applies
230/// before fetching. The fetch side has always been safe, because nothing can
231/// reach the network except through `net.rs` — but "safe to fetch" and "safe to
232/// write down" are different questions, and only the first had an answer.
233///
234/// Two paths took a URL from outside and stored it with no validation at all:
235/// `resolve_subscriptions` (any atproto client can write a subscription record
236/// into a user's repo) and the OPML import (`xmlUrl` is whatever the file says).
237/// `classify_feed_privacy` does not cover this — it deliberately returns
238/// `Public` for an unparseable URL, on the stated assumption that "the add path
239/// will reject it as malformed regardless", and those two paths are the ones
240/// that never had an add path to do the rejecting.
241///
242/// Note that `javascript:alert(1)` and `file:///etc/passwd` both *parse* cleanly
243/// as URLs, so parsing is not the check — the scheme is.
244pub fn is_storable_feed_url(url: &str, allow_at_uri: bool) -> bool {
245    // **`at://` is checked BEFORE `Url::parse`, because `Url::parse` cannot read
246    // the form that matters.** `at://did:plc:…/…` fails to parse with *invalid
247    // port number* — the colons in the DID are taken as a port separator — while
248    // the handle form `at://alice.example.com/…` parses fine. So adding `"at"`
249    // to the `matches!` below would appear to work and silently reject every
250    // DID-based at-URI, which is all of them in practice.
251    if let Some(rest) = crate::atproto::strip_at_prefix(url) {
252        // **Recognised case-insensitively, stored canonically.** Schemes are
253        // case-insensitive, so `At://` names the same publication — but
254        // `feeds.url` is UNIQUE, so accepting both spellings is two rows for
255        // one publication, the hazard the canonical-handle rule exists for.
256        // Recognising it here rather than letting it fall through to the
257        // generic checks is what keeps it out of the poller: nothing can fetch
258        // it under any spelling.
259        if !url.starts_with(crate::atproto::AT_URI_PREFIX) {
260            return false;
261        }
262        // **This gates STORING only — polling is handled by exclusion**, by
263        // kind rather than by any re-description of the URL, and
264        // `FeedKind::POLLABLE` is the one place the why is written down.
265        return allow_at_uri && is_storable_publication_uri(rest);
266    }
267    match Url::parse(url) {
268        Ok(u) => {
269            // No host check: for http(s) the `url` crate refuses every hostless
270            // spelling at parse (`http://`, `https://?q`, `http:///` are all
271            // "empty host") and turns `https:///x` into host `x`. A conjunct
272            // requiring a non-empty host was unreachable — a test hunt listed
273            // it as untested, and the honest answer was that no input reaches
274            // it. The `Err` arm below is what refuses a hostless URL.
275            matches!(u.scheme(), "http" | "https")
276        }
277        Err(_) => false,
278    }
279}
280
281/// The body of an `at://` URI — `<did-or-handle>/<collection>/<rkey>` — judged
282/// as a **storable feed**.
283///
284/// An allowlist entry, not a loosening: exactly one foreign collection is
285/// accepted, `site.standard.publication`. The two paths this guard exists for
286/// (`resolve_subscriptions`, the OPML import) take records written by any
287/// atproto client, so "it is an at-URI" is not a reason to store it — only "it
288/// is a publication this reader knows how to poll" is.
289fn is_storable_publication_uri(rest: &str) -> bool {
290    let mut parts = rest.split('/');
291    let (Some(authority), Some(collection), Some(rkey)) =
292        (parts.next(), parts.next(), parts.next())
293    else {
294        return false;
295    };
296    parts.next().is_none()
297        && collection == crate::lexicon::nsid::STANDARD_PUBLICATION
298        // **The rkey is validated against atproto's rules, not a blacklist.**
299        //
300        // A blacklist was the first attempt and it leaked twice: `is_control()`
301        // is Unicode category Cc only, so a bidi override (Cf) passed — and it
302        // reordered both the manage page and `scheduler.rs`'s `%feed.url` log
303        // line. Worse, nothing stopped a query string or fragment living inside
304        // the rkey, which satisfies the three-segment check and is exactly what
305        // `classify_feed_privacy`'s `at://` exemption keys off: a token
306        // smuggled there would have been declared public.
307        //
308        // An allowlist cannot leak the next character class someone finds.
309        // The charset alone still admitted `.`, `..` and a 10 000-byte key;
310        // `is_valid_rkey` carries the length and reserved-name rules too.
311        && crate::atproto::is_valid_rkey(rkey)
312        && is_storable_at_authority(authority)
313}
314
315/// The DID form only. `did:plc:` identifiers are validated by
316/// [`crate::oauth::identity::is_atproto_did`] rather than a `did:` prefix check,
317/// which would accept `did:plc:TOOSHORT`.
318///
319/// **The handle form is not storable, for the reason the canonical-handle rule
320/// already gave:** `feeds.url` is UNIQUE, so `at://alice.example.com/…` beside
321/// `at://did:plc:…/…` is two rows — two sidebar entries, and two polled copies
322/// once the reader is wired — for one publication. A handle is a mutable name
323/// for a DID; the row is keyed on the identity. Resolving a pasted or imported
324/// handle to its DID is the reader's job (#165 already resolves DIDs to their
325/// PDS), and belongs at input, not in storage.
326fn is_storable_at_authority(authority: &str) -> bool {
327    crate::oauth::identity::is_atproto_did(authority)
328}
329
330/// Classify whether a feed URL carries a secret credential in the URL itself.
331///
332/// Returns [`FeedPrivacy::Private`] (with a reason) when the URL looks like it
333/// embeds a token / key / auth credential, else [`FeedPrivacy::Public`].
334///
335/// **Design — provider-agnostic first.** The primary defence is a generic
336/// credential-in-URL heuristic that catches paid feeds from *any* provider, not
337/// just the ones we've named; a secondary known-provider table adds precision
338/// (and a nicer reason) for the common paid newsletters and private podcasts. We
339/// deliberately **bias toward flagging**: a false-positive block of a public feed
340/// is low-harm (the user just can't add that one feed yet), whereas a false
341/// negative would leak a paid secret onto the public network — high-harm.
342///
343/// Detection (any one is sufficient):
344/// 1. **Userinfo** — `https://user:pass@host/…` embeds credentials directly.
345/// 2. **Known private-feed path markers** — `PRIVATE_PATH_MARKERS`
346///    (`/feed/private/`, `/members/`, `/subscriber/`, …).
347/// 3. **Credential query parameters** — a query key in `SECRET_QUERY_KEYS` with
348///    a long/opaque value (Patreon `?auth=`, Ghost `?uuid=`, `?token=`, …).
349/// 4. **High-entropy opaque token segments** — a long opaque blob (hex ≥ 16,
350///    base64url ≥ 16, or a UUID) anywhere in the path or a query value, even
351///    without a telltale name.
352/// 5. **Known providers** — `KNOWN_PROVIDERS` host (+ optional marker) match.
353///
354/// An unparseable URL is treated as [`FeedPrivacy::Public`]: the add path rejects
355/// a malformed URL downstream anyway, and we don't want a parse quirk to
356/// misclassify.
357pub fn classify_feed_privacy(url: &str) -> FeedPrivacy {
358    // **`at://` is classified deliberately, and NOT doing so refused real
359    // subscriptions.** An atproto rkey is a TID — 13 base32-sortable characters
360    // — which is exactly what the generic "high-entropy token in path"
361    // heuristic below is looking for. Measured: without this arm,
362    // `at://did:plc:…/site.standard.publication/3lab2c4d5e6f7g8h` is
363    // classified PRIVATE and the subscription refused, while the DID form slips
364    // through only because it fails to parse as a `Url` at all.
365    //
366    // `Public` is the right answer: a publication is a public record in a
367    // public repo and the rkey is a handle, not a secret, so there is no
368    // private/paid shape for this scheme to carry.
369    if let Some(rest) = crate::atproto::strip_at_prefix(url) {
370        // A non-canonical scheme spelling is an at-URI this reader will not
371        // store, not an unparseable string for the `Err(_) => Public` arm below
372        // to wave through. Fail closed.
373        if !url.starts_with(crate::atproto::AT_URI_PREFIX) {
374            return FeedPrivacy::Private("non-canonical at:// scheme spelling".to_string());
375        }
376        // **Only a WELL-FORMED publication URI is exempt.** The first version
377        // of this was a bare prefix match, which declared any attacker-chosen
378        // string starting `at://` safe to publish — skipping the userinfo
379        // check, the known-provider table, the private-path markers, the
380        // secret-query keys and the entropy heuristics all at once. That is a
381        // regression against every one of them, on a path
382        // (`rename_subscription`) where this function is the only gate and the
383        // value is written to the user's PUBLIC repo.
384        //
385        // Anything else falls through to the generic checks below, which is
386        // where a credential-bearing string belongs.
387        if is_storable_publication_uri(rest) {
388            return FeedPrivacy::Public;
389        }
390        // **Malformed `at://` is REFUSED, not passed through.** Falling through
391        // reaches `Url::parse`, which fails on the DID form and lands on the
392        // `Err(_) => Public` arm below — whose justification is "the add path
393        // will reject it as a malformed URL regardless".
394        //
395        // **Defence in depth, not a live gate.** This comment used to say the
396        // justification is false on the `rename_subscription` path, "where this
397        // function is the only gate". That stopped being true when storability
398        // moved ahead of privacy on the repoint: a review then found no
399        // production caller can reach this arm at all — add pre-checks the
400        // `at://` prefix, rename and OPML and `resolve_subscriptions` all
401        // refuse a non-storable URL first. It stays because a fail-closed
402        // branch is worth its keep for the next caller that arrives without
403        // one, and because deleting a guard on the grounds that nothing
404        // currently reaches it is how the next one gets it wrong. It is not
405        // load-bearing today, and saying so is the honest version.
406        return FeedPrivacy::Private("not a well-formed at:// publication URI".to_string());
407    }
408    let parsed = match Url::parse(url) {
409        Ok(u) => u,
410        // Can't parse => the add path will reject it as a malformed URL regardless.
411        Err(_) => return FeedPrivacy::Public,
412    };
413
414    // (1) Userinfo (`https://user:pass@host/…`) — credentials in the authority.
415    if !parsed.username().is_empty() || parsed.password().is_some() {
416        return FeedPrivacy::Private("credentials in URL userinfo".to_string());
417    }
418
419    let path_lower = parsed.path().to_ascii_lowercase();
420    let query_lower = parsed.query().unwrap_or("").to_ascii_lowercase();
421    let host_lower = parsed.host_str().unwrap_or("").to_ascii_lowercase();
422
423    // (0) Public-feed allowlist. A handful of large, fully-public feed shapes
424    // carry a high-entropy-looking id in the query that would otherwise trip the
425    // generic entropy heuristic. YouTube channel/playlist RSS
426    // (`youtube.com/feeds/videos.xml?channel_id=UC…` / `?playlist_id=PL…`) is the
427    // canonical way any reader subscribes to a channel — the id is a PUBLIC
428    // handle, not a secret. Allowlist it before the generic checks so we don't
429    // false-block it. (Userinfo / known-provider markers are checked below and
430    // still apply, so this can't be used to smuggle a credential.)
431    if is_public_youtube_feed(&host_lower, &path_lower, &parsed) {
432        return FeedPrivacy::Public;
433    }
434
435    // (5) Known-provider precision layer (checked early so its specific reason
436    // wins over a generic one). Host substring + optional path/query marker.
437    for kp in KNOWN_PROVIDERS {
438        if host_lower.contains(kp.host_contains) {
439            let marker_ok = match kp.marker {
440                None => true,
441                Some(m) => {
442                    let m = m.to_ascii_lowercase();
443                    path_lower.contains(&m) || query_lower.contains(&m)
444                }
445            };
446            if marker_ok {
447                return FeedPrivacy::Private(kp.reason.to_string());
448            }
449        }
450    }
451
452    // (2) Known private-feed path markers.
453    for marker in PRIVATE_PATH_MARKERS {
454        if path_lower.contains(marker) {
455            return FeedPrivacy::Private(format!("private feed path `{marker}`"));
456        }
457    }
458
459    // (3) Credential query parameters with a long/opaque value.
460    for (k, v) in parsed.query_pairs() {
461        let key = k.as_ref().to_ascii_lowercase();
462        if SECRET_QUERY_KEYS.iter().any(|sk| *sk == key) && value_is_opaque(v.as_ref()) {
463            return FeedPrivacy::Private(format!("credential query parameter `{key}`"));
464        }
465    }
466
467    // (4) High-entropy opaque token segments (an embedded key/token with no
468    // telltale name): hex ≥ 16, base64url ≥ 16, or a UUID, in path or query.
469    // The dominant real-world private-podcast shape delivers the token as a
470    // *filename* (`<token>.rss` / `<token>.xml`) or affixed inside a larger
471    // segment (`feed-<uuid>`), so [`segment_hides_secret`] strips a trailing feed
472    // extension AND scans dot/underscore/hyphen-delimited sub-parts, not just the
473    // whole segment.
474    for seg in parsed.path().split('/').filter(|s| !s.is_empty()) {
475        if segment_hides_secret(seg) {
476            return FeedPrivacy::Private("high-entropy token in path".to_string());
477        }
478    }
479    for (_, v) in parsed.query_pairs() {
480        if looks_like_embedded_secret(v.as_ref()) {
481            return FeedPrivacy::Private("high-entropy token in query".to_string());
482        }
483    }
484
485    FeedPrivacy::Public
486}
487
488/// Whether a *named* credential query value (`?token=<v>`) is long/opaque enough
489/// to count as a secret. A short value (e.g. an enum like `?token=none`) is not.
490/// We treat a UUID, or anything ≥ 8 chars that isn't an obvious plain word, as
491/// opaque — named credential keys already signal intent, so the length bar is
492/// low.
493fn value_is_opaque(v: &str) -> bool {
494    if v.is_empty() {
495        return false;
496    }
497    if is_uuid(v) {
498        return true;
499    }
500    v.len() >= 8
501}
502
503/// Heuristic: does `s` look like an embedded secret (an opaque high-entropy
504/// token), as opposed to an ordinary slug or word? Matches a UUID, a hex string
505/// ≥ 16 chars, or a base64url-ish blob ≥ 16 chars that mixes letters and digits
506/// and isn't a hyphen/dot slug. Deliberately strict so it only fires on things
507/// that really look like keys — the named-marker and known-provider checks cover
508/// the rest.
509fn looks_like_embedded_secret(s: &str) -> bool {
510    if is_uuid(s) {
511        return true;
512    }
513    // Hex string ≥ 16 chars (e.g. a 32-char MD5-ish token).
514    if s.len() >= 16 && s.chars().all(|c| c.is_ascii_hexdigit()) {
515        return true;
516    }
517    // base64url-ish opaque blob ≥ 16 chars.
518    if s.len() < 16 {
519        return false;
520    }
521    // Hyphen/dot-heavy slugs (`this-is-a-normal-post-title`) are not secrets.
522    let separators = s
523        .bytes()
524        .filter(|b| *b == b'-' || *b == b'.' || *b == b' ')
525        .count();
526    if separators >= 3 {
527        return false;
528    }
529    // Must be plausibly token-charset: base64url alphabet only. `=` is accepted
530    // as base64 padding (it only ever appears trailing on a real blob, so a
531    // padded base64url token like `…dnc=` still counts).
532    let token_chars = s
533        .chars()
534        .filter(|c| c.is_ascii_alphanumeric() || matches!(c, '_' | '-' | '='))
535        .count();
536    if token_chars < s.chars().count() {
537        return false;
538    }
539    let has_alpha = s.chars().any(|c| c.is_ascii_alphabetic());
540    let has_digit = s.chars().any(|c| c.is_ascii_digit());
541    if !(has_alpha && has_digit) {
542        // A token almost always mixes letters and digits; a pure-alpha long
543        // segment is far more likely to be a normal (if long) slug/word.
544        return false;
545    }
546    // Distinct-character ratio: real tokens use most of the alphabet, words
547    // repeat a small set. Require >= 10 distinct chars for a 16+ char blob.
548    let mut seen = std::collections::HashSet::new();
549    for c in s.chars() {
550        seen.insert(c.to_ascii_lowercase());
551    }
552    seen.len() >= 10
553}
554
555/// Known feed/file extensions a token filename may wear (`<token>.rss`,
556/// `<token>.xml`, …). Stripped before the whole-segment secret test so a
557/// tokened *filename* — the dominant private-podcast URL shape — is still caught.
558const FEED_EXTENSIONS: &[&str] = &["rss", "xml", "atom", "json", "rss20"];
559
560/// Whether a single path segment hides an embedded secret. Beyond the plain
561/// whole-segment [`looks_like_embedded_secret`] test, this also catches the two
562/// real-world shapes that wrap a token so the whole segment is no longer a clean
563/// blob:
564///
565/// 1. **Token-as-filename** — `<token>.rss` / `<token>.xml`: strip a trailing
566///    feed extension and re-test the stem.
567/// 2. **Token affixed inside a larger segment** — `feed-<uuid>`, `<token>.xml`,
568///    `pod_<hex32>`: split on `.`/`_`/`-` and test each sub-part, so a
569///    high-entropy blob delimited by an affix is still found.
570fn segment_hides_secret(seg: &str) -> bool {
571    if looks_like_embedded_secret(seg) {
572        return true;
573    }
574    // (1) Strip a trailing known feed extension and re-test the stem.
575    if let Some((stem, ext)) = seg.rsplit_once('.') {
576        if FEED_EXTENSIONS.contains(&ext.to_ascii_lowercase().as_str())
577            && looks_like_embedded_secret(stem)
578        {
579            return true;
580        }
581    }
582    // (2) A UUID embedded with an affix (`feed-<uuid>`, `<uuid>-audio`) — the
583    // `-` delimiters inside the UUID mean a naive split can't see it, so scan for
584    // a canonical UUID substring directly.
585    if contains_uuid(seg) {
586        return true;
587    }
588    // (3) Scan `.`/`_`/`-`-delimited sub-parts for a high-entropy blob affixed to
589    // an ordinary word (`pod_<hex32>`, `<hex32>.mp3`). Only fires on multi-part
590    // segments (a single-part segment was already covered by the whole-segment
591    // test above), so a plain `my-normal-post-slug` — whose parts are short
592    // dictionary words — can't trip it.
593    if seg.contains(['.', '_', '-']) {
594        for part in seg.split(['.', '_', '-']).filter(|p| !p.is_empty()) {
595            if looks_like_embedded_secret(part) {
596                return true;
597            }
598        }
599    }
600    false
601}
602
603/// Whether `s` contains a canonical 8-4-4-4-12 UUID as a substring (allowing an
604/// affix on either side, e.g. `feed-<uuid>` or `<uuid>-audio`). Slides a 36-char
605/// window over the string and tests each with [`is_uuid`].
606fn contains_uuid(s: &str) -> bool {
607    const UUID_LEN: usize = 36; // 8+4+4+4+12 + 4 hyphens.
608    let bytes = s.as_bytes();
609    if bytes.len() < UUID_LEN {
610        return false;
611    }
612    // ASCII-only window: a UUID is pure ASCII hex/hyphen, so byte indexing is
613    // safe here (a multi-byte char in the window just fails is_uuid).
614    (0..=bytes.len() - UUID_LEN).any(|i| s.get(i..i + UUID_LEN).map(is_uuid).unwrap_or(false))
615}
616
617/// Whether this is a PUBLIC YouTube channel/playlist RSS feed
618/// (`www.youtube.com/feeds/videos.xml?channel_id=UC…` or `?playlist_id=PL…`).
619/// The channel/playlist id is a public handle, not a credential, so these feeds
620/// must NOT be flagged by the generic entropy heuristic. We require the exact
621/// public host + feeds path + one of the two public id keys, so this narrow
622/// allowlist can't be abused to smuggle a `?token=` past classification.
623fn is_public_youtube_feed(host_lower: &str, path_lower: &str, parsed: &Url) -> bool {
624    let host_ok = host_lower == "youtube.com"
625        || host_lower == "www.youtube.com"
626        || host_lower.ends_with(".youtube.com");
627    if !host_ok || !path_lower.starts_with("/feeds/videos.xml") {
628        return false;
629    }
630    // Only the public id keys may appear; a `token`/`auth`/… key means treat it
631    // as a normal (potentially private) URL and let the checks below run.
632    parsed.query_pairs().all(|(k, _)| {
633        let k = k.as_ref().to_ascii_lowercase();
634        k == "channel_id" || k == "playlist_id" || k == "user"
635    })
636}
637
638/// Whether `s` is a canonical 8-4-4-4-12 hyphenated UUID (any hex case).
639fn is_uuid(s: &str) -> bool {
640    let groups = [8usize, 4, 4, 4, 12];
641    let parts: Vec<&str> = s.split('-').collect();
642    if parts.len() != groups.len() {
643        return false;
644    }
645    parts
646        .iter()
647        .zip(groups.iter())
648        .all(|(p, &n)| p.len() == n && p.chars().all(|c| c.is_ascii_hexdigit()))
649}
650
651/// How long a single feed fetch may take before we give up.
652const FETCH_TIMEOUT: Duration = Duration::from_secs(30);
653
654/// Per-read idle timeout: cap the wait for the *next* body chunk, so a server
655/// that trickles bytes forever can't tie up a fetch under the total timeout.
656const READ_TIMEOUT: Duration = Duration::from_secs(15);
657
658/// Base backoff applied after a failed poll; the caller multiplies this by the
659/// feed's consecutive-error count (with a ceiling) to space out retries.
660const BACKOFF_BASE: Duration = Duration::from_secs(300);
661
662/// Ceiling on backoff so a persistently broken feed still gets retried daily.
663const BACKOFF_MAX: Duration = Duration::from_secs(24 * 3600);
664
665/// The outcome of polling a single feed. Lets the scheduler decide how to
666/// reschedule (and lets tests assert what happened) without inspecting the DB.
667#[derive(Debug, Clone, PartialEq, Eq)]
668pub enum PollOutcome {
669    /// The feed was fetched, parsed, and stored. `new_entries` is the number of
670    /// entries inserted or updated by this poll.
671    Updated { new_entries: u64 },
672    /// The server returned `304 Not Modified` — nothing changed, nothing stored.
673    NotModified,
674    /// The fetch or parse failed; the feed was left intact and skipped. Carries
675    /// the suggested backoff before the next attempt. Never a panic.
676    ///
677    /// **`kind` and `detail` are the reason, and they exist because their
678    /// absence cost a production investigation.** Until #159 the error was
679    /// logged here and discarded, so `feeds` recorded that a feed was failing
680    /// and never why — which is how sixty feeds broken by our own 304 handling
681    /// looked exactly like sixty dead blogs. `kind` is a small closed
682    /// vocabulary so failures can be counted by cause; `detail` is the message
683    /// for a human reading one row.
684    Failed {
685        backoff: Duration,
686        kind: FailureKind,
687        detail: String,
688    },
689}
690
691/// Why a poll failed, as a closed set.
692///
693/// Closed on purpose: the point is to *count* failures by cause, and a free-text
694/// kind cannot be counted. The detail string carries whatever else matters.
695#[derive(Debug, Clone, Copy, PartialEq, Eq)]
696pub enum FailureKind {
697    /// The request never produced a response — DNS, TLS, timeout, connection
698    /// refused, or a refusal by the SSRF guard.
699    Fetch,
700    /// A response arrived with a non-success status.
701    Status,
702    /// The body was too large, or reading it failed part-way.
703    Body,
704    /// The body arrived and is not a feed this parser can read.
705    Parse,
706}
707
708/// What the poller does with a feed row.
709///
710/// **A column, not a predicate.** "Can this be fetched?" used to be
711/// `substr(url, 1, 5) = 'at://'` spliced into four statements, with a fifth
712/// reader that had already drifted from them. Deciding it once, in Rust, at
713/// insert — and storing the answer — means SQL cannot disagree with the
714/// fetcher, and wiring the standard.site reader becomes a change to this
715/// function plus a dispatch, rather than an edit to every statement that
716/// mentions a URL.
717#[derive(Debug, Clone, Copy, PartialEq, Eq)]
718pub enum FeedKind {
719    /// An RSS/Atom/JSON feed document fetched over http(s).
720    Rss,
721    /// An `at://…/site.standard.publication/…` record pair in somebody's PDS.
722    /// Storable behind `FEATHERREADER_STANDARD_SITE`; **not yet pollable**, so
723    /// [`FeedKind::POLLABLE`] excludes it. Wiring the reader is what moves it.
724    Publication,
725}
726
727impl FeedKind {
728    /// The kinds the scheduler may select. The single place that changes when
729    /// the standard.site reader is wired to the poller.
730    ///
731    /// **This is the canonical home of the at:// exclusion; the other sites
732    /// point here.** It used to be a SQL string predicate in `store`, carrying
733    /// its own copy of the rule — which is how one reader (`count_feeds`) came
734    /// to drift from it unnoticed.
735    ///
736    /// Why an unpollable kind is skipped rather than failed: `poll_feed`
737    /// reaches `net::guarded_get`, whose `check_scheme` refuses any non-http(s)
738    /// scheme, and the standard.site reader is not yet wired to the scheduler.
739    /// Handing such a row to the poller does not leave the feature dormant — it
740    /// manufactures one permanent failure per row, which the public cause
741    /// histogram then reports as an unreachable publisher. Unsupported is not
742    /// broken, and telling those apart is the entire reason a failure cause is
743    /// recorded. (Rows like this exist: subscriptions written by other clients
744    /// before this reader refused the scheme.)
745    ///
746    /// **`store::count_feeds` is deliberately NOT filtered by this.** The
747    /// global ceiling bounds storage on a small box, and an unpollable row
748    /// occupies a row, so it counts against the cap. An earlier version of this
749    /// note listed the readers without naming the exception, which read as
750    /// completeness it did not have: a review found the ceiling consuming
751    /// capacity that appeared on no surface, since `/stats` measures the poller
752    /// and excludes these rows. `/admin/metrics` renders
753    /// `store::unpollable_feeds` for exactly that reason.
754    /// **Adding a kind here makes a population of rows due all at once.**
755    /// `store::due_feeds` orders `next_poll IS NOT NULL, next_poll ASC`, so a
756    /// row with no scheduled poll sorts ahead of every dated one. Rows that
757    /// were never pollable have no schedule, so the boot that reclassifies
758    /// them hands the poller a block of N rows that outrank every regular
759    /// feed until they drain — `ceil(N / batch)` ticks, measured, during which
760    /// `/stats` shows a climbing backlog and nothing logs why. Bounded and
761    /// harmless at ninety feeds; not at ten thousand. Whoever wires the next
762    /// kind should seed or stagger `next_poll` for the rows it admits.
763    pub const POLLABLE: &'static [FeedKind] = &[FeedKind::Rss];
764
765    /// The kinds whose entries the retention **window** applies to.
766    ///
767    /// **Age is the wrong retention policy for an archive, and that is a
768    /// measurement, not a preference.** Three real publications, read through
769    /// `standard_site::fetch` on 2026-09-27: Standard.site's newest document was
770    /// **131 days** old, Annotated's **109** (with its oldest at 373), and minus
771    /// listens' **241**. Against the instance default window of 14 days, every
772    /// one of them stored **zero** rows — a green poll, an empty feed, and an
773    /// info log as the only trace. Long-form publishing is not news-paced.
774    ///
775    /// So a publication is bounded by COUNT instead: `max_entries_per_feed` in
776    /// `insert_entries`, which caps a feed at the newest N plus up to N starred.
777    /// That is a real bound — it is what keeps this from being "retention off" —
778    /// and it is the one that suits a source whose value is its archive.
779    ///
780    /// A generous absolute ceiling still applies (`publication_retention_days`),
781    /// because "not aged out" must not mean "immortal": rows belonging to a feed
782    /// nobody polls any more would otherwise never be reaped at all, and the
783    /// per-feed trim only runs when a poll stores something.
784    pub const AGED: &'static [FeedKind] = &[FeedKind::Rss];
785
786    /// The column value. Stable — it is persisted.
787    pub fn as_str(self) -> &'static str {
788        match self {
789            FeedKind::Rss => "rss",
790            FeedKind::Publication => "publication",
791        }
792    }
793
794    /// A closed vocabulary on the way back in: a kind written by a newer build
795    /// is not silently read as one this build knows.
796    pub fn parse(raw: &str) -> Option<Self> {
797        match raw {
798            "rss" => Some(FeedKind::Rss),
799            "publication" => Some(FeedKind::Publication),
800            _ => None,
801        }
802    }
803
804    /// What a URL will be stored as. The only place the question is asked.
805    ///
806    /// `store::feeds.kind` is a cache of this function, re-derived from the URL
807    /// on every start and on every upsert, so a change here needs no migration
808    /// and SQL cannot hold an opinion of its own about what a row is.
809    ///
810    /// **Case-insensitive on purpose.** An earlier version was case-sensitive,
811    /// with a test pinning that a mixed-case `At://` row IS handed to the
812    /// poller — reasoning that if Rust does not call it an at-URI, neither
813    /// should anything else. That was wrong in the direction that matters: URL
814    /// schemes are case-insensitive, so `Url::parse` folds `At://` to scheme
815    /// `at` and `net::check_scheme` refuses it (the DID form does not parse at
816    /// all). The row could only fail, every tick, forever, and be published in
817    /// the `fetch` bucket as an unreachable publisher. Storing a non-canonical
818    /// spelling is separately refused, because `feeds.url` is UNIQUE.
819    pub fn of(url: &str) -> Self {
820        if crate::atproto::strip_at_prefix(url).is_some() {
821            FeedKind::Publication
822        } else {
823            FeedKind::Rss
824        }
825    }
826}
827
828/// Apply a [`PollOutcome`] to the feed's row: settle the error columns AND
829/// reschedule it. **Both halves, always, from one place.**
830///
831/// The scheduler did this inline. `web::add_subscription` then copied only the
832/// first half, so a poll taken off the scheduler could clear a stale failure's
833/// COUNT while leaving the feed parked on its stale backoff HORIZON — reported
834/// healthy, not polled for up to 24h. And its failures fed `consecutive_errors`
835/// with no reschedule, so repeated Subscribe clicks drove a shared feed to the
836/// 24h ceiling for every subscriber. Two copies of a sequence drift; this is
837/// the one copy.
838///
839/// `cadence` is the interval to use on success; the scheduler derives it from
840/// the feed's hint, a direct caller passes the configured default. A store
841/// failure is logged and the reschedule still attempted, so a hiccup writing
842/// the count cannot strand the feed at a NULL `next_poll` that `due_feeds`
843/// would then re-poll every tick.
844pub async fn settle_poll(
845    pool: &sqlx::SqlitePool,
846    url: &str,
847    outcome: &PollOutcome,
848    cadence: Duration,
849) {
850    let delay = match outcome {
851        // A 304 is a healthy poll: it proves the fetch worked and nothing changed.
852        PollOutcome::Updated { .. } | PollOutcome::NotModified => {
853            if let Err(err) = crate::store::reset_feed_errors(pool, url).await {
854                tracing::warn!(feed = %url, %err, "failed to reset feed error count");
855            }
856            cadence
857        }
858        PollOutcome::Failed {
859            backoff,
860            kind,
861            detail,
862        } => {
863            // Recompute from the feed's REAL consecutive-error count so a
864            // persistently-broken feed climbs toward the ceiling instead of
865            // retrying at the floor forever; fall back to the outcome's floor.
866            match crate::store::bump_feed_errors(pool, url, *kind, detail).await {
867                Ok(count) => backoff_for(count.max(1) as u32),
868                Err(err) => {
869                    tracing::warn!(feed = %url, %err, "failed to bump feed error count; using floor backoff");
870                    *backoff
871                }
872            }
873        }
874    };
875    if let Err(err) = crate::store::set_next_poll(pool, url, delay).await {
876        tracing::error!(feed = %url, %err, "failed to persist next_poll");
877    }
878}
879
880/// Cap on a failure detail, applied where the string is BUILT.
881///
882/// It was originally applied only inside `store::bump_feed_errors`, which
883/// bounded the database row and nothing else — and widening
884/// [`PollOutcome::Failed`] with this field had quietly opened a second sink:
885/// `web::add_subscription` logs `?outcome` at INFO on a user-facing request
886/// path. Bounding at construction bounds every sink, including ones added
887/// later by someone who never reads this comment.
888pub const MAX_FAILURE_DETAIL_CHARS: usize = 300;
889
890/// Render an error chain into a bounded [`PollOutcome::Failed`] detail.
891///
892/// `{e:#}` — the anyhow CHAIN, not just the outermost context. "fetching
893/// https://…" alone says nothing; the cause is the part that would have named
894/// the 304 bug in #159.
895pub fn failure_detail(err: impl std::fmt::Display) -> String {
896    let s = err.to_string();
897    if s.chars().count() <= MAX_FAILURE_DETAIL_CHARS {
898        return s;
899    }
900    s.chars().take(MAX_FAILURE_DETAIL_CHARS).collect()
901}
902
903impl FailureKind {
904    /// The stable string stored in `feeds.last_error_kind` and aggregated on
905    /// `/stats`. Changing one of these silently rewrites history in the
906    /// aggregate, so they are spelled out rather than derived from the variant.
907    pub fn as_str(self) -> &'static str {
908        match self {
909            Self::Fetch => "fetch",
910            Self::Status => "status",
911            Self::Body => "body",
912            Self::Parse => "parse",
913        }
914    }
915
916    /// Read back a persisted `last_error_kind`. `None` for anything this
917    /// version does not know, so a row written by a newer build is not
918    /// silently attributed to a cause this one recognises — the same contract
919    /// [`crate::metrics::Backend::parse`] keeps for the same reason.
920    pub fn parse(raw: &str) -> Option<Self> {
921        match raw {
922            "fetch" => Some(Self::Fetch),
923            "status" => Some(Self::Status),
924            "body" => Some(Self::Body),
925            "parse" => Some(Self::Parse),
926            _ => None,
927        }
928    }
929
930    /// Every variant, so a test can assert over the whole set rather than a
931    /// list that drifts when a variant is added.
932    pub const ALL: [Self; 4] = [Self::Fetch, Self::Status, Self::Body, Self::Parse];
933}
934
935/// Build a `reqwest::Client` configured for polite **and safe** feed fetching.
936///
937/// Callers should build this **once** and share it (connection pooling), then
938/// hand a reference to [`poll_feed`]. Kept here so the fetch policy (UA,
939/// timeout, redirect behaviour) lives with the code that depends on it.
940///
941/// Auto-redirect is **disabled** on purpose: feed URLs are untrusted, so
942/// redirects are followed manually by [`crate::net::guarded_get`], which
943/// re-validates the scheme + resolved IP of every hop (SSRF defence). A client
944/// that silently followed redirects could be bounced onto `169.254.169.254` or
945/// `127.0.0.1` between the guard's check and the connect.
946pub fn build_client() -> Result<Client> {
947    Client::builder()
948        .user_agent(crate::USER_AGENT)
949        .timeout(FETCH_TIMEOUT)
950        .read_timeout(READ_TIMEOUT)
951        // Ignore ambient proxy configuration, for the same reason the pinned
952        // client does: a proxied request hands the hostname to the proxy to
953        // resolve, so `net`'s IP checks never see the address they are meant to
954        // vet. See `net::build_pinned_client`.
955        .no_proxy()
956        // No auto-redirect: net::guarded_get follows + re-validates each hop.
957        .redirect(reqwest::redirect::Policy::none())
958        .build()
959        .context("failed to build feed HTTP client")
960}
961
962/// Compute the backoff for the `n`th consecutive failure (1-based), clamped to
963/// `BACKOFF_MAX`. Exponential in the error count so transient blips retry soon
964/// while a durably-broken feed backs off toward daily.
965///
966/// The scheduler passes the feed's persisted `consecutive_errors` count (see
967/// [`crate::store::bump_feed_errors`]) so a feed that keeps failing actually
968/// climbs toward `BACKOFF_MAX` instead of retrying at the floor forever.
969pub fn backoff_for(consecutive_errors: u32) -> Duration {
970    let n = consecutive_errors.max(1);
971    // Saturating shift: base * 2^(n-1), capped. Avoids overflow for large n.
972    let factor = 1u64.checked_shl(n.saturating_sub(1)).unwrap_or(u64::MAX);
973    let secs = BACKOFF_BASE
974        .as_secs()
975        .saturating_mul(factor)
976        .min(BACKOFF_MAX.as_secs());
977    Duration::from_secs(secs)
978}
979
980/// Fetch, parse, sanitize, normalize, and store a single feed.
981///
982/// Performs a conditional GET using the feed's stored `ETag` / `Last-Modified`.
983/// On `304` it returns [`PollOutcome::NotModified`] without touching entries. On
984/// `200` it parses with `feed-rs`, sanitizes every entry's HTML with `ammonia`,
985/// upserts the feed row (carrying the fresh validators) and inserts new entries
986/// (deduped by GUID). Any fetch/parse error is logged and returned as
987/// [`PollOutcome::Failed`] — it never panics and never propagates as `Err` for
988/// a merely-broken feed, so one bad publisher can't stall the scheduler.
989///
990/// `Err` is reserved for *store* failures (a broken local DB is a real error the
991/// caller should see), not for feed misbehaviour.
992///
993/// `max_entries_per_feed` caps how many entries this feed retains after insert
994/// (newest N by published date); `<= 0` disables the per-feed trim.
995pub async fn poll_feed(
996    pool: &SqlitePool,
997    client: &Client,
998    feed: &Feed,
999    max_entries_per_feed: i64,
1000) -> Result<PollOutcome> {
1001    // --- conditional GET (through the SSRF guard) ----------------------------
1002    // The guard re-validates the scheme + resolved IP of the target and of every
1003    // redirect hop, so a subscribed feed can't bounce the poller onto an
1004    // internal address (cloud metadata / loopback). Conditional-GET validators
1005    // ride along as extra headers.
1006    let mut extra: Vec<(reqwest::header::HeaderName, reqwest::header::HeaderValue)> = Vec::new();
1007    if let Some(etag) = feed.etag.as_deref() {
1008        if let Ok(v) = reqwest::header::HeaderValue::from_str(etag) {
1009            extra.push((IF_NONE_MATCH, v));
1010        }
1011    }
1012    if let Some(lm) = feed.last_modified.as_deref() {
1013        if let Ok(v) = reqwest::header::HeaderValue::from_str(lm) {
1014            extra.push((IF_MODIFIED_SINCE, v));
1015        }
1016    }
1017
1018    let resp = match crate::net::guarded_get(client, &feed.url, &extra).await {
1019        Ok(r) => r,
1020        Err(e) => {
1021            tracing::warn!(feed = %feed.url, error = %e, "feed fetch failed (or blocked by SSRF guard)");
1022            return Ok(PollOutcome::Failed {
1023                backoff: backoff_for(1),
1024                kind: FailureKind::Fetch,
1025                detail: failure_detail(format!("{e:#}")),
1026            });
1027        }
1028    };
1029
1030    let status = resp.status();
1031    if status == StatusCode::NOT_MODIFIED {
1032        tracing::debug!(feed = %feed.url, "feed not modified (304)");
1033        // Bump last_polled/next_poll only; leave validators + entries untouched.
1034        touch_polled(pool, &feed.url, None, None)
1035            .await
1036            .with_context(|| format!("touch_polled after 304 for {}", feed.url))?;
1037        return Ok(PollOutcome::NotModified);
1038    }
1039    if !status.is_success() {
1040        tracing::warn!(feed = %feed.url, %status, "feed returned non-success status");
1041        return Ok(PollOutcome::Failed {
1042            backoff: backoff_for(1),
1043            kind: FailureKind::Status,
1044            detail: failure_detail(status),
1045        });
1046    }
1047
1048    // Capture validators for the *next* conditional GET before consuming body.
1049    let new_etag = header_str(resp.headers().get(ETAG));
1050    let new_last_modified = header_str(resp.headers().get(LAST_MODIFIED));
1051
1052    // Stream the body with a hard byte cap, aborting mid-stream if it exceeds
1053    // it. We never trust Content-Length: reqwest's gzip layer strips it, so a
1054    // small gzip bomb could otherwise inflate to GBs before any size check.
1055    let body = match crate::net::read_capped(resp).await {
1056        Ok(b) => b,
1057        Err(e) => {
1058            tracing::warn!(feed = %feed.url, error = %e, "feed body rejected (too large / read error)");
1059            return Ok(PollOutcome::Failed {
1060                backoff: backoff_for(1),
1061                kind: FailureKind::Body,
1062                detail: failure_detail(format!("{e:#}")),
1063            });
1064        }
1065    };
1066
1067    // --- parse (malformed feed => log + skip, never panic) -------------------
1068    let parsed = match feed_rs::parser::parse(&body[..]) {
1069        Ok(f) => f,
1070        Err(e) => {
1071            tracing::warn!(feed = %feed.url, error = %e, "malformed feed; skipping");
1072            return Ok(PollOutcome::Failed {
1073                backoff: backoff_for(1),
1074                kind: FailureKind::Parse,
1075                detail: failure_detail(format!("{e:#}")),
1076            });
1077        }
1078    };
1079
1080    // --- normalize + sanitize ------------------------------------------------
1081    let (title, site_url) = feed_metadata(&parsed);
1082    let new_feed = NewFeed {
1083        url: feed.url.clone(),
1084        title,
1085        site_url,
1086        etag: new_etag,
1087        last_modified: new_last_modified,
1088        last_polled: Some(now_rfc3339()),
1089        next_poll: None, // the scheduler owns cadence; leave it to set next_poll.
1090    };
1091
1092    let entries: Vec<NewEntry> = parsed.entries.iter().map(normalize_entry).collect();
1093
1094    // --- store (a store failure IS a real error) -----------------------------
1095    let feed_id = store::upsert_feed(pool, &new_feed)
1096        .await
1097        .with_context(|| format!("upsert_feed for {}", feed.url))?;
1098    let n = store::insert_entries(pool, feed_id, &entries, max_entries_per_feed)
1099        .await
1100        .with_context(|| format!("insert_entries for {}", feed.url))?;
1101
1102    tracing::info!(feed = %feed.url, entries = n, "feed polled");
1103    Ok(PollOutcome::Updated { new_entries: n })
1104}
1105
1106/// Bump `last_polled` (and optionally validators) without changing entries —
1107/// used on the `304 Not Modified` path.
1108async fn touch_polled(
1109    pool: &SqlitePool,
1110    url: &str,
1111    etag: Option<String>,
1112    last_modified: Option<String>,
1113) -> Result<()> {
1114    // `None` means "keep current" — upsert_feed COALESCEs the validators, so a
1115    // 304 that repeats no headers leaves the stored ones untouched. This used to
1116    // re-read the row and re-supply them by hand because the upsert clobbered
1117    // unconditionally; the read-modify-write is gone now that the upsert is
1118    // honest, and with it a race where a concurrent poll's validators could be
1119    // read here and written back stale.
1120    let nf = NewFeed {
1121        url: url.to_string(),
1122        etag,
1123        last_modified,
1124        last_polled: Some(now_rfc3339()),
1125        ..Default::default()
1126    };
1127    store::upsert_feed(pool, &nf).await?;
1128    Ok(())
1129}
1130
1131/// Extract `(title, site_url)` from a parsed feed. `site_url` prefers an
1132/// `alternate`/no-rel HTML link over the feed's self link.
1133fn feed_metadata(parsed: &RawFeed) -> (Option<String>, Option<String>) {
1134    let title = parsed.title.as_ref().map(text_plain);
1135    let site_url = parsed
1136        .links
1137        .iter()
1138        // Prefer an explicit human-facing page: rel="alternate" or no rel at all.
1139        .find(|l| {
1140            l.rel.as_deref() == Some("alternate")
1141                || (l.rel.is_none()
1142                    && l.media_type.as_deref() != Some("application/rss+xml")
1143                    && l.media_type.as_deref() != Some("application/atom+xml"))
1144        })
1145        .or_else(|| {
1146            parsed
1147                .links
1148                .iter()
1149                .find(|l| l.rel.as_deref() != Some("self"))
1150        })
1151        .or_else(|| parsed.links.first())
1152        .map(|l| l.href.clone());
1153    (title, site_url)
1154}
1155
1156/// Turn a parsed [`RawEntry`] into the store's [`NewEntry`], sanitizing HTML.
1157///
1158/// Content preference: full `content` body, else `summary`. Whichever is chosen
1159/// is **always** passed through [`sanitize_html`] before storage. GUID falls
1160/// back to the entry link, then to a stable hash of title+link, so an entry
1161/// missing an `id` still deduplicates instead of being re-inserted forever.
1162fn normalize_entry(e: &RawEntry) -> NewEntry {
1163    let url = entry_link(e);
1164    let content_html = e
1165        .content
1166        .as_ref()
1167        .and_then(|c| c.body.as_deref())
1168        .or_else(|| e.summary.as_ref().map(|t| t.content.as_str()))
1169        .map(sanitize_html);
1170
1171    // GUID may use the raw link (dedup key only, never rendered), so prefer the
1172    // entry's first raw link for identity even when it's not a safe href.
1173    let guid = if !e.id.trim().is_empty() {
1174        e.id.trim().to_string()
1175    } else if let Some(link) = raw_entry_link(e) {
1176        link
1177    } else {
1178        // Last resort: derive a stable id so re-fetches dedup rather than dupe.
1179        stable_guid(e)
1180    };
1181
1182    NewEntry {
1183        guid,
1184        url,
1185        title: e.title.as_ref().map(text_plain),
1186        author: entry_author(e),
1187        published: entry_time(e),
1188        content_html,
1189        fetched_at: None, // store defaults to "now".
1190    }
1191}
1192
1193/// The raw best-permalink URL for an entry (no scheme filtering) — used only as
1194/// a dedup GUID, never rendered as an href.
1195fn raw_entry_link(e: &RawEntry) -> Option<String> {
1196    e.links
1197        .iter()
1198        .find(|l| l.rel.as_deref() == Some("alternate") || l.rel.is_none())
1199        .or_else(|| e.links.first())
1200        .map(|l| l.href.clone())
1201}
1202
1203/// The best display/permalink URL for an entry, **scheme-allow-listed** so it is
1204/// safe to render as an `href`: prefer `rel="alternate"` or a no-rel link, else
1205/// the first link — but only if it is an `http`/`https` URL. A `javascript:` or
1206/// `data:` permalink (a stored-XSS vector that survives HTML escaping, since it
1207/// carries no HTML-special characters) is dropped here at ingest, before it can
1208/// ever reach the store or a template.
1209fn entry_link(e: &RawEntry) -> Option<String> {
1210    raw_entry_link(e).and_then(|href| crate::net::safe_link(&href))
1211}
1212
1213/// First author name, if any.
1214fn entry_author(e: &RawEntry) -> Option<String> {
1215    e.authors.first().map(|p| p.name.clone())
1216}
1217
1218/// Best publication time (published, else updated) as an RFC3339 string.
1219fn entry_time(e: &RawEntry) -> Option<String> {
1220    e.published.or(e.updated).map(fmt_time)
1221}
1222
1223/// Extract the plain string content of a feed [`Text`] node.
1224fn text_plain(t: &Text) -> String {
1225    t.content.trim().to_string()
1226}
1227
1228/// Sanitize hostile feed **HTML** with ammonia's whitelist cleaner. Applied to
1229/// every RSS/Atom entry body unconditionally, because every one of them is
1230/// markup.
1231///
1232/// **Not for plain text.** An earlier version of this comment claimed it was
1233/// "safe on plain text too (it will simply escape/strip as needed)". It is
1234/// not: `clean` PARSES its input, so a bare `<` in prose swallows the rest —
1235/// `"if x<y then z"` comes back as `"if x"`. That sentence is how a
1236/// plain-text field got run through here once already. Use
1237/// [`plain_text_to_html`].
1238pub(crate) fn sanitize_html(raw: &str) -> String {
1239    ammonia::clean(raw)
1240}
1241
1242/// Render **plain text** into the HTML the `content_html` column holds.
1243///
1244/// **Not [`sanitize_html`].** `ammonia::clean` parses its input as markup, so a
1245/// bare `<` in prose swallows the rest: measured here, `"if x<y then z"` comes
1246/// back as `"if x"`. That is correct for an RSS body, which IS markup, and
1247/// silent data loss for a field a lexicon defines as text. Escape first, then
1248/// add the only markup this needs — line breaks, which the column's consumer
1249/// renders as HTML and would otherwise collapse.
1250pub(crate) fn plain_text_to_html(raw: &str) -> String {
1251    let escaped = raw
1252        .replace('&', "&amp;")
1253        .replace('<', "&lt;")
1254        .replace('>', "&gt;");
1255    // Safe by order: every `<` from the input is already `&lt;` before this
1256    // adds a real tag.
1257    escaped.replace('\n', "<br>")
1258}
1259
1260/// Format a chrono timestamp as RFC3339 (UTC, seconds precision) to match the
1261/// store's string columns.
1262pub(crate) fn fmt_time(dt: DateTime<Utc>) -> String {
1263    dt.to_rfc3339_opts(SecondsFormat::Secs, true)
1264}
1265
1266/// "Now" in the store's RFC3339 shape.
1267fn now_rfc3339() -> String {
1268    Utc::now().to_rfc3339_opts(SecondsFormat::Secs, true)
1269}
1270
1271/// A stable GUID derived from an entry's title + first link, for feeds that
1272/// supply neither an id nor a usable link id. Deterministic so re-fetches dedup.
1273fn stable_guid(e: &RawEntry) -> String {
1274    use std::hash::{Hash, Hasher};
1275    let mut h = std::collections::hash_map::DefaultHasher::new();
1276    e.title.as_ref().map(|t| t.content.as_str()).hash(&mut h);
1277    e.links.first().map(|l| l.href.as_str()).hash(&mut h);
1278    e.summary.as_ref().map(|s| s.content.as_str()).hash(&mut h);
1279    format!("featherreader:synthetic:{:016x}", h.finish())
1280}
1281
1282/// Decode an HTTP header value to an owned `String`, dropping non-UTF-8 values.
1283fn header_str(v: Option<&reqwest::header::HeaderValue>) -> Option<String> {
1284    v.and_then(|h| h.to_str().ok()).map(str::to_string)
1285}
1286
1287/// Discover a feed URL from a site's HTML via
1288/// `<link rel="alternate" type="application/rss+xml|atom+xml" href="…">`.
1289///
1290/// Returns the first RSS/Atom autodiscovery link found, resolved against the
1291/// page URL if the `href` is relative. This is what lets a user paste a *site*
1292/// URL and have FeatherReader find the actual feed ("subscribe by URL").
1293/// Returns `None` if the HTML carries no autodiscovery link.
1294///
1295/// The `base` is the URL the HTML was fetched from, used to resolve relative
1296/// `href`s. Pass `None` to only accept absolute hrefs.
1297pub fn discover_feed(site_html: &str, base: Option<&Url>) -> Option<Url> {
1298    // Parse the HTML with html5ever (via ammonia's dependency graph is separate;
1299    // use a light hand-rolled scan over <link> tags to avoid a new dependency).
1300    // We look for <link ...> elements whose rel contains "alternate" and whose
1301    // type is an RSS/Atom feed media type, and take the href.
1302    for tag in link_tags(site_html) {
1303        let rel = attr(&tag, "rel").unwrap_or_default().to_ascii_lowercase();
1304        let typ = attr(&tag, "type").unwrap_or_default().to_ascii_lowercase();
1305        let is_feed_type = typ.contains("application/rss+xml")
1306            || typ.contains("application/atom+xml")
1307            || typ.contains("application/feed+json")
1308            || typ.contains("application/json");
1309        // rel="alternate" is the standard; be lenient and also accept a bare
1310        // feed type with any rel, but require the feed media type either way.
1311        let rel_ok = rel.split_whitespace().any(|r| r == "alternate") || rel.is_empty();
1312        if is_feed_type && rel_ok {
1313            if let Some(href) = attr(&tag, "href") {
1314                let href = href.trim();
1315                if href.is_empty() {
1316                    continue;
1317                }
1318                // Absolute URL wins directly; otherwise resolve against `base`.
1319                // Either way, only http(s): the href is publisher-controlled and
1320                // `Url::parse` accepts any scheme, so this is where an `at://`
1321                // (or `file:`, `javascript:`) alternate would otherwise become
1322                // the URL the add path stores — after its input gate has run.
1323                // Skip, don't stop: a later real feed link still wins.
1324                let resolved = match Url::parse(href) {
1325                    Ok(u) => Some(u),
1326                    Err(_) => base.and_then(|b| b.join(href).ok()),
1327                };
1328                match resolved {
1329                    Some(u) if matches!(u.scheme(), "http" | "https") => return Some(u),
1330                    _ => continue,
1331                }
1332            }
1333        }
1334    }
1335    None
1336}
1337
1338/// Extract the raw text of every `<link ...>` tag (self-closing or not) from an
1339/// HTML string. A deliberately small, allocation-light scan — feed
1340/// autodiscovery does not need a full DOM, and avoiding one keeps the dependency
1341/// surface minimal (design bias: boring, small-dependency).
1342fn link_tags(html: &str) -> Vec<String> {
1343    let mut out = Vec::new();
1344    let bytes = html.as_bytes();
1345    let lower = html.to_ascii_lowercase();
1346    let mut search_from = 0usize;
1347    while let Some(rel_idx) = lower[search_from..].find("<link") {
1348        let start = search_from + rel_idx;
1349        // Ensure it's a tag boundary ("<link" followed by whitespace, '>' or '/').
1350        let after = bytes.get(start + 5).copied();
1351        let boundary = matches!(after, Some(b) if b == b' ' || b == b'\t' || b == b'\n' || b == b'\r' || b == b'>' || b == b'/');
1352        if !boundary {
1353            search_from = start + 5;
1354            continue;
1355        }
1356        // Find the closing '>' for this tag.
1357        if let Some(end_rel) = html[start..].find('>') {
1358            let end = start + end_rel;
1359            out.push(html[start..=end].to_string());
1360            search_from = end + 1;
1361        } else {
1362            break;
1363        }
1364    }
1365    out
1366}
1367
1368/// Read an attribute value from a single tag string, handling both single- and
1369/// double-quoted values. Case-insensitive attribute name match.
1370fn attr(tag: &str, name: &str) -> Option<String> {
1371    let lower = tag.to_ascii_lowercase();
1372    let needle = format!("{name}=");
1373    let mut from = 0usize;
1374    while let Some(rel) = lower[from..].find(&needle) {
1375        let name_start = from + rel;
1376        // Guard against matching a suffix of a longer attribute name
1377        // (e.g. matching "type=" inside "mytype="): the char before must be a
1378        // tag/whitespace boundary.
1379        let ok_prefix = name_start == 0
1380            || matches!(
1381                tag.as_bytes().get(name_start - 1),
1382                Some(b' ') | Some(b'\t') | Some(b'\n') | Some(b'\r') | Some(b'<')
1383            );
1384        let val_start = name_start + needle.len();
1385        if !ok_prefix {
1386            from = val_start;
1387            continue;
1388        }
1389        let rest = &tag[val_start..];
1390        let quote = rest.chars().next();
1391        let value = match quote {
1392            Some('"') => rest[1..].split('"').next(),
1393            Some('\'') => rest[1..].split('\'').next(),
1394            // Unquoted: read up to whitespace, '>' or '/'.
1395            _ => rest
1396                .split(|c: char| c.is_whitespace() || c == '>' || c == '/')
1397                .next(),
1398        };
1399        return value.map(str::to_string);
1400    }
1401    None
1402}
1403
1404#[cfg(test)]
1405mod tests {
1406    use super::*;
1407
1408    const RSS_SAMPLE: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
1409<rss version="2.0">
1410  <channel>
1411    <title>Example RSS Feed</title>
1412    <link>https://example.com/</link>
1413    <description>An example feed for tests</description>
1414    <item>
1415      <title>First post</title>
1416      <link>https://example.com/first</link>
1417      <guid>https://example.com/first</guid>
1418      <author>alice@example.com (Alice)</author>
1419      <pubDate>Fri, 10 Jul 2026 08:00:00 GMT</pubDate>
1420      <description><![CDATA[<p>Hello <b>world</b>.</p><script>alert('xss')</script><img src="x" onerror="alert(1)">]]></description>
1421    </item>
1422    <item>
1423      <title>Second post</title>
1424      <link>https://example.com/second</link>
1425      <guid>guid-second</guid>
1426      <pubDate>Sat, 11 Jul 2026 08:00:00 GMT</pubDate>
1427      <description><![CDATA[<a href="javascript:alert(1)">click</a><a href="https://ok.example/">ok</a>]]></description>
1428    </item>
1429  </channel>
1430</rss>"#;
1431
1432    const ATOM_SAMPLE: &str = r#"<?xml version="1.0" encoding="utf-8"?>
1433<feed xmlns="http://www.w3.org/2005/Atom">
1434  <title>Example Atom Feed</title>
1435  <link rel="alternate" href="https://atom.example.com/"/>
1436  <link rel="self" href="https://atom.example.com/feed.xml"/>
1437  <id>urn:uuid:feed-1</id>
1438  <updated>2026-07-11T08:00:00Z</updated>
1439  <entry>
1440    <title>Atom entry</title>
1441    <id>urn:uuid:entry-1</id>
1442    <link rel="alternate" href="https://atom.example.com/a"/>
1443    <author><name>Bob</name></author>
1444    <updated>2026-07-11T08:00:00Z</updated>
1445    <content type="html"><![CDATA[<p>Safe <em>text</em>.</p><script>steal()</script><iframe src="evil"></iframe>]]></content>
1446  </entry>
1447</feed>"#;
1448
1449    /// [`ATOM_SAMPLE`] with the `self` link before the `alternate` one.
1450    const ATOM_SELF_FIRST: &str = r#"<?xml version="1.0" encoding="utf-8"?>
1451<feed xmlns="http://www.w3.org/2005/Atom">
1452  <title>Example Atom Feed</title>
1453  <link rel="self" href="https://atom.example.com/feed.xml"/>
1454  <link rel="alternate" href="https://atom.example.com/"/>
1455  <id>urn:uuid:feed-1</id>
1456  <updated>2026-07-11T08:00:00Z</updated>
1457  <entry>
1458    <title>Atom entry</title>
1459    <id>urn:uuid:entry-1</id>
1460    <link rel="alternate" href="https://atom.example.com/a"/>
1461    <author><name>Bob</name></author>
1462    <updated>2026-07-11T08:00:00Z</updated>
1463    <content type="html"><![CDATA[<p>Safe <em>text</em>.</p><script>steal()</script><iframe src="evil"></iframe>]]></content>
1464  </entry>
1465</feed>"#;
1466
1467    /// Parse a static RSS sample through feed-rs + our normalize/sanitize path
1468    /// (no network) and assert the entries come out sanitized and well-shaped.
1469    #[test]
1470    fn rss_parses_and_sanitizes() {
1471        let parsed = feed_rs::parser::parse(RSS_SAMPLE.as_bytes()).expect("RSS should parse");
1472        assert_eq!(
1473            parsed.title.as_ref().map(text_plain).as_deref(),
1474            Some("Example RSS Feed")
1475        );
1476        assert_eq!(parsed.entries.len(), 2);
1477
1478        let (title, site) = feed_metadata(&parsed);
1479        assert_eq!(title.as_deref(), Some("Example RSS Feed"));
1480        assert_eq!(site.as_deref(), Some("https://example.com/"));
1481
1482        let e0 = normalize_entry(&parsed.entries[0]);
1483        assert_eq!(e0.guid, "https://example.com/first");
1484        assert_eq!(e0.title.as_deref(), Some("First post"));
1485        assert_eq!(e0.url.as_deref(), Some("https://example.com/first"));
1486        assert!(e0.published.is_some());
1487        let html0 = e0.content_html.expect("content present");
1488        // Sanitized: benign markup kept, script + onerror stripped.
1489        assert!(html0.contains("Hello"));
1490        assert!(html0.contains("<b>world</b>") || html0.contains("<b>"));
1491        assert!(!html0.to_ascii_lowercase().contains("<script"));
1492        assert!(!html0.to_ascii_lowercase().contains("onerror"));
1493        assert!(!html0.to_ascii_lowercase().contains("alert"));
1494
1495        // Second entry: javascript: URL scrubbed, safe link kept.
1496        let e1 = normalize_entry(&parsed.entries[1]);
1497        assert_eq!(e1.guid, "guid-second");
1498        let html1 = e1.content_html.expect("content present");
1499        assert!(!html1.to_ascii_lowercase().contains("javascript:"));
1500        assert!(html1.contains("https://ok.example/"));
1501    }
1502
1503    /// Same, for an Atom sample: alternate link is the site URL, dangerous
1504    /// elements are stripped from entry content.
1505    /// **`rel="alternate"` is preferred over a `rel="self"` listed FIRST.** In
1506    /// `ATOM_SAMPLE` the alternate link is already first, so "prefer alternate"
1507    /// and "take the first link" were indistinguishable; the selection could
1508    /// be replaced by `links.first()` with the suite green. A feed listing
1509    /// `self` first — very common in Atom — would store the feed XML URL as
1510    /// the subscription's `siteUrl`, published to the reader's PDS.
1511    #[test]
1512    fn atom_prefers_alternate_over_a_self_link_listed_first() {
1513        let parsed = feed_rs::parser::parse(ATOM_SELF_FIRST.as_bytes()).expect("Atom should parse");
1514        let (title, site) = feed_metadata(&parsed);
1515        assert_eq!(title.as_deref(), Some("Example Atom Feed"));
1516        // alternate link preferred over rel="self".
1517        assert_eq!(site.as_deref(), Some("https://atom.example.com/"));
1518    }
1519
1520    #[test]
1521    fn atom_parses_and_sanitizes() {
1522        let parsed = feed_rs::parser::parse(ATOM_SAMPLE.as_bytes()).expect("Atom should parse");
1523        let (title, site) = feed_metadata(&parsed);
1524        assert_eq!(title.as_deref(), Some("Example Atom Feed"));
1525        // alternate link preferred over rel="self".
1526        assert_eq!(site.as_deref(), Some("https://atom.example.com/"));
1527
1528        assert_eq!(parsed.entries.len(), 1);
1529        let e = normalize_entry(&parsed.entries[0]);
1530        assert_eq!(e.guid, "urn:uuid:entry-1");
1531        assert_eq!(e.title.as_deref(), Some("Atom entry"));
1532        assert_eq!(e.author.as_deref(), Some("Bob"));
1533        assert_eq!(e.url.as_deref(), Some("https://atom.example.com/a"));
1534        let html = e.content_html.expect("content present");
1535        assert!(html.contains("Safe"));
1536        assert!(!html.to_ascii_lowercase().contains("<script"));
1537        assert!(!html.to_ascii_lowercase().contains("<iframe"));
1538    }
1539
1540    /// **The rkey obeys all of atproto's record-key rules, not just the
1541    /// charset.** `.` and `..` are reserved and the length is 1..=512; the
1542    /// charset alone admitted both and a 10 000-character key, into a UNIQUE
1543    /// column and the user's public PDS. The rule is the one the repo's own
1544    /// TID tests already state.
1545    #[test]
1546    fn an_rkey_must_obey_atprotos_length_and_dot_rules() {
1547        let uri = |rkey: &str| {
1548            format!("at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/{rkey}")
1549        };
1550        assert!(
1551            !is_storable_feed_url(&uri("."), true),
1552            "`.` is a reserved rkey"
1553        );
1554        assert!(
1555            !is_storable_feed_url(&uri(".."), true),
1556            "`..` is a reserved rkey"
1557        );
1558        assert!(
1559            !is_storable_feed_url(&uri(&"a".repeat(513)), true),
1560            "an rkey over 512 bytes was accepted"
1561        );
1562        assert!(
1563            is_storable_feed_url(&uri(&"a".repeat(512)), true),
1564            "an rkey of exactly 512 bytes is valid"
1565        );
1566        assert!(is_storable_feed_url(&uri("3lab2c4d5e6f7g8h"), true));
1567    }
1568
1569    /// **Every `KNOWN_PROVIDERS` row is pinned by its own reason.**
1570    ///
1571    /// The provider test used URLs that the generic heuristics catch on their
1572    /// own — Substack's `/feed/private/` is also a path marker, Patreon's
1573    /// `?auth=…` an opaque secret key — and never asserted the reason. With
1574    /// 17 of 18 rows deleted, the suite stayed green. The provider layer runs
1575    /// FIRST so its specific reason wins; asserting the reason pins each row
1576    /// even where a generic rule would still refuse the URL. Values are kept
1577    /// short and plain so the generic query rule (`value_is_opaque`) does not
1578    /// fire — most of these are refused by the provider row alone.
1579    #[test]
1580    fn each_known_provider_is_caught_by_its_own_row() {
1581        for (url, reason) in [
1582            (
1583                "https://author.substack.com/feed/private/x",
1584                "Substack private feed",
1585            ),
1586            (
1587                "https://www.patreon.com/rss/creator?auth=ab",
1588                "Patreon member feed",
1589            ),
1590            ("https://blog.ghost.io/rss/?uuid=x", "Ghost members feed"),
1591            (
1592                "https://buttondown.email/me/rss?token=x",
1593                "Buttondown premium feed",
1594            ),
1595            (
1596                "https://buttondown.com/me/rss?token=x",
1597                "Buttondown premium feed",
1598            ),
1599            (
1600                "https://rss.beehiiv.com/feeds/x.xml?token=x",
1601                "Beehiiv premium feed",
1602            ),
1603            (
1604                "https://example.memberful.com/feed",
1605                "Memberful members feed",
1606            ),
1607            ("https://example.pico.link/feed", "Pico member feed"),
1608            ("https://steadyhq.com/rss/example", "Steady member feed"),
1609            (
1610                "https://example.supercast.com/feed",
1611                "Supercast private podcast",
1612            ),
1613            (
1614                "https://example.supercast.tech/feed",
1615                "Supercast private podcast",
1616            ),
1617            (
1618                "https://example.supportingcast.fm/feed",
1619                "Supporting Cast private podcast",
1620            ),
1621            (
1622                "https://feeds.redcircle.com/x?private=1",
1623                "RedCircle private podcast",
1624            ),
1625            (
1626                "https://feeds.megaphone.fm/x?token=x",
1627                "Megaphone private podcast",
1628            ),
1629            (
1630                "https://feeds.acast.com/public/shows/x?token=x",
1631                "Acast+ private podcast",
1632            ),
1633            (
1634                "https://omny.fm/shows/x/playlists/podcast.rss?token=x",
1635                "Omny private podcast",
1636            ),
1637            (
1638                "https://podcasts.apple.com/feed/x?token=x",
1639                "Apple subscriber podcast",
1640            ),
1641            (
1642                "https://anchor.spotify.com/s/x/podcast/rss?token=x",
1643                "Spotify subscriber podcast",
1644            ),
1645        ] {
1646            match classify_feed_privacy(url) {
1647                FeedPrivacy::Private(r) => {
1648                    assert_eq!(r, reason, "{url} was refused by another rule")
1649                }
1650                FeedPrivacy::Public => panic!("{url} was not refused at all"),
1651            }
1652        }
1653    }
1654
1655    /// **Plain text is escaped, not sanitised.** `ammonia::clean` parses its
1656    /// input as markup, so a `<` in prose swallows everything after it:
1657    /// measured in this tree, `"if x<y then z"` becomes `"if x"`. That is the
1658    /// right function for an RSS body (which IS markup) and exactly the wrong
1659    /// one for a field the lexicon defines as plain text — it silently deletes
1660    /// the reader's content.
1661    #[test]
1662    fn plain_text_is_escaped_rather_than_swallowed() {
1663        assert_eq!(
1664            plain_text_to_html("Vec<String> is a type"),
1665            "Vec&lt;String&gt; is a type"
1666        );
1667        assert_eq!(plain_text_to_html("if x<y then z"), "if x&lt;y then z");
1668        assert_eq!(plain_text_to_html("a & b"), "a &amp; b");
1669        // Line structure survives into a field rendered as HTML.
1670        assert_eq!(plain_text_to_html("one\ntwo"), "one<br>two");
1671        // And it is still safe: the escaping happens before any markup is added.
1672        let hostile = plain_text_to_html("<script>alert(1)</script>");
1673        assert!(!hostile.contains("<script"), "{hostile}");
1674    }
1675
1676    #[test]
1677    fn discover_finds_rss_link() {
1678        let html = r#"<!doctype html><html><head>
1679            <title>Blog</title>
1680            <link rel="stylesheet" href="/style.css">
1681            <link rel="alternate" type="application/rss+xml" title="RSS" href="/feed.xml">
1682        </head><body>hi</body></html>"#;
1683        let base = Url::parse("https://blog.example.com/").unwrap();
1684        let found = discover_feed(html, Some(&base)).expect("should discover feed");
1685        assert_eq!(found.as_str(), "https://blog.example.com/feed.xml");
1686    }
1687
1688    #[test]
1689    fn discover_finds_atom_absolute_link() {
1690        let html = r#"<head><link rel="alternate" type="application/atom+xml" href="https://x.example/atom"></head>"#;
1691        let found = discover_feed(html, None).expect("should discover absolute feed");
1692        assert_eq!(found.as_str(), "https://x.example/atom");
1693    }
1694
1695    /// **Autodiscovery only ever yields an http(s) URL.**
1696    ///
1697    /// The href is publisher-controlled and `Url::parse` accepts any scheme, so
1698    /// a page could hand the add path an `at://` publication URI (or anything
1699    /// else) that the user never typed — and the add path's input gate has
1700    /// already run by then. A non-http(s) alternate is skipped, not returned,
1701    /// so a later real feed link still wins.
1702    #[test]
1703    fn discover_skips_a_non_http_alternate() {
1704        let at_link = r#"<link rel="alternate" type="application/rss+xml" href="at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h">"#;
1705        assert!(
1706            discover_feed(&format!("<head>{at_link}</head>"), None).is_none(),
1707            "an at:// alternate was handed back as a feed URL"
1708        );
1709        let ftp_link =
1710            r#"<link rel="alternate" type="application/atom+xml" href="ftp://x.example/atom">"#;
1711        assert!(discover_feed(&format!("<head>{ftp_link}</head>"), None).is_none());
1712
1713        let real =
1714            r#"<link rel="alternate" type="application/atom+xml" href="https://x.example/atom">"#;
1715        let found = discover_feed(&format!("<head>{at_link}{real}</head>"), None)
1716            .expect("the http(s) link after a skipped one must still be found");
1717        assert_eq!(found.as_str(), "https://x.example/atom");
1718    }
1719
1720    #[test]
1721    fn discover_returns_none_without_feed_link() {
1722        let html =
1723            r#"<head><link rel="stylesheet" href="/s.css"><link rel="icon" href="/f.ico"></head>"#;
1724        assert!(discover_feed(html, None).is_none());
1725    }
1726
1727    #[test]
1728    fn synthetic_guid_is_stable_and_dedups() {
1729        // An item with neither guid nor link: feed-rs will hash the link (absent)
1730        // to a UUID id, but to exercise *our* synthetic fallback we clear the id
1731        // on the parsed entry and confirm normalize yields a deterministic guid.
1732        let xml = r#"<?xml version="1.0"?><rss version="2.0"><channel>
1733            <title>t</title>
1734            <item><title>only a title</title><description>body</description></item>
1735        </channel></rss>"#;
1736        let mut parsed = feed_rs::parser::parse(xml.as_bytes()).expect("parse");
1737        parsed.entries[0].id.clear();
1738        parsed.entries[0].links.clear();
1739        let g1 = normalize_entry(&parsed.entries[0]).guid;
1740        let g2 = normalize_entry(&parsed.entries[0]).guid;
1741        assert_eq!(g1, g2);
1742        assert!(g1.starts_with("featherreader:synthetic:"));
1743    }
1744
1745    #[test]
1746    fn entry_link_scheme_allowlist_neutralizes_javascript() {
1747        // An entry whose only link is a javascript: URL must yield no href.
1748        let xml = r#"<?xml version="1.0"?><rss version="2.0"><channel>
1749            <title>t</title>
1750            <item>
1751              <title>evil</title>
1752              <link>javascript:alert(document.domain)</link>
1753              <guid>evil-1</guid>
1754            </item>
1755        </channel></rss>"#;
1756        let parsed = feed_rs::parser::parse(xml.as_bytes()).expect("parse");
1757        let e = normalize_entry(&parsed.entries[0]);
1758        // url is dropped (not a safe http(s) link)…
1759        assert_eq!(e.url, None);
1760        // …but the entry still dedups (guid preserved from <guid>).
1761        assert_eq!(e.guid, "evil-1");
1762
1763        // A data: URL is likewise dropped.
1764        let xml2 = r#"<?xml version="1.0"?><rss version="2.0"><channel>
1765            <title>t</title>
1766            <item><title>d</title><link>data:text/html,<script>1</script></link><guid>d1</guid></item>
1767        </channel></rss>"#;
1768        let parsed2 = feed_rs::parser::parse(xml2.as_bytes()).expect("parse");
1769        let e2 = normalize_entry(&parsed2.entries[0]);
1770        assert_eq!(e2.url, None);
1771
1772        // A normal https link survives.
1773        let xml3 = r#"<?xml version="1.0"?><rss version="2.0"><channel>
1774            <title>t</title>
1775            <item><title>ok</title><link>https://ok.example/post</link><guid>ok1</guid></item>
1776        </channel></rss>"#;
1777        let parsed3 = feed_rs::parser::parse(xml3.as_bytes()).expect("parse");
1778        let e3 = normalize_entry(&parsed3.entries[0]);
1779        assert_eq!(e3.url.as_deref(), Some("https://ok.example/post"));
1780    }
1781
1782    #[test]
1783    fn classify_privacy_flags_secret_urls_across_providers() {
1784        // --- Known providers: newsletters ---
1785        // Substack private feed path.
1786        assert!(
1787            classify_feed_privacy("https://author.substack.com/feed/private/deadbeefcafe1234")
1788                .is_private()
1789        );
1790        // Patreon ?auth= member feed.
1791        assert!(classify_feed_privacy(
1792            "https://www.patreon.com/rss/author?auth=Zm9vYmFyc2VjcmV0dG9rZW4"
1793        )
1794        .is_private());
1795        // Ghost members feed via ?uuid=.
1796        assert!(classify_feed_privacy(
1797            "https://blog.ghost.io/rss/?uuid=1f2e3d4c-5b6a-7089-90ab-cdef01234567"
1798        )
1799        .is_private());
1800
1801        // --- Known providers: private podcasts ---
1802        // Supporting Cast tokened podcast feed.
1803        assert!(classify_feed_privacy(
1804            "https://feeds.supportingcast.fm/show/abcdef0123456789abcdef01"
1805        )
1806        .is_private());
1807        // Supercast private podcast (host alone is enough).
1808        assert!(classify_feed_privacy("https://feeds.supercast.com/12345/rss").is_private());
1809
1810        // --- Generic, provider-agnostic heuristic ---
1811        // Named credential query params with an opaque value.
1812        assert!(
1813            classify_feed_privacy("https://example.com/feed?token=Zm9vYmFyc2VjcmV0").is_private()
1814        );
1815        assert!(
1816            classify_feed_privacy("https://example.com/feed?key=Zm9vYmFyc2VjcmV0").is_private()
1817        );
1818        assert!(
1819            classify_feed_privacy("https://example.com/feed?secret=Zm9vYmFyc2VjcmV0").is_private()
1820        );
1821        // Userinfo credentials in the authority.
1822        assert!(classify_feed_privacy("https://user:pass@example.com/feed").is_private());
1823        // A `/private/` path segment on an unknown host.
1824        assert!(classify_feed_privacy("https://blog.example.com/private/rss").is_private());
1825        // `/members/` path convention.
1826        assert!(classify_feed_privacy("https://news.example.com/members/feed.xml").is_private());
1827        // A high-entropy opaque token embedded in the path with no telltale name.
1828        assert!(
1829            classify_feed_privacy("https://feeds.example.com/aB3xK9zQ7mP2rT5wL8nD4vF6")
1830                .is_private()
1831        );
1832        // A bare UUID path segment (many tokened feeds).
1833        assert!(classify_feed_privacy(
1834            "https://feeds.example.com/1f2e3d4c-5b6a-7089-90ab-cdef01234567"
1835        )
1836        .is_private());
1837    }
1838
1839    /// The dominant real-world private-podcast shape delivers the token as a
1840    /// FILENAME (`<token>.rss` / `<token>.xml`) or affixed inside a larger
1841    /// segment (`feed-<uuid>`). Named providers are caught by their host rule;
1842    /// these are UNKNOWN-provider CDNs that must still be caught by the generic
1843    /// backstop, so the secret is never fetched or stored.
1844    #[test]
1845    fn classify_privacy_catches_tokened_filenames_on_unknown_hosts() {
1846        // **A stem only the extension-strip branch can see.** Every other case
1847        // here is also caught by the sub-part scan (branch 3) or the UUID scan,
1848        // so deleting the strip left the suite green — `FEED_EXTENSIONS` was
1849        // effectively dead. This stem's `-`-separated parts are each too short
1850        // to look like a secret on their own, and the whole segment fails on
1851        // the `.` — only stripping `.rss` and re-testing the 26-char stem sees
1852        // it. That is exactly the shape a hyphen-bearing base64url token
1853        // filename takes.
1854        assert!(
1855            classify_feed_privacy("https://cdn.example/feeds/aB3xK9pQ-7mZ2vN8w-Qr5tYuW.rss")
1856                .is_private(),
1857            "a token stem visible only after stripping the extension was not caught"
1858        );
1859        // hex-32 token as an .xml filename.
1860        assert!(classify_feed_privacy(
1861            "https://cdn.somepod.io/f/a1b2c3d4e5f60718293a4b5c6d7e8f90.xml"
1862        )
1863        .is_private());
1864        // hex-32 token as a .rss filename on an unknown CDN.
1865        assert!(classify_feed_privacy(
1866            "https://dcs.megaphone.example/network/a1b2c3d4e5f60718293a4b5c6d7e8f90.rss"
1867        )
1868        .is_private());
1869        // UUID + .xml filename.
1870        assert!(classify_feed_privacy(
1871            "https://brandnew.example/feed/1f2e3d4c-5b6a-7089-90ab-cdef01234567.xml"
1872        )
1873        .is_private());
1874        // UUID affixed with a prefix (`feed-<uuid>`) — split can't see it, the
1875        // UUID-substring scan must.
1876        assert!(classify_feed_privacy(
1877            "https://x.example/feed-1f2e3d4c-5b6a-7089-90ab-cdef01234567"
1878        )
1879        .is_private());
1880        // UUID + .rss suffix.
1881        assert!(classify_feed_privacy(
1882            "https://x.example/1f2e3d4c-5b6a-7089-90ab-cdef01234567.rss"
1883        )
1884        .is_private());
1885        // hex-16 token as an .xml filename.
1886        assert!(classify_feed_privacy("https://x.example/feed/9f8e7d6c5b4a3928.xml").is_private());
1887        // A base64url token with `=` padding as a clean path segment.
1888        assert!(
1889            classify_feed_privacy("https://cdn.pod.io/f/YWJjZGVmZ2hpamtsbW5vcHFyc3R1dnc=")
1890                .is_private()
1891        );
1892    }
1893
1894    /// YouTube channel/playlist RSS feeds are FULLY PUBLIC (the id is a public
1895    /// handle, not a secret) and are the standard way to subscribe to a channel —
1896    /// they must NOT be false-blocked by the generic entropy heuristic.
1897    #[test]
1898    fn classify_privacy_allows_public_youtube_feeds() {
1899        assert_eq!(
1900            classify_feed_privacy(
1901                "https://www.youtube.com/feeds/videos.xml?channel_id=UC-lHJZR3Gqxm24_Vd_AJ5Yw"
1902            ),
1903            FeedPrivacy::Public
1904        );
1905        assert_eq!(
1906            classify_feed_privacy(
1907                "https://www.youtube.com/feeds/videos.xml?playlist_id=PLFgquLnL59alCl_2TQvOiD5Vgm1hCaGSI"
1908            ),
1909            FeedPrivacy::Public
1910        );
1911        // Bare host form too.
1912        assert_eq!(
1913            classify_feed_privacy(
1914                "https://youtube.com/feeds/videos.xml?channel_id=UC-lHJZR3Gqxm24_Vd_AJ5Yw"
1915            ),
1916            FeedPrivacy::Public
1917        );
1918        // The allowlist is narrow: a `token=` on the YouTube feeds path still
1919        // classifies private (can't smuggle a credential through the allowlist).
1920        assert!(classify_feed_privacy(
1921            "https://www.youtube.com/feeds/videos.xml?token=Zm9vYmFyc2VjcmV0dG9rZW4"
1922        )
1923        .is_private());
1924    }
1925
1926    #[test]
1927    fn classify_privacy_leaves_normal_public_feeds_public() {
1928        // Plain feed documents.
1929        assert_eq!(
1930            classify_feed_privacy("https://example.com/feed.xml"),
1931            FeedPrivacy::Public
1932        );
1933        assert_eq!(
1934            classify_feed_privacy("https://blog.example.com/rss"),
1935            FeedPrivacy::Public
1936        );
1937        assert_eq!(
1938            classify_feed_privacy("https://blog.example.com/rss.xml"),
1939            FeedPrivacy::Public
1940        );
1941        // A Substack PUBLIC feed (`/feed`, not `/feed/private/`) stays public.
1942        assert_eq!(
1943            classify_feed_privacy("https://author.substack.com/feed"),
1944            FeedPrivacy::Public
1945        );
1946        // A WordPress `/feed` endpoint.
1947        assert_eq!(
1948            classify_feed_privacy("https://wordpress.example.com/feed/"),
1949            FeedPrivacy::Public
1950        );
1951        // A plain Atom feed.
1952        assert_eq!(
1953            classify_feed_privacy("https://example.org/atom.xml"),
1954            FeedPrivacy::Public
1955        );
1956        // A long, hyphenated slug must NOT be mistaken for an embedded secret.
1957        assert_eq!(
1958            classify_feed_privacy("https://example.com/2026/07/my-first-long-blog-post-title/feed"),
1959            FeedPrivacy::Public
1960        );
1961        // A benign query key that merely contains "key" as a substring is fine.
1962        assert_eq!(
1963            classify_feed_privacy("https://example.com/feed?keyword=rust"),
1964            FeedPrivacy::Public
1965        );
1966        // A short, non-opaque value on a named key (e.g. an enum) is not a secret.
1967        assert_eq!(
1968            classify_feed_privacy("https://example.com/feed?p=2"),
1969            FeedPrivacy::Public
1970        );
1971        // An empty credential value is not a secret.
1972        assert_eq!(
1973            classify_feed_privacy("https://example.com/feed?token="),
1974            FeedPrivacy::Public
1975        );
1976        // A hyphenated slug ending in a feed extension must NOT be seen as a
1977        // tokened filename (the stem is short dictionary words, not a blob).
1978        assert_eq!(
1979            classify_feed_privacy("https://example.com/my-first-long-blog-post.xml"),
1980            FeedPrivacy::Public
1981        );
1982        // A short hex episode id in an .xml filename (< 16 chars) is not a secret.
1983        assert_eq!(
1984            classify_feed_privacy("https://example.com/episodes/ab12cd.xml"),
1985            FeedPrivacy::Public
1986        );
1987        // A dotted host-style filename slug stays public.
1988        assert_eq!(
1989            classify_feed_privacy("https://example.com/category/tech-news/feed.xml"),
1990            FeedPrivacy::Public
1991        );
1992        // Unparseable URL: treated as Public (add path rejects it downstream).
1993        assert_eq!(classify_feed_privacy("not a url"), FeedPrivacy::Public);
1994    }
1995
1996    /// **The detail is bounded where it is CONSTRUCTED, not only where it is
1997    /// stored.**
1998    ///
1999    /// Review found that widening `PollOutcome::Failed` with this field opened a
2000    /// second sink nobody looked at: `web.rs`'s `add_subscription` logs
2001    /// `?outcome` at INFO on a user-facing request path, so the whole
2002    /// untruncated anyhow chain — redirect-hop URLs, the SSRF guard's refusal
2003    /// text naming a resolved internal address — went to the access log.
2004    ///
2005    /// Bounding inside `bump_feed_errors` protected the database and nothing
2006    /// else. Bounding at construction protects every sink, including the ones
2007    /// added later.
2008    #[test]
2009    fn a_failure_detail_is_bounded_at_construction() {
2010        let huge = "x".repeat(10_000);
2011        let outcome = PollOutcome::Failed {
2012            backoff: BACKOFF_BASE,
2013            kind: FailureKind::Fetch,
2014            detail: failure_detail(&huge),
2015        };
2016        let PollOutcome::Failed { detail, .. } = &outcome else {
2017            panic!("wrong variant");
2018        };
2019        assert!(
2020            detail.chars().count() <= MAX_FAILURE_DETAIL_CHARS,
2021            "detail was {} chars",
2022            detail.chars().count(),
2023        );
2024        // And the Debug rendering — which is what actually reached the log — is
2025        // bounded with it.
2026        assert!(format!("{outcome:?}").len() < 1_000);
2027    }
2028
2029    /// **Every failure kind has its own label, and they round-trip.**
2030    ///
2031    /// Review found that collapsing all four `as_str` arms to `"fetch"` left
2032    /// the whole suite green: every test of these columns passed string
2033    /// literals, so nothing tied a variant to its label. A histogram whose
2034    /// buckets all say the same thing is worse than no histogram — it reports a
2035    /// single confident cause for four different failures.
2036    ///
2037    /// Asserted over `ALL` rather than a hand-written list, so adding a variant
2038    /// without a label fails here instead of silently sharing one.
2039    #[test]
2040    fn every_failure_kind_has_a_distinct_round_tripping_label() {
2041        let mut seen = std::collections::BTreeSet::new();
2042        for kind in FailureKind::ALL {
2043            let label = kind.as_str();
2044            assert!(
2045                seen.insert(label),
2046                "{label:?} is used by more than one FailureKind",
2047            );
2048            assert_eq!(
2049                FailureKind::parse(label),
2050                Some(kind),
2051                "{label:?} does not read back as the kind that wrote it",
2052            );
2053        }
2054        assert_eq!(seen.len(), FailureKind::ALL.len());
2055        // A label from a newer build is not attributed to a cause this one
2056        // knows — the `metrics::Backend::parse` contract.
2057        assert_eq!(FailureKind::parse("quota"), None);
2058    }
2059
2060    /// **Escalation reaches `settle_poll`.** `backoff_for` grows with the
2061    /// count and is tested alone; nothing asserted that the poll path passes
2062    /// the COUNT in. `backoff_for(1)` in its place left the whole suite green
2063    /// — a permanently dead feed retrying forever at the first-failure floor,
2064    /// which the comment on that line says must not happen.
2065    #[tokio::test]
2066    async fn backoff_escalates_with_consecutive_failures() -> anyhow::Result<()> {
2067        let pool = crate::store::init_url("sqlite::memory:").await?;
2068        let url = "https://dead.example/feed.xml";
2069        crate::store::upsert_feed(
2070            &pool,
2071            &crate::store::NewFeed {
2072                url: url.to_string(),
2073                ..Default::default()
2074            },
2075        )
2076        .await?;
2077        for _ in 0..5 {
2078            crate::store::bump_feed_errors(&pool, url, FailureKind::Fetch, "down").await?;
2079        }
2080        let before = chrono::Utc::now();
2081        settle_poll(
2082            &pool,
2083            url,
2084            &PollOutcome::Failed {
2085                backoff: Duration::from_secs(300),
2086                kind: FailureKind::Fetch,
2087                detail: "still down".to_string(),
2088            },
2089            Duration::from_secs(3600),
2090        )
2091        .await;
2092        let next: String = sqlx::query_scalar("SELECT next_poll FROM feeds WHERE url = ?1")
2093            .bind(url)
2094            .fetch_one(&pool)
2095            .await?;
2096        let next = chrono::DateTime::parse_from_rfc3339(&next)?.with_timezone(&chrono::Utc);
2097        let delay = (next - before).num_seconds();
2098        let expected = backoff_for(6).as_secs() as i64;
2099        assert!(
2100            (delay - expected).abs() <= 60,
2101            "sixth failure scheduled {delay}s out; escalation says {expected}s"
2102        );
2103        assert!(
2104            delay > backoff_for(1).as_secs() as i64 + 60,
2105            "the sixth failure landed on the first-failure floor"
2106        );
2107        Ok(())
2108    }
2109
2110    #[test]
2111    fn backoff_grows_and_is_capped() {
2112        assert_eq!(backoff_for(1), BACKOFF_BASE);
2113        assert!(backoff_for(2) > backoff_for(1));
2114        assert_eq!(backoff_for(100), BACKOFF_MAX);
2115    }
2116
2117    /// **Storable and pollable are ONE decision.**
2118    ///
2119    /// Review found the sequencing error this closes: making `at://` storable
2120    /// while nothing can poll it does not leave the feature dormant, it creates
2121    /// permanent failures that the cause histogram then publishes as
2122    /// unreachable publishers — the exact conflation it exists to end.
2123    #[test]
2124    fn an_at_uri_is_not_storable_while_standard_site_is_off() {
2125        let uri = "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab";
2126        assert!(
2127            !is_storable_feed_url(uri, false),
2128            "stored a feed nothing can poll"
2129        );
2130        assert!(is_storable_feed_url(uri, true));
2131        assert!(is_storable_feed_url("https://example.com/feed.xml", false));
2132        assert!(is_storable_feed_url("https://example.com/feed.xml", true));
2133    }
2134
2135    /// **A non-canonical scheme spelling is recognised and refused.** URL
2136    /// schemes are case-insensitive, so `At://` names the same thing as
2137    /// `at://` — but `feeds.url` is UNIQUE, so accepting both is two rows for
2138    /// one publication. Recognised (not passed through to the generic checks
2139    /// as if it were an ordinary URL), then refused for the spelling.
2140    #[test]
2141    fn a_non_canonical_at_uri_spelling_is_recognised_and_refused() {
2142        for odd in [
2143            "At://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
2144            "AT://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
2145        ] {
2146            assert!(!is_storable_feed_url(odd, true), "stored {odd:?}");
2147            // Fails CLOSED: it is an at-URI this reader will not store, not an
2148            // unparseable string that the `Err(_) => Public` arm waves through.
2149            assert!(
2150                classify_feed_privacy(odd).is_private(),
2151                "{odd:?} was declared publishable"
2152            );
2153        }
2154    }
2155
2156    /// **The handle form is not storable — the DID form is the identity.**
2157    ///
2158    /// `feeds.url` is UNIQUE; a handle and its DID would be two rows for one
2159    /// publication, and a handle can change hands. Every spelling is refused,
2160    /// canonical or not; resolving one to a DID is the input path's job.
2161    #[test]
2162    fn a_handle_form_publication_uri_is_not_storable() {
2163        for authority in [
2164            "alice.example.com",
2165            "EXAMPLE.COM",
2166            "169.254.169.254",
2167            "pds.internal",
2168            "printer.local",
2169            "host:8080",
2170            "-.-",
2171            "a b.c",
2172        ] {
2173            let uri = format!("at://{authority}/site.standard.publication/3lab");
2174            assert!(
2175                !is_storable_feed_url(&uri, true),
2176                "accepted authority {authority:?}"
2177            );
2178        }
2179    }
2180
2181    /// A control character or space in the at-URI is refused: `scheduler.rs`
2182    /// logs `%feed.url` with Display, and `feeds.url` is UNIQUE.
2183    #[test]
2184    fn an_at_uri_with_control_characters_is_not_storable() {
2185        for bad in [
2186            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab\n",
2187            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab ",
2188            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3l\tab",
2189        ] {
2190            assert!(!is_storable_feed_url(bad, true), "accepted {bad:?}");
2191        }
2192    }
2193
2194    /// **The `at://` exemption is a REGRESSION unless it is narrow.**
2195    ///
2196    /// Fan-out review found the arm I added was a bare prefix match, so *any*
2197    /// attacker-chosen string starting `at://` was declared safe to publish —
2198    /// skipping the userinfo check, the known-provider table, the private-path
2199    /// markers, the secret-query keys and the entropy heuristics. Measured
2200    /// against `main`, these three went from `Private` to `Public`.
2201    ///
2202    /// That matters because `rename_subscription` caches the URL AND rewrites
2203    /// the user's PUBLIC PDS record, with `classify_feed_privacy` as its only
2204    /// gate.
2205    #[test]
2206    fn a_credential_bearing_at_uri_is_still_private() {
2207        for hostile in [
2208            "at://user:pass@private.example.com/feed/private/TOKEN?apikey=deadbeefdeadbeef",
2209            "at://patreon.com/rss/12345?auth=deadbeefdeadbeefdeadbeef",
2210            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab?apikey=sekrit",
2211        ] {
2212            assert!(
2213                matches!(classify_feed_privacy(hostile), FeedPrivacy::Private(_)),
2214                "declared public: {hostile}"
2215            );
2216        }
2217    }
2218
2219    /// An rkey is `[A-Za-z0-9._:~-]` per atproto. Without that, a query string
2220    /// or path fragment smuggled into the rkey satisfies the three-segment
2221    /// check — which is what the exemption above keys off.
2222    #[test]
2223    fn an_rkey_outside_the_atproto_charset_is_not_storable() {
2224        for bad in [
2225            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab?apikey=sekrit",
2226            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab#frag",
2227            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab%2Fevil",
2228            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/caf\u{e9}",
2229            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab\u{202e}x",
2230        ] {
2231            assert!(!is_storable_feed_url(bad, true), "accepted rkey in {bad:?}");
2232        }
2233        // The legitimate charset still passes.
2234        for good in [
2235            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h",
2236            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/a.b_c~d-e",
2237        ] {
2238            assert!(is_storable_feed_url(good, true), "refused {good:?}");
2239        }
2240    }
2241
2242    /// **A real rkey is a TID, and a TID looks exactly like a secret.**
2243    ///
2244    /// Without an explicit `at://` arm, `classify_feed_privacy` runs the generic
2245    /// "high-entropy token in path" heuristic over the rkey. Measured: a
2246    /// realistic 16-char rkey on a handle-form at-URI is classified PRIVATE and
2247    /// the subscription REFUSED. The DID form escaped only because it fails to
2248    /// parse as a `Url` at all — so the bug was invisible from that side.
2249    ///
2250    /// The first version of this test used the rkey `3lab`, which is too short
2251    /// to trip the heuristic, so it passed with and without the fix.
2252    #[test]
2253    fn a_realistic_at_uri_rkey_is_not_mistaken_for_a_secret() {
2254        for uri in [
2255            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h",
2256            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/aB3xK9pQ7mZ2vN8w",
2257            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h",
2258        ] {
2259            assert_eq!(
2260                classify_feed_privacy(uri),
2261                FeedPrivacy::Public,
2262                "a publication rkey was mistaken for a credential: {uri}"
2263            );
2264        }
2265    }
2266
2267    /// **The DID form is the one that matters, and the one `Url::parse` cannot
2268    /// read.**
2269    ///
2270    /// `Url::parse("at://did:plc:…/…")` fails with *invalid port number* — the
2271    /// colons in the DID are taken as a port separator. So the obvious
2272    /// implementation, adding `"at"` to the `matches!` on `u.scheme()`, silently
2273    /// rejects every DID-based at-URI while appearing to work: the handle form
2274    /// (`at://alice.example.com/…`) parses fine and would pass such a test.
2275    ///
2276    /// All 19 at-URI rows in production are the DID form. A test written with a
2277    /// handle would have passed against an implementation that cannot store a
2278    /// single one of them.
2279    #[test]
2280    fn a_did_form_publication_uri_is_storable() {
2281        assert!(is_storable_feed_url(
2282            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
2283            true
2284        ));
2285    }
2286
2287    /// **An allowlist entry, not a loosening.** `at://` is accepted for exactly
2288    /// one foreign collection. Any other collection is somebody else's lexicon
2289    /// arriving through a path (`resolve_subscriptions`, OPML import) that takes
2290    /// records from outside with no add-path to reject them.
2291    #[test]
2292    fn an_at_uri_for_another_collection_is_not_storable() {
2293        assert!(!is_storable_feed_url(
2294            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/community.lexicon.rss.subscription/3lab",
2295            true
2296        ));
2297        assert!(!is_storable_feed_url(
2298            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/app.bsky.feed.post/3lab",
2299            true
2300        ));
2301    }
2302
2303    /// The malformed shapes, each of which a naive `split('/')` would accept.
2304    #[test]
2305    fn a_malformed_at_uri_is_not_storable() {
2306        for bad in [
2307            "at://",
2308            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc",
2309            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication",
2310            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/",
2311            "at:///site.standard.publication/3lab",
2312            "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab/extra",
2313            "at://not-a-did-or-handle/site.standard.publication/3lab",
2314            "at://did:plc:TOOSHORT/site.standard.publication/3lab",
2315        ] {
2316            assert!(!is_storable_feed_url(bad, true), "accepted {bad:?}");
2317        }
2318    }
2319
2320    /// **The reason the function exists, unchanged.** Mutating the new branch to
2321    /// accept any scheme makes this fail while the at-URI tests keep passing —
2322    /// that asymmetry is what says the change was an allowlist entry.
2323    #[test]
2324    fn the_refused_schemes_are_still_refused() {
2325        for bad in [
2326            "javascript:alert(1)",
2327            "file:///etc/passwd",
2328            "data:text/html,<script>",
2329            "ftp://example.com/feed.xml",
2330            "at:did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
2331        ] {
2332            assert!(!is_storable_feed_url(bad, true), "accepted {bad:?}");
2333        }
2334        // A hostless http(s) URL is a PARSE error, not a parsed URL with no
2335        // host — `https:///feed.xml` even parses as host `feed.xml`. What
2336        // refuses these is the `Err` arm, so that is what this pins.
2337        for hostless in ["http://", "https://?q=1", "http:///"] {
2338            assert!(
2339                !is_storable_feed_url(hostless, true),
2340                "a hostless URL {hostless:?} was storable"
2341            );
2342        }
2343        assert!(is_storable_feed_url("https://example.com/feed.xml", true));
2344        assert!(is_storable_feed_url("http://example.com/feed.xml", true));
2345    }
2346}