feather_reader/feed.rs
1//! Feed fetch → parse → sanitize → store pipeline.
2//!
3//! This is the module that turns a feed URL into rows in the [`store`]. It is
4//! deliberately conservative on three axes, because a feed reader ingests
5//! **hostile, arbitrary web input**:
6//!
7//! 1. **Politeness** — fetches use a **conditional GET** (`If-None-Match` /
8//! `If-Modified-Since` from the stored `ETag` / `Last-Modified`), an
9//! identifiable [`crate::USER_AGENT`], a request timeout, and a simple
10//! exponential backoff hint on error. A `304 Not Modified` is a no-op:
11//! the feed is untouched apart from bumping its next-poll time.
12//! 2. **Safety** — every entry's HTML is run through `ammonia` before it is
13//! ever stored (and therefore before it is ever rendered). Scripts, event
14//! handlers, `javascript:` URLs, tracking pixels' dangerous attributes, and
15//! other XSS vectors are stripped. Feeds carrying `<script>` is not
16//! hypothetical; treat all feed HTML as untrusted.
17//! 3. **Robustness** — a malformed feed is **logged and skipped**, never a
18//! panic. One bad publisher must not take down the poller. All non-test
19//! paths use `Result`/`anyhow`; there are no `unwrap`/`expect`s.
20//!
21//! The normalized shape written to the store is the store's own
22//! [`store::NewFeed`] / [`store::NewEntry`]; dedup is by feed-native GUID via
23//! [`store::insert_entries`]'s `ON CONFLICT (feed_id, guid)` upsert.
24
25use std::time::Duration;
26
27use anyhow::{Context, Result};
28use chrono::{DateTime, SecondsFormat, Utc};
29use feed_rs::model::{Entry as RawEntry, Feed as RawFeed, Text};
30use reqwest::header::{ETAG, IF_MODIFIED_SINCE, IF_NONE_MATCH, LAST_MODIFIED};
31use reqwest::{Client, StatusCode};
32use sqlx::SqlitePool;
33use url::Url;
34
35use crate::store::{self, Feed, NewEntry, NewFeed};
36
37/// The privacy classification of a feed URL — the output of
38/// [`classify_feed_privacy`].
39///
40/// A **private** feed carries a secret (a token / key / auth credential) *in the
41/// URL itself* — a Substack `…/feed/private/<token>`, a Patreon `?auth=…` feed,
42/// a Ghost members `?uuid=` feed, a private-podcast token feed (Supercast,
43/// Supporting Cast, tokened Megaphone/Acast+), and so on. FeatherReader stores a
44/// user's subscriptions as records in their **public PDS** (unauthenticated
45/// `getRecord` / `listRecords` + the firehose, retained even after delete), so
46/// writing such a URL anywhere — the PDS *or* the server's own store — would risk
47/// leaking paid / members-only access.
48///
49/// **Decision (stopgap until atproto permissioned data ships): FeatherReader
50/// supports PUBLIC feeds only.** A feed classified [`FeedPrivacy::Private`] is
51/// *refused* at the add / import boundary — never fetched, never stored, never
52/// written to the PDS. There is no local-secret fallback and no override: the
53/// server holds NO private secret, ever, which keeps "your data lives in your
54/// public PDS" 100% honest.
55#[derive(Debug, Clone, PartialEq, Eq)]
56pub enum FeedPrivacy {
57 /// No secret detected in the URL; safe to add as a public feed.
58 Public,
59 /// A secret was detected in the URL. The `String` is a short, human-readable
60 /// reason (for logging / the skip report), e.g. `"substack private feed
61 /// path"`. The feed is refused — not fetched, stored, or written anywhere.
62 Private(String),
63}
64
65impl FeedPrivacy {
66 /// Whether this classification is [`FeedPrivacy::Private`].
67 pub fn is_private(&self) -> bool {
68 matches!(self, FeedPrivacy::Private(_))
69 }
70}
71
72/// Query-parameter *keys* that, when present with a long/opaque value, mark a URL
73/// as carrying a secret. Conservative and lowercase-compared; matched as a whole
74/// key (case-insensitive) so a benign `keyword=` does NOT trip `key`. This is the
75/// generic, provider-agnostic credential-in-query defence — it catches paid
76/// feeds from providers we've never heard of. Covers Patreon (`auth`), Ghost
77/// members (`uuid`), token-in-query feeds (`token`/`key`/`k`/`sig`/`hash`), and
78/// the long tail (`access`/`apikey`/`private`/`password`/`u`/`s`/`p`/…).
79const SECRET_QUERY_KEYS: &[&str] = &[
80 "token", "key", "auth", "secret", "k", "sig", "hash", "access", "apikey", "api_key", "uuid",
81 "id", "u", "s", "p", "private", "password", "pw",
82];
83
84/// Path *segments* / fragments that mark a private-feed URL shape. Matched as a
85/// case-insensitive substring of the (lowercased) path so `/feed/private/<tok>`,
86/// `/members/…`, `/subscriber/…` etc. all trip regardless of the token that
87/// follows. Provider-agnostic: many paid providers expose members-only feeds
88/// under one of these path conventions.
89const PRIVATE_PATH_MARKERS: &[&str] = &[
90 "/private/",
91 "/feed/private/",
92 "/rss/private/",
93 "/private-feed/",
94 "/members/",
95 "/member/",
96 "/subscriber/",
97];
98
99/// A KNOWN paid/private feed provider, matched by host substring + (optionally) a
100/// path/query marker specific to that provider. This is the **secondary**,
101/// precision layer on top of the generic heuristic — it names providers so the
102/// skip report can say *why* and so we catch provider-specific shapes that the
103/// generic pass might rate as borderline. Data-driven and easy to extend: add a
104/// row, don't touch the matcher.
105struct KnownProvider {
106 /// Substring that must appear in the URL host (lowercased), e.g.
107 /// `substack.com`.
108 host_contains: &'static str,
109 /// Optional lowercased substring that must appear in the path-or-query for a
110 /// match (a provider's private-feed marker). `None` = the host alone is
111 /// enough (used for hosts that ONLY serve private/tokened feeds).
112 marker: Option<&'static str>,
113 /// Human-readable reason for the skip report.
114 reason: &'static str,
115}
116
117/// The known-provider table. Covers paid NEWSLETTERS and private PODCASTS — an
118/// RSS reader ingests both. Kept intentionally verbose/commented so it's obvious
119/// what each row targets and safe to extend.
120const KNOWN_PROVIDERS: &[KnownProvider] = &[
121 // --- Paid newsletters -------------------------------------------------
122 // Substack private feed: author.substack.com/feed/private/<token>.
123 KnownProvider {
124 host_contains: "substack.com",
125 marker: Some("/feed/private/"),
126 reason: "Substack private feed",
127 },
128 // Patreon RSS carries the member token as ?auth=.
129 KnownProvider {
130 host_contains: "patreon.com",
131 marker: Some("auth="),
132 reason: "Patreon member feed",
133 },
134 // Ghost members feed: ?uuid=<member-uuid> (or a members token path).
135 KnownProvider {
136 host_contains: "ghost.io",
137 marker: Some("uuid="),
138 reason: "Ghost members feed",
139 },
140 // Buttondown paid RSS uses a per-subscriber token in the path/query.
141 KnownProvider {
142 host_contains: "buttondown.email",
143 marker: Some("token"),
144 reason: "Buttondown premium feed",
145 },
146 KnownProvider {
147 host_contains: "buttondown.com",
148 marker: Some("token"),
149 reason: "Buttondown premium feed",
150 },
151 // Beehiiv premium RSS carries a subscriber token.
152 KnownProvider {
153 host_contains: "beehiiv.com",
154 marker: Some("token"),
155 reason: "Beehiiv premium feed",
156 },
157 // Memberful-gated feeds (host or ?auth token).
158 KnownProvider {
159 host_contains: "memberful.com",
160 marker: None,
161 reason: "Memberful members feed",
162 },
163 // Pico / Steady member feeds.
164 KnownProvider {
165 host_contains: "pico.link",
166 marker: None,
167 reason: "Pico member feed",
168 },
169 KnownProvider {
170 host_contains: "steadyhq.com",
171 marker: None,
172 reason: "Steady member feed",
173 },
174 // --- Private podcasts -------------------------------------------------
175 // Supercast private podcast feeds (host serves tokened member feeds only).
176 KnownProvider {
177 host_contains: "supercast.com",
178 marker: None,
179 reason: "Supercast private podcast",
180 },
181 KnownProvider {
182 host_contains: "supercast.tech",
183 marker: None,
184 reason: "Supercast private podcast",
185 },
186 // Supporting Cast private podcast feeds (supportingcast.fm).
187 KnownProvider {
188 host_contains: "supportingcast.fm",
189 marker: None,
190 reason: "Supporting Cast private podcast",
191 },
192 // RedCircle private/exclusive feeds.
193 KnownProvider {
194 host_contains: "redcircle.com",
195 marker: Some("private"),
196 reason: "RedCircle private podcast",
197 },
198 // Private/tokened Megaphone, Acast+, and Omny feeds carry an access token.
199 KnownProvider {
200 host_contains: "megaphone.fm",
201 marker: Some("token"),
202 reason: "Megaphone private podcast",
203 },
204 KnownProvider {
205 host_contains: "acast.com",
206 marker: Some("token"),
207 reason: "Acast+ private podcast",
208 },
209 KnownProvider {
210 host_contains: "omny.fm",
211 marker: Some("token"),
212 reason: "Omny private podcast",
213 },
214 // Apple / Spotify subscriber podcast feeds carry a per-listener token.
215 KnownProvider {
216 host_contains: "podcasts.apple.com",
217 marker: Some("token"),
218 reason: "Apple subscriber podcast",
219 },
220 KnownProvider {
221 host_contains: "spotify.com",
222 marker: Some("token"),
223 reason: "Spotify subscriber podcast",
224 },
225];
226
227/// Whether a URL may be **stored or published** as a feed URL at all.
228///
229/// This is the storage-side twin of the scheme check `net::check_scheme` applies
230/// before fetching. The fetch side has always been safe, because nothing can
231/// reach the network except through `net.rs` — but "safe to fetch" and "safe to
232/// write down" are different questions, and only the first had an answer.
233///
234/// Two paths took a URL from outside and stored it with no validation at all:
235/// `resolve_subscriptions` (any atproto client can write a subscription record
236/// into a user's repo) and the OPML import (`xmlUrl` is whatever the file says).
237/// `classify_feed_privacy` does not cover this — it deliberately returns
238/// `Public` for an unparseable URL, on the stated assumption that "the add path
239/// will reject it as malformed regardless", and those two paths are the ones
240/// that never had an add path to do the rejecting.
241///
242/// Note that `javascript:alert(1)` and `file:///etc/passwd` both *parse* cleanly
243/// as URLs, so parsing is not the check — the scheme is.
244pub fn is_storable_feed_url(url: &str, allow_at_uri: bool) -> bool {
245 // **`at://` is checked BEFORE `Url::parse`, because `Url::parse` cannot read
246 // the form that matters.** `at://did:plc:…/…` fails to parse with *invalid
247 // port number* — the colons in the DID are taken as a port separator — while
248 // the handle form `at://alice.example.com/…` parses fine. So adding `"at"`
249 // to the `matches!` below would appear to work and silently reject every
250 // DID-based at-URI, which is all of them in practice.
251 if let Some(rest) = crate::atproto::strip_at_prefix(url) {
252 // **Recognised case-insensitively, stored canonically.** Schemes are
253 // case-insensitive, so `At://` names the same publication — but
254 // `feeds.url` is UNIQUE, so accepting both spellings is two rows for
255 // one publication, the hazard the canonical-handle rule exists for.
256 // Recognising it here rather than letting it fall through to the
257 // generic checks is what keeps it out of the poller: nothing can fetch
258 // it under any spelling.
259 if !url.starts_with(crate::atproto::AT_URI_PREFIX) {
260 return false;
261 }
262 // **This gates STORING only — polling is handled by exclusion**, by
263 // kind rather than by any re-description of the URL, and
264 // `FeedKind::POLLABLE` is the one place the why is written down.
265 return allow_at_uri && is_storable_publication_uri(rest);
266 }
267 match Url::parse(url) {
268 Ok(u) => {
269 // No host check: for http(s) the `url` crate refuses every hostless
270 // spelling at parse (`http://`, `https://?q`, `http:///` are all
271 // "empty host") and turns `https:///x` into host `x`. A conjunct
272 // requiring a non-empty host was unreachable — a test hunt listed
273 // it as untested, and the honest answer was that no input reaches
274 // it. The `Err` arm below is what refuses a hostless URL.
275 matches!(u.scheme(), "http" | "https")
276 }
277 Err(_) => false,
278 }
279}
280
281/// The body of an `at://` URI — `<did-or-handle>/<collection>/<rkey>` — judged
282/// as a **storable feed**.
283///
284/// An allowlist entry, not a loosening: exactly one foreign collection is
285/// accepted, `site.standard.publication`. The two paths this guard exists for
286/// (`resolve_subscriptions`, the OPML import) take records written by any
287/// atproto client, so "it is an at-URI" is not a reason to store it — only "it
288/// is a publication this reader knows how to poll" is.
289fn is_storable_publication_uri(rest: &str) -> bool {
290 let mut parts = rest.split('/');
291 let (Some(authority), Some(collection), Some(rkey)) =
292 (parts.next(), parts.next(), parts.next())
293 else {
294 return false;
295 };
296 parts.next().is_none()
297 && collection == crate::lexicon::nsid::STANDARD_PUBLICATION
298 // **The rkey is validated against atproto's rules, not a blacklist.**
299 //
300 // A blacklist was the first attempt and it leaked twice: `is_control()`
301 // is Unicode category Cc only, so a bidi override (Cf) passed — and it
302 // reordered both the manage page and `scheduler.rs`'s `%feed.url` log
303 // line. Worse, nothing stopped a query string or fragment living inside
304 // the rkey, which satisfies the three-segment check and is exactly what
305 // `classify_feed_privacy`'s `at://` exemption keys off: a token
306 // smuggled there would have been declared public.
307 //
308 // An allowlist cannot leak the next character class someone finds.
309 // The charset alone still admitted `.`, `..` and a 10 000-byte key;
310 // `is_valid_rkey` carries the length and reserved-name rules too.
311 && crate::atproto::is_valid_rkey(rkey)
312 && is_storable_at_authority(authority)
313}
314
315/// The DID form only. `did:plc:` identifiers are validated by
316/// [`crate::oauth::identity::is_atproto_did`] rather than a `did:` prefix check,
317/// which would accept `did:plc:TOOSHORT`.
318///
319/// **The handle form is not storable, for the reason the canonical-handle rule
320/// already gave:** `feeds.url` is UNIQUE, so `at://alice.example.com/…` beside
321/// `at://did:plc:…/…` is two rows — two sidebar entries, and two polled copies
322/// once the reader is wired — for one publication. A handle is a mutable name
323/// for a DID; the row is keyed on the identity. Resolving a pasted or imported
324/// handle to its DID is the reader's job (#165 already resolves DIDs to their
325/// PDS), and belongs at input, not in storage.
326fn is_storable_at_authority(authority: &str) -> bool {
327 crate::oauth::identity::is_atproto_did(authority)
328}
329
330/// Classify whether a feed URL carries a secret credential in the URL itself.
331///
332/// Returns [`FeedPrivacy::Private`] (with a reason) when the URL looks like it
333/// embeds a token / key / auth credential, else [`FeedPrivacy::Public`].
334///
335/// **Design — provider-agnostic first.** The primary defence is a generic
336/// credential-in-URL heuristic that catches paid feeds from *any* provider, not
337/// just the ones we've named; a secondary known-provider table adds precision
338/// (and a nicer reason) for the common paid newsletters and private podcasts. We
339/// deliberately **bias toward flagging**: a false-positive block of a public feed
340/// is low-harm (the user just can't add that one feed yet), whereas a false
341/// negative would leak a paid secret onto the public network — high-harm.
342///
343/// Detection (any one is sufficient):
344/// 1. **Userinfo** — `https://user:pass@host/…` embeds credentials directly.
345/// 2. **Known private-feed path markers** — `PRIVATE_PATH_MARKERS`
346/// (`/feed/private/`, `/members/`, `/subscriber/`, …).
347/// 3. **Credential query parameters** — a query key in `SECRET_QUERY_KEYS` with
348/// a long/opaque value (Patreon `?auth=`, Ghost `?uuid=`, `?token=`, …).
349/// 4. **High-entropy opaque token segments** — a long opaque blob (hex ≥ 16,
350/// base64url ≥ 16, or a UUID) anywhere in the path or a query value, even
351/// without a telltale name.
352/// 5. **Known providers** — `KNOWN_PROVIDERS` host (+ optional marker) match.
353///
354/// An unparseable URL is treated as [`FeedPrivacy::Public`]: the add path rejects
355/// a malformed URL downstream anyway, and we don't want a parse quirk to
356/// misclassify.
357pub fn classify_feed_privacy(url: &str) -> FeedPrivacy {
358 // **`at://` is classified deliberately, and NOT doing so refused real
359 // subscriptions.** An atproto rkey is a TID — 13 base32-sortable characters
360 // — which is exactly what the generic "high-entropy token in path"
361 // heuristic below is looking for. Measured: without this arm,
362 // `at://did:plc:…/site.standard.publication/3lab2c4d5e6f7g8h` is
363 // classified PRIVATE and the subscription refused, while the DID form slips
364 // through only because it fails to parse as a `Url` at all.
365 //
366 // `Public` is the right answer: a publication is a public record in a
367 // public repo and the rkey is a handle, not a secret, so there is no
368 // private/paid shape for this scheme to carry.
369 if let Some(rest) = crate::atproto::strip_at_prefix(url) {
370 // A non-canonical scheme spelling is an at-URI this reader will not
371 // store, not an unparseable string for the `Err(_) => Public` arm below
372 // to wave through. Fail closed.
373 if !url.starts_with(crate::atproto::AT_URI_PREFIX) {
374 return FeedPrivacy::Private("non-canonical at:// scheme spelling".to_string());
375 }
376 // **Only a WELL-FORMED publication URI is exempt.** The first version
377 // of this was a bare prefix match, which declared any attacker-chosen
378 // string starting `at://` safe to publish — skipping the userinfo
379 // check, the known-provider table, the private-path markers, the
380 // secret-query keys and the entropy heuristics all at once. That is a
381 // regression against every one of them, on a path
382 // (`rename_subscription`) where this function is the only gate and the
383 // value is written to the user's PUBLIC repo.
384 //
385 // Anything else falls through to the generic checks below, which is
386 // where a credential-bearing string belongs.
387 if is_storable_publication_uri(rest) {
388 return FeedPrivacy::Public;
389 }
390 // **Malformed `at://` is REFUSED, not passed through.** Falling through
391 // reaches `Url::parse`, which fails on the DID form and lands on the
392 // `Err(_) => Public` arm below — whose justification is "the add path
393 // will reject it as a malformed URL regardless".
394 //
395 // **Defence in depth, not a live gate.** This comment used to say the
396 // justification is false on the `rename_subscription` path, "where this
397 // function is the only gate". That stopped being true when storability
398 // moved ahead of privacy on the repoint: a review then found no
399 // production caller can reach this arm at all — add pre-checks the
400 // `at://` prefix, rename and OPML and `resolve_subscriptions` all
401 // refuse a non-storable URL first. It stays because a fail-closed
402 // branch is worth its keep for the next caller that arrives without
403 // one, and because deleting a guard on the grounds that nothing
404 // currently reaches it is how the next one gets it wrong. It is not
405 // load-bearing today, and saying so is the honest version.
406 return FeedPrivacy::Private("not a well-formed at:// publication URI".to_string());
407 }
408 let parsed = match Url::parse(url) {
409 Ok(u) => u,
410 // Can't parse => the add path will reject it as a malformed URL regardless.
411 Err(_) => return FeedPrivacy::Public,
412 };
413
414 // (1) Userinfo (`https://user:pass@host/…`) — credentials in the authority.
415 if !parsed.username().is_empty() || parsed.password().is_some() {
416 return FeedPrivacy::Private("credentials in URL userinfo".to_string());
417 }
418
419 let path_lower = parsed.path().to_ascii_lowercase();
420 let query_lower = parsed.query().unwrap_or("").to_ascii_lowercase();
421 let host_lower = parsed.host_str().unwrap_or("").to_ascii_lowercase();
422
423 // (0) Public-feed allowlist. A handful of large, fully-public feed shapes
424 // carry a high-entropy-looking id in the query that would otherwise trip the
425 // generic entropy heuristic. YouTube channel/playlist RSS
426 // (`youtube.com/feeds/videos.xml?channel_id=UC…` / `?playlist_id=PL…`) is the
427 // canonical way any reader subscribes to a channel — the id is a PUBLIC
428 // handle, not a secret. Allowlist it before the generic checks so we don't
429 // false-block it. (Userinfo / known-provider markers are checked below and
430 // still apply, so this can't be used to smuggle a credential.)
431 if is_public_youtube_feed(&host_lower, &path_lower, &parsed) {
432 return FeedPrivacy::Public;
433 }
434
435 // (5) Known-provider precision layer (checked early so its specific reason
436 // wins over a generic one). Host substring + optional path/query marker.
437 for kp in KNOWN_PROVIDERS {
438 if host_lower.contains(kp.host_contains) {
439 let marker_ok = match kp.marker {
440 None => true,
441 Some(m) => {
442 let m = m.to_ascii_lowercase();
443 path_lower.contains(&m) || query_lower.contains(&m)
444 }
445 };
446 if marker_ok {
447 return FeedPrivacy::Private(kp.reason.to_string());
448 }
449 }
450 }
451
452 // (2) Known private-feed path markers.
453 for marker in PRIVATE_PATH_MARKERS {
454 if path_lower.contains(marker) {
455 return FeedPrivacy::Private(format!("private feed path `{marker}`"));
456 }
457 }
458
459 // (3) Credential query parameters with a long/opaque value.
460 for (k, v) in parsed.query_pairs() {
461 let key = k.as_ref().to_ascii_lowercase();
462 if SECRET_QUERY_KEYS.iter().any(|sk| *sk == key) && value_is_opaque(v.as_ref()) {
463 return FeedPrivacy::Private(format!("credential query parameter `{key}`"));
464 }
465 }
466
467 // (4) High-entropy opaque token segments (an embedded key/token with no
468 // telltale name): hex ≥ 16, base64url ≥ 16, or a UUID, in path or query.
469 // The dominant real-world private-podcast shape delivers the token as a
470 // *filename* (`<token>.rss` / `<token>.xml`) or affixed inside a larger
471 // segment (`feed-<uuid>`), so [`segment_hides_secret`] strips a trailing feed
472 // extension AND scans dot/underscore/hyphen-delimited sub-parts, not just the
473 // whole segment.
474 for seg in parsed.path().split('/').filter(|s| !s.is_empty()) {
475 if segment_hides_secret(seg) {
476 return FeedPrivacy::Private("high-entropy token in path".to_string());
477 }
478 }
479 for (_, v) in parsed.query_pairs() {
480 if looks_like_embedded_secret(v.as_ref()) {
481 return FeedPrivacy::Private("high-entropy token in query".to_string());
482 }
483 }
484
485 FeedPrivacy::Public
486}
487
488/// Whether a *named* credential query value (`?token=<v>`) is long/opaque enough
489/// to count as a secret. A short value (e.g. an enum like `?token=none`) is not.
490/// We treat a UUID, or anything ≥ 8 chars that isn't an obvious plain word, as
491/// opaque — named credential keys already signal intent, so the length bar is
492/// low.
493fn value_is_opaque(v: &str) -> bool {
494 if v.is_empty() {
495 return false;
496 }
497 if is_uuid(v) {
498 return true;
499 }
500 v.len() >= 8
501}
502
503/// Heuristic: does `s` look like an embedded secret (an opaque high-entropy
504/// token), as opposed to an ordinary slug or word? Matches a UUID, a hex string
505/// ≥ 16 chars, or a base64url-ish blob ≥ 16 chars that mixes letters and digits
506/// and isn't a hyphen/dot slug. Deliberately strict so it only fires on things
507/// that really look like keys — the named-marker and known-provider checks cover
508/// the rest.
509fn looks_like_embedded_secret(s: &str) -> bool {
510 if is_uuid(s) {
511 return true;
512 }
513 // Hex string ≥ 16 chars (e.g. a 32-char MD5-ish token).
514 if s.len() >= 16 && s.chars().all(|c| c.is_ascii_hexdigit()) {
515 return true;
516 }
517 // base64url-ish opaque blob ≥ 16 chars.
518 if s.len() < 16 {
519 return false;
520 }
521 // Hyphen/dot-heavy slugs (`this-is-a-normal-post-title`) are not secrets.
522 let separators = s
523 .bytes()
524 .filter(|b| *b == b'-' || *b == b'.' || *b == b' ')
525 .count();
526 if separators >= 3 {
527 return false;
528 }
529 // Must be plausibly token-charset: base64url alphabet only. `=` is accepted
530 // as base64 padding (it only ever appears trailing on a real blob, so a
531 // padded base64url token like `…dnc=` still counts).
532 let token_chars = s
533 .chars()
534 .filter(|c| c.is_ascii_alphanumeric() || matches!(c, '_' | '-' | '='))
535 .count();
536 if token_chars < s.chars().count() {
537 return false;
538 }
539 let has_alpha = s.chars().any(|c| c.is_ascii_alphabetic());
540 let has_digit = s.chars().any(|c| c.is_ascii_digit());
541 if !(has_alpha && has_digit) {
542 // A token almost always mixes letters and digits; a pure-alpha long
543 // segment is far more likely to be a normal (if long) slug/word.
544 return false;
545 }
546 // Distinct-character ratio: real tokens use most of the alphabet, words
547 // repeat a small set. Require >= 10 distinct chars for a 16+ char blob.
548 let mut seen = std::collections::HashSet::new();
549 for c in s.chars() {
550 seen.insert(c.to_ascii_lowercase());
551 }
552 seen.len() >= 10
553}
554
555/// Known feed/file extensions a token filename may wear (`<token>.rss`,
556/// `<token>.xml`, …). Stripped before the whole-segment secret test so a
557/// tokened *filename* — the dominant private-podcast URL shape — is still caught.
558const FEED_EXTENSIONS: &[&str] = &["rss", "xml", "atom", "json", "rss20"];
559
560/// Whether a single path segment hides an embedded secret. Beyond the plain
561/// whole-segment [`looks_like_embedded_secret`] test, this also catches the two
562/// real-world shapes that wrap a token so the whole segment is no longer a clean
563/// blob:
564///
565/// 1. **Token-as-filename** — `<token>.rss` / `<token>.xml`: strip a trailing
566/// feed extension and re-test the stem.
567/// 2. **Token affixed inside a larger segment** — `feed-<uuid>`, `<token>.xml`,
568/// `pod_<hex32>`: split on `.`/`_`/`-` and test each sub-part, so a
569/// high-entropy blob delimited by an affix is still found.
570fn segment_hides_secret(seg: &str) -> bool {
571 if looks_like_embedded_secret(seg) {
572 return true;
573 }
574 // (1) Strip a trailing known feed extension and re-test the stem.
575 if let Some((stem, ext)) = seg.rsplit_once('.') {
576 if FEED_EXTENSIONS.contains(&ext.to_ascii_lowercase().as_str())
577 && looks_like_embedded_secret(stem)
578 {
579 return true;
580 }
581 }
582 // (2) A UUID embedded with an affix (`feed-<uuid>`, `<uuid>-audio`) — the
583 // `-` delimiters inside the UUID mean a naive split can't see it, so scan for
584 // a canonical UUID substring directly.
585 if contains_uuid(seg) {
586 return true;
587 }
588 // (3) Scan `.`/`_`/`-`-delimited sub-parts for a high-entropy blob affixed to
589 // an ordinary word (`pod_<hex32>`, `<hex32>.mp3`). Only fires on multi-part
590 // segments (a single-part segment was already covered by the whole-segment
591 // test above), so a plain `my-normal-post-slug` — whose parts are short
592 // dictionary words — can't trip it.
593 if seg.contains(['.', '_', '-']) {
594 for part in seg.split(['.', '_', '-']).filter(|p| !p.is_empty()) {
595 if looks_like_embedded_secret(part) {
596 return true;
597 }
598 }
599 }
600 false
601}
602
603/// Whether `s` contains a canonical 8-4-4-4-12 UUID as a substring (allowing an
604/// affix on either side, e.g. `feed-<uuid>` or `<uuid>-audio`). Slides a 36-char
605/// window over the string and tests each with [`is_uuid`].
606fn contains_uuid(s: &str) -> bool {
607 const UUID_LEN: usize = 36; // 8+4+4+4+12 + 4 hyphens.
608 let bytes = s.as_bytes();
609 if bytes.len() < UUID_LEN {
610 return false;
611 }
612 // ASCII-only window: a UUID is pure ASCII hex/hyphen, so byte indexing is
613 // safe here (a multi-byte char in the window just fails is_uuid).
614 (0..=bytes.len() - UUID_LEN).any(|i| s.get(i..i + UUID_LEN).map(is_uuid).unwrap_or(false))
615}
616
617/// Whether this is a PUBLIC YouTube channel/playlist RSS feed
618/// (`www.youtube.com/feeds/videos.xml?channel_id=UC…` or `?playlist_id=PL…`).
619/// The channel/playlist id is a public handle, not a credential, so these feeds
620/// must NOT be flagged by the generic entropy heuristic. We require the exact
621/// public host + feeds path + one of the two public id keys, so this narrow
622/// allowlist can't be abused to smuggle a `?token=` past classification.
623fn is_public_youtube_feed(host_lower: &str, path_lower: &str, parsed: &Url) -> bool {
624 let host_ok = host_lower == "youtube.com"
625 || host_lower == "www.youtube.com"
626 || host_lower.ends_with(".youtube.com");
627 if !host_ok || !path_lower.starts_with("/feeds/videos.xml") {
628 return false;
629 }
630 // Only the public id keys may appear; a `token`/`auth`/… key means treat it
631 // as a normal (potentially private) URL and let the checks below run.
632 parsed.query_pairs().all(|(k, _)| {
633 let k = k.as_ref().to_ascii_lowercase();
634 k == "channel_id" || k == "playlist_id" || k == "user"
635 })
636}
637
638/// Whether `s` is a canonical 8-4-4-4-12 hyphenated UUID (any hex case).
639fn is_uuid(s: &str) -> bool {
640 let groups = [8usize, 4, 4, 4, 12];
641 let parts: Vec<&str> = s.split('-').collect();
642 if parts.len() != groups.len() {
643 return false;
644 }
645 parts
646 .iter()
647 .zip(groups.iter())
648 .all(|(p, &n)| p.len() == n && p.chars().all(|c| c.is_ascii_hexdigit()))
649}
650
651/// How long a single feed fetch may take before we give up.
652const FETCH_TIMEOUT: Duration = Duration::from_secs(30);
653
654/// Per-read idle timeout: cap the wait for the *next* body chunk, so a server
655/// that trickles bytes forever can't tie up a fetch under the total timeout.
656const READ_TIMEOUT: Duration = Duration::from_secs(15);
657
658/// Base backoff applied after a failed poll; the caller multiplies this by the
659/// feed's consecutive-error count (with a ceiling) to space out retries.
660const BACKOFF_BASE: Duration = Duration::from_secs(300);
661
662/// Ceiling on backoff so a persistently broken feed still gets retried daily.
663const BACKOFF_MAX: Duration = Duration::from_secs(24 * 3600);
664
665/// The outcome of polling a single feed. Lets the scheduler decide how to
666/// reschedule (and lets tests assert what happened) without inspecting the DB.
667#[derive(Debug, Clone, PartialEq, Eq)]
668pub enum PollOutcome {
669 /// The feed was fetched, parsed, and stored. `new_entries` is the number of
670 /// entries inserted or updated by this poll.
671 Updated { new_entries: u64 },
672 /// The server returned `304 Not Modified` — nothing changed, nothing stored.
673 NotModified,
674 /// The fetch or parse failed; the feed was left intact and skipped. Carries
675 /// the suggested backoff before the next attempt. Never a panic.
676 ///
677 /// **`kind` and `detail` are the reason, and they exist because their
678 /// absence cost a production investigation.** Until #159 the error was
679 /// logged here and discarded, so `feeds` recorded that a feed was failing
680 /// and never why — which is how sixty feeds broken by our own 304 handling
681 /// looked exactly like sixty dead blogs. `kind` is a small closed
682 /// vocabulary so failures can be counted by cause; `detail` is the message
683 /// for a human reading one row.
684 Failed {
685 backoff: Duration,
686 kind: FailureKind,
687 detail: String,
688 },
689}
690
691/// Why a poll failed, as a closed set.
692///
693/// Closed on purpose: the point is to *count* failures by cause, and a free-text
694/// kind cannot be counted. The detail string carries whatever else matters.
695#[derive(Debug, Clone, Copy, PartialEq, Eq)]
696pub enum FailureKind {
697 /// The request never produced a response — DNS, TLS, timeout, connection
698 /// refused, or a refusal by the SSRF guard.
699 Fetch,
700 /// A response arrived with a non-success status.
701 Status,
702 /// The body was too large, or reading it failed part-way.
703 Body,
704 /// The body arrived and is not a feed this parser can read.
705 Parse,
706}
707
708/// What the poller does with a feed row.
709///
710/// **A column, not a predicate.** "Can this be fetched?" used to be
711/// `substr(url, 1, 5) = 'at://'` spliced into four statements, with a fifth
712/// reader that had already drifted from them. Deciding it once, in Rust, at
713/// insert — and storing the answer — means SQL cannot disagree with the
714/// fetcher, and wiring the standard.site reader becomes a change to this
715/// function plus a dispatch, rather than an edit to every statement that
716/// mentions a URL.
717#[derive(Debug, Clone, Copy, PartialEq, Eq)]
718pub enum FeedKind {
719 /// An RSS/Atom/JSON feed document fetched over http(s).
720 Rss,
721 /// An `at://…/site.standard.publication/…` record pair in somebody's PDS.
722 /// Storable behind `FEATHERREADER_STANDARD_SITE`; **not yet pollable**, so
723 /// [`FeedKind::POLLABLE`] excludes it. Wiring the reader is what moves it.
724 Publication,
725}
726
727impl FeedKind {
728 /// The kinds the scheduler may select. The single place that changes when
729 /// the standard.site reader is wired to the poller.
730 ///
731 /// **This is the canonical home of the at:// exclusion; the other sites
732 /// point here.** It used to be a SQL string predicate in `store`, carrying
733 /// its own copy of the rule — which is how one reader (`count_feeds`) came
734 /// to drift from it unnoticed.
735 ///
736 /// Why an unpollable kind is skipped rather than failed: `poll_feed`
737 /// reaches `net::guarded_get`, whose `check_scheme` refuses any non-http(s)
738 /// scheme, and the standard.site reader is not yet wired to the scheduler.
739 /// Handing such a row to the poller does not leave the feature dormant — it
740 /// manufactures one permanent failure per row, which the public cause
741 /// histogram then reports as an unreachable publisher. Unsupported is not
742 /// broken, and telling those apart is the entire reason a failure cause is
743 /// recorded. (Rows like this exist: subscriptions written by other clients
744 /// before this reader refused the scheme.)
745 ///
746 /// **`store::count_feeds` is deliberately NOT filtered by this.** The
747 /// global ceiling bounds storage on a small box, and an unpollable row
748 /// occupies a row, so it counts against the cap. An earlier version of this
749 /// note listed the readers without naming the exception, which read as
750 /// completeness it did not have: a review found the ceiling consuming
751 /// capacity that appeared on no surface, since `/stats` measures the poller
752 /// and excludes these rows. `/admin/metrics` renders
753 /// `store::unpollable_feeds` for exactly that reason.
754 /// **Adding a kind here makes a population of rows due all at once.**
755 /// `store::due_feeds` orders `next_poll IS NOT NULL, next_poll ASC`, so a
756 /// row with no scheduled poll sorts ahead of every dated one. Rows that
757 /// were never pollable have no schedule, so the boot that reclassifies
758 /// them hands the poller a block of N rows that outrank every regular
759 /// feed until they drain — `ceil(N / batch)` ticks, measured, during which
760 /// `/stats` shows a climbing backlog and nothing logs why. Bounded and
761 /// harmless at ninety feeds; not at ten thousand. Whoever wires the next
762 /// kind should seed or stagger `next_poll` for the rows it admits.
763 pub const POLLABLE: &'static [FeedKind] = &[FeedKind::Rss];
764
765 /// The kinds whose entries the retention **window** applies to.
766 ///
767 /// **Age is the wrong retention policy for an archive, and that is a
768 /// measurement, not a preference.** Three real publications, read through
769 /// `standard_site::fetch` on 2026-09-27: Standard.site's newest document was
770 /// **131 days** old, Annotated's **109** (with its oldest at 373), and minus
771 /// listens' **241**. Against the instance default window of 14 days, every
772 /// one of them stored **zero** rows — a green poll, an empty feed, and an
773 /// info log as the only trace. Long-form publishing is not news-paced.
774 ///
775 /// So a publication is bounded by COUNT instead: `max_entries_per_feed` in
776 /// `insert_entries`, which caps a feed at the newest N plus up to N starred.
777 /// That is a real bound — it is what keeps this from being "retention off" —
778 /// and it is the one that suits a source whose value is its archive.
779 ///
780 /// A generous absolute ceiling still applies (`publication_retention_days`),
781 /// because "not aged out" must not mean "immortal": rows belonging to a feed
782 /// nobody polls any more would otherwise never be reaped at all, and the
783 /// per-feed trim only runs when a poll stores something.
784 pub const AGED: &'static [FeedKind] = &[FeedKind::Rss];
785
786 /// The column value. Stable — it is persisted.
787 pub fn as_str(self) -> &'static str {
788 match self {
789 FeedKind::Rss => "rss",
790 FeedKind::Publication => "publication",
791 }
792 }
793
794 /// A closed vocabulary on the way back in: a kind written by a newer build
795 /// is not silently read as one this build knows.
796 pub fn parse(raw: &str) -> Option<Self> {
797 match raw {
798 "rss" => Some(FeedKind::Rss),
799 "publication" => Some(FeedKind::Publication),
800 _ => None,
801 }
802 }
803
804 /// What a URL will be stored as. The only place the question is asked.
805 ///
806 /// `store::feeds.kind` is a cache of this function, re-derived from the URL
807 /// on every start and on every upsert, so a change here needs no migration
808 /// and SQL cannot hold an opinion of its own about what a row is.
809 ///
810 /// **Case-insensitive on purpose.** An earlier version was case-sensitive,
811 /// with a test pinning that a mixed-case `At://` row IS handed to the
812 /// poller — reasoning that if Rust does not call it an at-URI, neither
813 /// should anything else. That was wrong in the direction that matters: URL
814 /// schemes are case-insensitive, so `Url::parse` folds `At://` to scheme
815 /// `at` and `net::check_scheme` refuses it (the DID form does not parse at
816 /// all). The row could only fail, every tick, forever, and be published in
817 /// the `fetch` bucket as an unreachable publisher. Storing a non-canonical
818 /// spelling is separately refused, because `feeds.url` is UNIQUE.
819 pub fn of(url: &str) -> Self {
820 if crate::atproto::strip_at_prefix(url).is_some() {
821 FeedKind::Publication
822 } else {
823 FeedKind::Rss
824 }
825 }
826}
827
828/// Apply a [`PollOutcome`] to the feed's row: settle the error columns AND
829/// reschedule it. **Both halves, always, from one place.**
830///
831/// The scheduler did this inline. `web::add_subscription` then copied only the
832/// first half, so a poll taken off the scheduler could clear a stale failure's
833/// COUNT while leaving the feed parked on its stale backoff HORIZON — reported
834/// healthy, not polled for up to 24h. And its failures fed `consecutive_errors`
835/// with no reschedule, so repeated Subscribe clicks drove a shared feed to the
836/// 24h ceiling for every subscriber. Two copies of a sequence drift; this is
837/// the one copy.
838///
839/// `cadence` is the interval to use on success; the scheduler derives it from
840/// the feed's hint, a direct caller passes the configured default. A store
841/// failure is logged and the reschedule still attempted, so a hiccup writing
842/// the count cannot strand the feed at a NULL `next_poll` that `due_feeds`
843/// would then re-poll every tick.
844pub async fn settle_poll(
845 pool: &sqlx::SqlitePool,
846 url: &str,
847 outcome: &PollOutcome,
848 cadence: Duration,
849) {
850 let delay = match outcome {
851 // A 304 is a healthy poll: it proves the fetch worked and nothing changed.
852 PollOutcome::Updated { .. } | PollOutcome::NotModified => {
853 if let Err(err) = crate::store::reset_feed_errors(pool, url).await {
854 tracing::warn!(feed = %url, %err, "failed to reset feed error count");
855 }
856 cadence
857 }
858 PollOutcome::Failed {
859 backoff,
860 kind,
861 detail,
862 } => {
863 // Recompute from the feed's REAL consecutive-error count so a
864 // persistently-broken feed climbs toward the ceiling instead of
865 // retrying at the floor forever; fall back to the outcome's floor.
866 match crate::store::bump_feed_errors(pool, url, *kind, detail).await {
867 Ok(count) => backoff_for(count.max(1) as u32),
868 Err(err) => {
869 tracing::warn!(feed = %url, %err, "failed to bump feed error count; using floor backoff");
870 *backoff
871 }
872 }
873 }
874 };
875 if let Err(err) = crate::store::set_next_poll(pool, url, delay).await {
876 tracing::error!(feed = %url, %err, "failed to persist next_poll");
877 }
878}
879
880/// Cap on a failure detail, applied where the string is BUILT.
881///
882/// It was originally applied only inside `store::bump_feed_errors`, which
883/// bounded the database row and nothing else — and widening
884/// [`PollOutcome::Failed`] with this field had quietly opened a second sink:
885/// `web::add_subscription` logs `?outcome` at INFO on a user-facing request
886/// path. Bounding at construction bounds every sink, including ones added
887/// later by someone who never reads this comment.
888pub const MAX_FAILURE_DETAIL_CHARS: usize = 300;
889
890/// Render an error chain into a bounded [`PollOutcome::Failed`] detail.
891///
892/// `{e:#}` — the anyhow CHAIN, not just the outermost context. "fetching
893/// https://…" alone says nothing; the cause is the part that would have named
894/// the 304 bug in #159.
895pub fn failure_detail(err: impl std::fmt::Display) -> String {
896 let s = err.to_string();
897 if s.chars().count() <= MAX_FAILURE_DETAIL_CHARS {
898 return s;
899 }
900 s.chars().take(MAX_FAILURE_DETAIL_CHARS).collect()
901}
902
903impl FailureKind {
904 /// The stable string stored in `feeds.last_error_kind` and aggregated on
905 /// `/stats`. Changing one of these silently rewrites history in the
906 /// aggregate, so they are spelled out rather than derived from the variant.
907 pub fn as_str(self) -> &'static str {
908 match self {
909 Self::Fetch => "fetch",
910 Self::Status => "status",
911 Self::Body => "body",
912 Self::Parse => "parse",
913 }
914 }
915
916 /// Read back a persisted `last_error_kind`. `None` for anything this
917 /// version does not know, so a row written by a newer build is not
918 /// silently attributed to a cause this one recognises — the same contract
919 /// [`crate::metrics::Backend::parse`] keeps for the same reason.
920 pub fn parse(raw: &str) -> Option<Self> {
921 match raw {
922 "fetch" => Some(Self::Fetch),
923 "status" => Some(Self::Status),
924 "body" => Some(Self::Body),
925 "parse" => Some(Self::Parse),
926 _ => None,
927 }
928 }
929
930 /// Every variant, so a test can assert over the whole set rather than a
931 /// list that drifts when a variant is added.
932 pub const ALL: [Self; 4] = [Self::Fetch, Self::Status, Self::Body, Self::Parse];
933}
934
935/// Build a `reqwest::Client` configured for polite **and safe** feed fetching.
936///
937/// Callers should build this **once** and share it (connection pooling), then
938/// hand a reference to [`poll_feed`]. Kept here so the fetch policy (UA,
939/// timeout, redirect behaviour) lives with the code that depends on it.
940///
941/// Auto-redirect is **disabled** on purpose: feed URLs are untrusted, so
942/// redirects are followed manually by [`crate::net::guarded_get`], which
943/// re-validates the scheme + resolved IP of every hop (SSRF defence). A client
944/// that silently followed redirects could be bounced onto `169.254.169.254` or
945/// `127.0.0.1` between the guard's check and the connect.
946pub fn build_client() -> Result<Client> {
947 Client::builder()
948 .user_agent(crate::USER_AGENT)
949 .timeout(FETCH_TIMEOUT)
950 .read_timeout(READ_TIMEOUT)
951 // Ignore ambient proxy configuration, for the same reason the pinned
952 // client does: a proxied request hands the hostname to the proxy to
953 // resolve, so `net`'s IP checks never see the address they are meant to
954 // vet. See `net::build_pinned_client`.
955 .no_proxy()
956 // No auto-redirect: net::guarded_get follows + re-validates each hop.
957 .redirect(reqwest::redirect::Policy::none())
958 .build()
959 .context("failed to build feed HTTP client")
960}
961
962/// Compute the backoff for the `n`th consecutive failure (1-based), clamped to
963/// `BACKOFF_MAX`. Exponential in the error count so transient blips retry soon
964/// while a durably-broken feed backs off toward daily.
965///
966/// The scheduler passes the feed's persisted `consecutive_errors` count (see
967/// [`crate::store::bump_feed_errors`]) so a feed that keeps failing actually
968/// climbs toward `BACKOFF_MAX` instead of retrying at the floor forever.
969pub fn backoff_for(consecutive_errors: u32) -> Duration {
970 let n = consecutive_errors.max(1);
971 // Saturating shift: base * 2^(n-1), capped. Avoids overflow for large n.
972 let factor = 1u64.checked_shl(n.saturating_sub(1)).unwrap_or(u64::MAX);
973 let secs = BACKOFF_BASE
974 .as_secs()
975 .saturating_mul(factor)
976 .min(BACKOFF_MAX.as_secs());
977 Duration::from_secs(secs)
978}
979
980/// Fetch, parse, sanitize, normalize, and store a single feed.
981///
982/// Performs a conditional GET using the feed's stored `ETag` / `Last-Modified`.
983/// On `304` it returns [`PollOutcome::NotModified`] without touching entries. On
984/// `200` it parses with `feed-rs`, sanitizes every entry's HTML with `ammonia`,
985/// upserts the feed row (carrying the fresh validators) and inserts new entries
986/// (deduped by GUID). Any fetch/parse error is logged and returned as
987/// [`PollOutcome::Failed`] — it never panics and never propagates as `Err` for
988/// a merely-broken feed, so one bad publisher can't stall the scheduler.
989///
990/// `Err` is reserved for *store* failures (a broken local DB is a real error the
991/// caller should see), not for feed misbehaviour.
992///
993/// `max_entries_per_feed` caps how many entries this feed retains after insert
994/// (newest N by published date); `<= 0` disables the per-feed trim.
995pub async fn poll_feed(
996 pool: &SqlitePool,
997 client: &Client,
998 feed: &Feed,
999 max_entries_per_feed: i64,
1000) -> Result<PollOutcome> {
1001 // --- conditional GET (through the SSRF guard) ----------------------------
1002 // The guard re-validates the scheme + resolved IP of the target and of every
1003 // redirect hop, so a subscribed feed can't bounce the poller onto an
1004 // internal address (cloud metadata / loopback). Conditional-GET validators
1005 // ride along as extra headers.
1006 let mut extra: Vec<(reqwest::header::HeaderName, reqwest::header::HeaderValue)> = Vec::new();
1007 if let Some(etag) = feed.etag.as_deref() {
1008 if let Ok(v) = reqwest::header::HeaderValue::from_str(etag) {
1009 extra.push((IF_NONE_MATCH, v));
1010 }
1011 }
1012 if let Some(lm) = feed.last_modified.as_deref() {
1013 if let Ok(v) = reqwest::header::HeaderValue::from_str(lm) {
1014 extra.push((IF_MODIFIED_SINCE, v));
1015 }
1016 }
1017
1018 let resp = match crate::net::guarded_get(client, &feed.url, &extra).await {
1019 Ok(r) => r,
1020 Err(e) => {
1021 tracing::warn!(feed = %feed.url, error = %e, "feed fetch failed (or blocked by SSRF guard)");
1022 return Ok(PollOutcome::Failed {
1023 backoff: backoff_for(1),
1024 kind: FailureKind::Fetch,
1025 detail: failure_detail(format!("{e:#}")),
1026 });
1027 }
1028 };
1029
1030 let status = resp.status();
1031 if status == StatusCode::NOT_MODIFIED {
1032 tracing::debug!(feed = %feed.url, "feed not modified (304)");
1033 // Bump last_polled/next_poll only; leave validators + entries untouched.
1034 touch_polled(pool, &feed.url, None, None)
1035 .await
1036 .with_context(|| format!("touch_polled after 304 for {}", feed.url))?;
1037 return Ok(PollOutcome::NotModified);
1038 }
1039 if !status.is_success() {
1040 tracing::warn!(feed = %feed.url, %status, "feed returned non-success status");
1041 return Ok(PollOutcome::Failed {
1042 backoff: backoff_for(1),
1043 kind: FailureKind::Status,
1044 detail: failure_detail(status),
1045 });
1046 }
1047
1048 // Capture validators for the *next* conditional GET before consuming body.
1049 let new_etag = header_str(resp.headers().get(ETAG));
1050 let new_last_modified = header_str(resp.headers().get(LAST_MODIFIED));
1051
1052 // Stream the body with a hard byte cap, aborting mid-stream if it exceeds
1053 // it. We never trust Content-Length: reqwest's gzip layer strips it, so a
1054 // small gzip bomb could otherwise inflate to GBs before any size check.
1055 let body = match crate::net::read_capped(resp).await {
1056 Ok(b) => b,
1057 Err(e) => {
1058 tracing::warn!(feed = %feed.url, error = %e, "feed body rejected (too large / read error)");
1059 return Ok(PollOutcome::Failed {
1060 backoff: backoff_for(1),
1061 kind: FailureKind::Body,
1062 detail: failure_detail(format!("{e:#}")),
1063 });
1064 }
1065 };
1066
1067 // --- parse (malformed feed => log + skip, never panic) -------------------
1068 let parsed = match feed_rs::parser::parse(&body[..]) {
1069 Ok(f) => f,
1070 Err(e) => {
1071 tracing::warn!(feed = %feed.url, error = %e, "malformed feed; skipping");
1072 return Ok(PollOutcome::Failed {
1073 backoff: backoff_for(1),
1074 kind: FailureKind::Parse,
1075 detail: failure_detail(format!("{e:#}")),
1076 });
1077 }
1078 };
1079
1080 // --- normalize + sanitize ------------------------------------------------
1081 let (title, site_url) = feed_metadata(&parsed);
1082 let new_feed = NewFeed {
1083 url: feed.url.clone(),
1084 title,
1085 site_url,
1086 etag: new_etag,
1087 last_modified: new_last_modified,
1088 last_polled: Some(now_rfc3339()),
1089 next_poll: None, // the scheduler owns cadence; leave it to set next_poll.
1090 };
1091
1092 let entries: Vec<NewEntry> = parsed.entries.iter().map(normalize_entry).collect();
1093
1094 // --- store (a store failure IS a real error) -----------------------------
1095 let feed_id = store::upsert_feed(pool, &new_feed)
1096 .await
1097 .with_context(|| format!("upsert_feed for {}", feed.url))?;
1098 let n = store::insert_entries(pool, feed_id, &entries, max_entries_per_feed)
1099 .await
1100 .with_context(|| format!("insert_entries for {}", feed.url))?;
1101
1102 tracing::info!(feed = %feed.url, entries = n, "feed polled");
1103 Ok(PollOutcome::Updated { new_entries: n })
1104}
1105
1106/// Bump `last_polled` (and optionally validators) without changing entries —
1107/// used on the `304 Not Modified` path.
1108async fn touch_polled(
1109 pool: &SqlitePool,
1110 url: &str,
1111 etag: Option<String>,
1112 last_modified: Option<String>,
1113) -> Result<()> {
1114 // `None` means "keep current" — upsert_feed COALESCEs the validators, so a
1115 // 304 that repeats no headers leaves the stored ones untouched. This used to
1116 // re-read the row and re-supply them by hand because the upsert clobbered
1117 // unconditionally; the read-modify-write is gone now that the upsert is
1118 // honest, and with it a race where a concurrent poll's validators could be
1119 // read here and written back stale.
1120 let nf = NewFeed {
1121 url: url.to_string(),
1122 etag,
1123 last_modified,
1124 last_polled: Some(now_rfc3339()),
1125 ..Default::default()
1126 };
1127 store::upsert_feed(pool, &nf).await?;
1128 Ok(())
1129}
1130
1131/// Extract `(title, site_url)` from a parsed feed. `site_url` prefers an
1132/// `alternate`/no-rel HTML link over the feed's self link.
1133fn feed_metadata(parsed: &RawFeed) -> (Option<String>, Option<String>) {
1134 let title = parsed.title.as_ref().map(text_plain);
1135 let site_url = parsed
1136 .links
1137 .iter()
1138 // Prefer an explicit human-facing page: rel="alternate" or no rel at all.
1139 .find(|l| {
1140 l.rel.as_deref() == Some("alternate")
1141 || (l.rel.is_none()
1142 && l.media_type.as_deref() != Some("application/rss+xml")
1143 && l.media_type.as_deref() != Some("application/atom+xml"))
1144 })
1145 .or_else(|| {
1146 parsed
1147 .links
1148 .iter()
1149 .find(|l| l.rel.as_deref() != Some("self"))
1150 })
1151 .or_else(|| parsed.links.first())
1152 .map(|l| l.href.clone());
1153 (title, site_url)
1154}
1155
1156/// Turn a parsed [`RawEntry`] into the store's [`NewEntry`], sanitizing HTML.
1157///
1158/// Content preference: full `content` body, else `summary`. Whichever is chosen
1159/// is **always** passed through [`sanitize_html`] before storage. GUID falls
1160/// back to the entry link, then to a stable hash of title+link, so an entry
1161/// missing an `id` still deduplicates instead of being re-inserted forever.
1162fn normalize_entry(e: &RawEntry) -> NewEntry {
1163 let url = entry_link(e);
1164 let content_html = e
1165 .content
1166 .as_ref()
1167 .and_then(|c| c.body.as_deref())
1168 .or_else(|| e.summary.as_ref().map(|t| t.content.as_str()))
1169 .map(sanitize_html);
1170
1171 // GUID may use the raw link (dedup key only, never rendered), so prefer the
1172 // entry's first raw link for identity even when it's not a safe href.
1173 let guid = if !e.id.trim().is_empty() {
1174 e.id.trim().to_string()
1175 } else if let Some(link) = raw_entry_link(e) {
1176 link
1177 } else {
1178 // Last resort: derive a stable id so re-fetches dedup rather than dupe.
1179 stable_guid(e)
1180 };
1181
1182 NewEntry {
1183 guid,
1184 url,
1185 title: e.title.as_ref().map(text_plain),
1186 author: entry_author(e),
1187 published: entry_time(e),
1188 content_html,
1189 fetched_at: None, // store defaults to "now".
1190 }
1191}
1192
1193/// The raw best-permalink URL for an entry (no scheme filtering) — used only as
1194/// a dedup GUID, never rendered as an href.
1195fn raw_entry_link(e: &RawEntry) -> Option<String> {
1196 e.links
1197 .iter()
1198 .find(|l| l.rel.as_deref() == Some("alternate") || l.rel.is_none())
1199 .or_else(|| e.links.first())
1200 .map(|l| l.href.clone())
1201}
1202
1203/// The best display/permalink URL for an entry, **scheme-allow-listed** so it is
1204/// safe to render as an `href`: prefer `rel="alternate"` or a no-rel link, else
1205/// the first link — but only if it is an `http`/`https` URL. A `javascript:` or
1206/// `data:` permalink (a stored-XSS vector that survives HTML escaping, since it
1207/// carries no HTML-special characters) is dropped here at ingest, before it can
1208/// ever reach the store or a template.
1209fn entry_link(e: &RawEntry) -> Option<String> {
1210 raw_entry_link(e).and_then(|href| crate::net::safe_link(&href))
1211}
1212
1213/// First author name, if any.
1214fn entry_author(e: &RawEntry) -> Option<String> {
1215 e.authors.first().map(|p| p.name.clone())
1216}
1217
1218/// Best publication time (published, else updated) as an RFC3339 string.
1219fn entry_time(e: &RawEntry) -> Option<String> {
1220 e.published.or(e.updated).map(fmt_time)
1221}
1222
1223/// Extract the plain string content of a feed [`Text`] node.
1224fn text_plain(t: &Text) -> String {
1225 t.content.trim().to_string()
1226}
1227
1228/// Sanitize hostile feed **HTML** with ammonia's whitelist cleaner. Applied to
1229/// every RSS/Atom entry body unconditionally, because every one of them is
1230/// markup.
1231///
1232/// **Not for plain text.** An earlier version of this comment claimed it was
1233/// "safe on plain text too (it will simply escape/strip as needed)". It is
1234/// not: `clean` PARSES its input, so a bare `<` in prose swallows the rest —
1235/// `"if x<y then z"` comes back as `"if x"`. That sentence is how a
1236/// plain-text field got run through here once already. Use
1237/// [`plain_text_to_html`].
1238pub(crate) fn sanitize_html(raw: &str) -> String {
1239 ammonia::clean(raw)
1240}
1241
1242/// Render **plain text** into the HTML the `content_html` column holds.
1243///
1244/// **Not [`sanitize_html`].** `ammonia::clean` parses its input as markup, so a
1245/// bare `<` in prose swallows the rest: measured here, `"if x<y then z"` comes
1246/// back as `"if x"`. That is correct for an RSS body, which IS markup, and
1247/// silent data loss for a field a lexicon defines as text. Escape first, then
1248/// add the only markup this needs — line breaks, which the column's consumer
1249/// renders as HTML and would otherwise collapse.
1250pub(crate) fn plain_text_to_html(raw: &str) -> String {
1251 let escaped = raw
1252 .replace('&', "&")
1253 .replace('<', "<")
1254 .replace('>', ">");
1255 // Safe by order: every `<` from the input is already `<` before this
1256 // adds a real tag.
1257 escaped.replace('\n', "<br>")
1258}
1259
1260/// Format a chrono timestamp as RFC3339 (UTC, seconds precision) to match the
1261/// store's string columns.
1262pub(crate) fn fmt_time(dt: DateTime<Utc>) -> String {
1263 dt.to_rfc3339_opts(SecondsFormat::Secs, true)
1264}
1265
1266/// "Now" in the store's RFC3339 shape.
1267fn now_rfc3339() -> String {
1268 Utc::now().to_rfc3339_opts(SecondsFormat::Secs, true)
1269}
1270
1271/// A stable GUID derived from an entry's title + first link, for feeds that
1272/// supply neither an id nor a usable link id. Deterministic so re-fetches dedup.
1273fn stable_guid(e: &RawEntry) -> String {
1274 use std::hash::{Hash, Hasher};
1275 let mut h = std::collections::hash_map::DefaultHasher::new();
1276 e.title.as_ref().map(|t| t.content.as_str()).hash(&mut h);
1277 e.links.first().map(|l| l.href.as_str()).hash(&mut h);
1278 e.summary.as_ref().map(|s| s.content.as_str()).hash(&mut h);
1279 format!("featherreader:synthetic:{:016x}", h.finish())
1280}
1281
1282/// Decode an HTTP header value to an owned `String`, dropping non-UTF-8 values.
1283fn header_str(v: Option<&reqwest::header::HeaderValue>) -> Option<String> {
1284 v.and_then(|h| h.to_str().ok()).map(str::to_string)
1285}
1286
1287/// Discover a feed URL from a site's HTML via
1288/// `<link rel="alternate" type="application/rss+xml|atom+xml" href="…">`.
1289///
1290/// Returns the first RSS/Atom autodiscovery link found, resolved against the
1291/// page URL if the `href` is relative. This is what lets a user paste a *site*
1292/// URL and have FeatherReader find the actual feed ("subscribe by URL").
1293/// Returns `None` if the HTML carries no autodiscovery link.
1294///
1295/// The `base` is the URL the HTML was fetched from, used to resolve relative
1296/// `href`s. Pass `None` to only accept absolute hrefs.
1297pub fn discover_feed(site_html: &str, base: Option<&Url>) -> Option<Url> {
1298 // Parse the HTML with html5ever (via ammonia's dependency graph is separate;
1299 // use a light hand-rolled scan over <link> tags to avoid a new dependency).
1300 // We look for <link ...> elements whose rel contains "alternate" and whose
1301 // type is an RSS/Atom feed media type, and take the href.
1302 for tag in link_tags(site_html) {
1303 let rel = attr(&tag, "rel").unwrap_or_default().to_ascii_lowercase();
1304 let typ = attr(&tag, "type").unwrap_or_default().to_ascii_lowercase();
1305 let is_feed_type = typ.contains("application/rss+xml")
1306 || typ.contains("application/atom+xml")
1307 || typ.contains("application/feed+json")
1308 || typ.contains("application/json");
1309 // rel="alternate" is the standard; be lenient and also accept a bare
1310 // feed type with any rel, but require the feed media type either way.
1311 let rel_ok = rel.split_whitespace().any(|r| r == "alternate") || rel.is_empty();
1312 if is_feed_type && rel_ok {
1313 if let Some(href) = attr(&tag, "href") {
1314 let href = href.trim();
1315 if href.is_empty() {
1316 continue;
1317 }
1318 // Absolute URL wins directly; otherwise resolve against `base`.
1319 // Either way, only http(s): the href is publisher-controlled and
1320 // `Url::parse` accepts any scheme, so this is where an `at://`
1321 // (or `file:`, `javascript:`) alternate would otherwise become
1322 // the URL the add path stores — after its input gate has run.
1323 // Skip, don't stop: a later real feed link still wins.
1324 let resolved = match Url::parse(href) {
1325 Ok(u) => Some(u),
1326 Err(_) => base.and_then(|b| b.join(href).ok()),
1327 };
1328 match resolved {
1329 Some(u) if matches!(u.scheme(), "http" | "https") => return Some(u),
1330 _ => continue,
1331 }
1332 }
1333 }
1334 }
1335 None
1336}
1337
1338/// Extract the raw text of every `<link ...>` tag (self-closing or not) from an
1339/// HTML string. A deliberately small, allocation-light scan — feed
1340/// autodiscovery does not need a full DOM, and avoiding one keeps the dependency
1341/// surface minimal (design bias: boring, small-dependency).
1342fn link_tags(html: &str) -> Vec<String> {
1343 let mut out = Vec::new();
1344 let bytes = html.as_bytes();
1345 let lower = html.to_ascii_lowercase();
1346 let mut search_from = 0usize;
1347 while let Some(rel_idx) = lower[search_from..].find("<link") {
1348 let start = search_from + rel_idx;
1349 // Ensure it's a tag boundary ("<link" followed by whitespace, '>' or '/').
1350 let after = bytes.get(start + 5).copied();
1351 let boundary = matches!(after, Some(b) if b == b' ' || b == b'\t' || b == b'\n' || b == b'\r' || b == b'>' || b == b'/');
1352 if !boundary {
1353 search_from = start + 5;
1354 continue;
1355 }
1356 // Find the closing '>' for this tag.
1357 if let Some(end_rel) = html[start..].find('>') {
1358 let end = start + end_rel;
1359 out.push(html[start..=end].to_string());
1360 search_from = end + 1;
1361 } else {
1362 break;
1363 }
1364 }
1365 out
1366}
1367
1368/// Read an attribute value from a single tag string, handling both single- and
1369/// double-quoted values. Case-insensitive attribute name match.
1370fn attr(tag: &str, name: &str) -> Option<String> {
1371 let lower = tag.to_ascii_lowercase();
1372 let needle = format!("{name}=");
1373 let mut from = 0usize;
1374 while let Some(rel) = lower[from..].find(&needle) {
1375 let name_start = from + rel;
1376 // Guard against matching a suffix of a longer attribute name
1377 // (e.g. matching "type=" inside "mytype="): the char before must be a
1378 // tag/whitespace boundary.
1379 let ok_prefix = name_start == 0
1380 || matches!(
1381 tag.as_bytes().get(name_start - 1),
1382 Some(b' ') | Some(b'\t') | Some(b'\n') | Some(b'\r') | Some(b'<')
1383 );
1384 let val_start = name_start + needle.len();
1385 if !ok_prefix {
1386 from = val_start;
1387 continue;
1388 }
1389 let rest = &tag[val_start..];
1390 let quote = rest.chars().next();
1391 let value = match quote {
1392 Some('"') => rest[1..].split('"').next(),
1393 Some('\'') => rest[1..].split('\'').next(),
1394 // Unquoted: read up to whitespace, '>' or '/'.
1395 _ => rest
1396 .split(|c: char| c.is_whitespace() || c == '>' || c == '/')
1397 .next(),
1398 };
1399 return value.map(str::to_string);
1400 }
1401 None
1402}
1403
1404#[cfg(test)]
1405mod tests {
1406 use super::*;
1407
1408 const RSS_SAMPLE: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
1409<rss version="2.0">
1410 <channel>
1411 <title>Example RSS Feed</title>
1412 <link>https://example.com/</link>
1413 <description>An example feed for tests</description>
1414 <item>
1415 <title>First post</title>
1416 <link>https://example.com/first</link>
1417 <guid>https://example.com/first</guid>
1418 <author>alice@example.com (Alice)</author>
1419 <pubDate>Fri, 10 Jul 2026 08:00:00 GMT</pubDate>
1420 <description><![CDATA[<p>Hello <b>world</b>.</p><script>alert('xss')</script><img src="x" onerror="alert(1)">]]></description>
1421 </item>
1422 <item>
1423 <title>Second post</title>
1424 <link>https://example.com/second</link>
1425 <guid>guid-second</guid>
1426 <pubDate>Sat, 11 Jul 2026 08:00:00 GMT</pubDate>
1427 <description><![CDATA[<a href="javascript:alert(1)">click</a><a href="https://ok.example/">ok</a>]]></description>
1428 </item>
1429 </channel>
1430</rss>"#;
1431
1432 const ATOM_SAMPLE: &str = r#"<?xml version="1.0" encoding="utf-8"?>
1433<feed xmlns="http://www.w3.org/2005/Atom">
1434 <title>Example Atom Feed</title>
1435 <link rel="alternate" href="https://atom.example.com/"/>
1436 <link rel="self" href="https://atom.example.com/feed.xml"/>
1437 <id>urn:uuid:feed-1</id>
1438 <updated>2026-07-11T08:00:00Z</updated>
1439 <entry>
1440 <title>Atom entry</title>
1441 <id>urn:uuid:entry-1</id>
1442 <link rel="alternate" href="https://atom.example.com/a"/>
1443 <author><name>Bob</name></author>
1444 <updated>2026-07-11T08:00:00Z</updated>
1445 <content type="html"><![CDATA[<p>Safe <em>text</em>.</p><script>steal()</script><iframe src="evil"></iframe>]]></content>
1446 </entry>
1447</feed>"#;
1448
1449 /// [`ATOM_SAMPLE`] with the `self` link before the `alternate` one.
1450 const ATOM_SELF_FIRST: &str = r#"<?xml version="1.0" encoding="utf-8"?>
1451<feed xmlns="http://www.w3.org/2005/Atom">
1452 <title>Example Atom Feed</title>
1453 <link rel="self" href="https://atom.example.com/feed.xml"/>
1454 <link rel="alternate" href="https://atom.example.com/"/>
1455 <id>urn:uuid:feed-1</id>
1456 <updated>2026-07-11T08:00:00Z</updated>
1457 <entry>
1458 <title>Atom entry</title>
1459 <id>urn:uuid:entry-1</id>
1460 <link rel="alternate" href="https://atom.example.com/a"/>
1461 <author><name>Bob</name></author>
1462 <updated>2026-07-11T08:00:00Z</updated>
1463 <content type="html"><![CDATA[<p>Safe <em>text</em>.</p><script>steal()</script><iframe src="evil"></iframe>]]></content>
1464 </entry>
1465</feed>"#;
1466
1467 /// Parse a static RSS sample through feed-rs + our normalize/sanitize path
1468 /// (no network) and assert the entries come out sanitized and well-shaped.
1469 #[test]
1470 fn rss_parses_and_sanitizes() {
1471 let parsed = feed_rs::parser::parse(RSS_SAMPLE.as_bytes()).expect("RSS should parse");
1472 assert_eq!(
1473 parsed.title.as_ref().map(text_plain).as_deref(),
1474 Some("Example RSS Feed")
1475 );
1476 assert_eq!(parsed.entries.len(), 2);
1477
1478 let (title, site) = feed_metadata(&parsed);
1479 assert_eq!(title.as_deref(), Some("Example RSS Feed"));
1480 assert_eq!(site.as_deref(), Some("https://example.com/"));
1481
1482 let e0 = normalize_entry(&parsed.entries[0]);
1483 assert_eq!(e0.guid, "https://example.com/first");
1484 assert_eq!(e0.title.as_deref(), Some("First post"));
1485 assert_eq!(e0.url.as_deref(), Some("https://example.com/first"));
1486 assert!(e0.published.is_some());
1487 let html0 = e0.content_html.expect("content present");
1488 // Sanitized: benign markup kept, script + onerror stripped.
1489 assert!(html0.contains("Hello"));
1490 assert!(html0.contains("<b>world</b>") || html0.contains("<b>"));
1491 assert!(!html0.to_ascii_lowercase().contains("<script"));
1492 assert!(!html0.to_ascii_lowercase().contains("onerror"));
1493 assert!(!html0.to_ascii_lowercase().contains("alert"));
1494
1495 // Second entry: javascript: URL scrubbed, safe link kept.
1496 let e1 = normalize_entry(&parsed.entries[1]);
1497 assert_eq!(e1.guid, "guid-second");
1498 let html1 = e1.content_html.expect("content present");
1499 assert!(!html1.to_ascii_lowercase().contains("javascript:"));
1500 assert!(html1.contains("https://ok.example/"));
1501 }
1502
1503 /// Same, for an Atom sample: alternate link is the site URL, dangerous
1504 /// elements are stripped from entry content.
1505 /// **`rel="alternate"` is preferred over a `rel="self"` listed FIRST.** In
1506 /// `ATOM_SAMPLE` the alternate link is already first, so "prefer alternate"
1507 /// and "take the first link" were indistinguishable; the selection could
1508 /// be replaced by `links.first()` with the suite green. A feed listing
1509 /// `self` first — very common in Atom — would store the feed XML URL as
1510 /// the subscription's `siteUrl`, published to the reader's PDS.
1511 #[test]
1512 fn atom_prefers_alternate_over_a_self_link_listed_first() {
1513 let parsed = feed_rs::parser::parse(ATOM_SELF_FIRST.as_bytes()).expect("Atom should parse");
1514 let (title, site) = feed_metadata(&parsed);
1515 assert_eq!(title.as_deref(), Some("Example Atom Feed"));
1516 // alternate link preferred over rel="self".
1517 assert_eq!(site.as_deref(), Some("https://atom.example.com/"));
1518 }
1519
1520 #[test]
1521 fn atom_parses_and_sanitizes() {
1522 let parsed = feed_rs::parser::parse(ATOM_SAMPLE.as_bytes()).expect("Atom should parse");
1523 let (title, site) = feed_metadata(&parsed);
1524 assert_eq!(title.as_deref(), Some("Example Atom Feed"));
1525 // alternate link preferred over rel="self".
1526 assert_eq!(site.as_deref(), Some("https://atom.example.com/"));
1527
1528 assert_eq!(parsed.entries.len(), 1);
1529 let e = normalize_entry(&parsed.entries[0]);
1530 assert_eq!(e.guid, "urn:uuid:entry-1");
1531 assert_eq!(e.title.as_deref(), Some("Atom entry"));
1532 assert_eq!(e.author.as_deref(), Some("Bob"));
1533 assert_eq!(e.url.as_deref(), Some("https://atom.example.com/a"));
1534 let html = e.content_html.expect("content present");
1535 assert!(html.contains("Safe"));
1536 assert!(!html.to_ascii_lowercase().contains("<script"));
1537 assert!(!html.to_ascii_lowercase().contains("<iframe"));
1538 }
1539
1540 /// **The rkey obeys all of atproto's record-key rules, not just the
1541 /// charset.** `.` and `..` are reserved and the length is 1..=512; the
1542 /// charset alone admitted both and a 10 000-character key, into a UNIQUE
1543 /// column and the user's public PDS. The rule is the one the repo's own
1544 /// TID tests already state.
1545 #[test]
1546 fn an_rkey_must_obey_atprotos_length_and_dot_rules() {
1547 let uri = |rkey: &str| {
1548 format!("at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/{rkey}")
1549 };
1550 assert!(
1551 !is_storable_feed_url(&uri("."), true),
1552 "`.` is a reserved rkey"
1553 );
1554 assert!(
1555 !is_storable_feed_url(&uri(".."), true),
1556 "`..` is a reserved rkey"
1557 );
1558 assert!(
1559 !is_storable_feed_url(&uri(&"a".repeat(513)), true),
1560 "an rkey over 512 bytes was accepted"
1561 );
1562 assert!(
1563 is_storable_feed_url(&uri(&"a".repeat(512)), true),
1564 "an rkey of exactly 512 bytes is valid"
1565 );
1566 assert!(is_storable_feed_url(&uri("3lab2c4d5e6f7g8h"), true));
1567 }
1568
1569 /// **Every `KNOWN_PROVIDERS` row is pinned by its own reason.**
1570 ///
1571 /// The provider test used URLs that the generic heuristics catch on their
1572 /// own — Substack's `/feed/private/` is also a path marker, Patreon's
1573 /// `?auth=…` an opaque secret key — and never asserted the reason. With
1574 /// 17 of 18 rows deleted, the suite stayed green. The provider layer runs
1575 /// FIRST so its specific reason wins; asserting the reason pins each row
1576 /// even where a generic rule would still refuse the URL. Values are kept
1577 /// short and plain so the generic query rule (`value_is_opaque`) does not
1578 /// fire — most of these are refused by the provider row alone.
1579 #[test]
1580 fn each_known_provider_is_caught_by_its_own_row() {
1581 for (url, reason) in [
1582 (
1583 "https://author.substack.com/feed/private/x",
1584 "Substack private feed",
1585 ),
1586 (
1587 "https://www.patreon.com/rss/creator?auth=ab",
1588 "Patreon member feed",
1589 ),
1590 ("https://blog.ghost.io/rss/?uuid=x", "Ghost members feed"),
1591 (
1592 "https://buttondown.email/me/rss?token=x",
1593 "Buttondown premium feed",
1594 ),
1595 (
1596 "https://buttondown.com/me/rss?token=x",
1597 "Buttondown premium feed",
1598 ),
1599 (
1600 "https://rss.beehiiv.com/feeds/x.xml?token=x",
1601 "Beehiiv premium feed",
1602 ),
1603 (
1604 "https://example.memberful.com/feed",
1605 "Memberful members feed",
1606 ),
1607 ("https://example.pico.link/feed", "Pico member feed"),
1608 ("https://steadyhq.com/rss/example", "Steady member feed"),
1609 (
1610 "https://example.supercast.com/feed",
1611 "Supercast private podcast",
1612 ),
1613 (
1614 "https://example.supercast.tech/feed",
1615 "Supercast private podcast",
1616 ),
1617 (
1618 "https://example.supportingcast.fm/feed",
1619 "Supporting Cast private podcast",
1620 ),
1621 (
1622 "https://feeds.redcircle.com/x?private=1",
1623 "RedCircle private podcast",
1624 ),
1625 (
1626 "https://feeds.megaphone.fm/x?token=x",
1627 "Megaphone private podcast",
1628 ),
1629 (
1630 "https://feeds.acast.com/public/shows/x?token=x",
1631 "Acast+ private podcast",
1632 ),
1633 (
1634 "https://omny.fm/shows/x/playlists/podcast.rss?token=x",
1635 "Omny private podcast",
1636 ),
1637 (
1638 "https://podcasts.apple.com/feed/x?token=x",
1639 "Apple subscriber podcast",
1640 ),
1641 (
1642 "https://anchor.spotify.com/s/x/podcast/rss?token=x",
1643 "Spotify subscriber podcast",
1644 ),
1645 ] {
1646 match classify_feed_privacy(url) {
1647 FeedPrivacy::Private(r) => {
1648 assert_eq!(r, reason, "{url} was refused by another rule")
1649 }
1650 FeedPrivacy::Public => panic!("{url} was not refused at all"),
1651 }
1652 }
1653 }
1654
1655 /// **Plain text is escaped, not sanitised.** `ammonia::clean` parses its
1656 /// input as markup, so a `<` in prose swallows everything after it:
1657 /// measured in this tree, `"if x<y then z"` becomes `"if x"`. That is the
1658 /// right function for an RSS body (which IS markup) and exactly the wrong
1659 /// one for a field the lexicon defines as plain text — it silently deletes
1660 /// the reader's content.
1661 #[test]
1662 fn plain_text_is_escaped_rather_than_swallowed() {
1663 assert_eq!(
1664 plain_text_to_html("Vec<String> is a type"),
1665 "Vec<String> is a type"
1666 );
1667 assert_eq!(plain_text_to_html("if x<y then z"), "if x<y then z");
1668 assert_eq!(plain_text_to_html("a & b"), "a & b");
1669 // Line structure survives into a field rendered as HTML.
1670 assert_eq!(plain_text_to_html("one\ntwo"), "one<br>two");
1671 // And it is still safe: the escaping happens before any markup is added.
1672 let hostile = plain_text_to_html("<script>alert(1)</script>");
1673 assert!(!hostile.contains("<script"), "{hostile}");
1674 }
1675
1676 #[test]
1677 fn discover_finds_rss_link() {
1678 let html = r#"<!doctype html><html><head>
1679 <title>Blog</title>
1680 <link rel="stylesheet" href="/style.css">
1681 <link rel="alternate" type="application/rss+xml" title="RSS" href="/feed.xml">
1682 </head><body>hi</body></html>"#;
1683 let base = Url::parse("https://blog.example.com/").unwrap();
1684 let found = discover_feed(html, Some(&base)).expect("should discover feed");
1685 assert_eq!(found.as_str(), "https://blog.example.com/feed.xml");
1686 }
1687
1688 #[test]
1689 fn discover_finds_atom_absolute_link() {
1690 let html = r#"<head><link rel="alternate" type="application/atom+xml" href="https://x.example/atom"></head>"#;
1691 let found = discover_feed(html, None).expect("should discover absolute feed");
1692 assert_eq!(found.as_str(), "https://x.example/atom");
1693 }
1694
1695 /// **Autodiscovery only ever yields an http(s) URL.**
1696 ///
1697 /// The href is publisher-controlled and `Url::parse` accepts any scheme, so
1698 /// a page could hand the add path an `at://` publication URI (or anything
1699 /// else) that the user never typed — and the add path's input gate has
1700 /// already run by then. A non-http(s) alternate is skipped, not returned,
1701 /// so a later real feed link still wins.
1702 #[test]
1703 fn discover_skips_a_non_http_alternate() {
1704 let at_link = r#"<link rel="alternate" type="application/rss+xml" href="at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h">"#;
1705 assert!(
1706 discover_feed(&format!("<head>{at_link}</head>"), None).is_none(),
1707 "an at:// alternate was handed back as a feed URL"
1708 );
1709 let ftp_link =
1710 r#"<link rel="alternate" type="application/atom+xml" href="ftp://x.example/atom">"#;
1711 assert!(discover_feed(&format!("<head>{ftp_link}</head>"), None).is_none());
1712
1713 let real =
1714 r#"<link rel="alternate" type="application/atom+xml" href="https://x.example/atom">"#;
1715 let found = discover_feed(&format!("<head>{at_link}{real}</head>"), None)
1716 .expect("the http(s) link after a skipped one must still be found");
1717 assert_eq!(found.as_str(), "https://x.example/atom");
1718 }
1719
1720 #[test]
1721 fn discover_returns_none_without_feed_link() {
1722 let html =
1723 r#"<head><link rel="stylesheet" href="/s.css"><link rel="icon" href="/f.ico"></head>"#;
1724 assert!(discover_feed(html, None).is_none());
1725 }
1726
1727 #[test]
1728 fn synthetic_guid_is_stable_and_dedups() {
1729 // An item with neither guid nor link: feed-rs will hash the link (absent)
1730 // to a UUID id, but to exercise *our* synthetic fallback we clear the id
1731 // on the parsed entry and confirm normalize yields a deterministic guid.
1732 let xml = r#"<?xml version="1.0"?><rss version="2.0"><channel>
1733 <title>t</title>
1734 <item><title>only a title</title><description>body</description></item>
1735 </channel></rss>"#;
1736 let mut parsed = feed_rs::parser::parse(xml.as_bytes()).expect("parse");
1737 parsed.entries[0].id.clear();
1738 parsed.entries[0].links.clear();
1739 let g1 = normalize_entry(&parsed.entries[0]).guid;
1740 let g2 = normalize_entry(&parsed.entries[0]).guid;
1741 assert_eq!(g1, g2);
1742 assert!(g1.starts_with("featherreader:synthetic:"));
1743 }
1744
1745 #[test]
1746 fn entry_link_scheme_allowlist_neutralizes_javascript() {
1747 // An entry whose only link is a javascript: URL must yield no href.
1748 let xml = r#"<?xml version="1.0"?><rss version="2.0"><channel>
1749 <title>t</title>
1750 <item>
1751 <title>evil</title>
1752 <link>javascript:alert(document.domain)</link>
1753 <guid>evil-1</guid>
1754 </item>
1755 </channel></rss>"#;
1756 let parsed = feed_rs::parser::parse(xml.as_bytes()).expect("parse");
1757 let e = normalize_entry(&parsed.entries[0]);
1758 // url is dropped (not a safe http(s) link)…
1759 assert_eq!(e.url, None);
1760 // …but the entry still dedups (guid preserved from <guid>).
1761 assert_eq!(e.guid, "evil-1");
1762
1763 // A data: URL is likewise dropped.
1764 let xml2 = r#"<?xml version="1.0"?><rss version="2.0"><channel>
1765 <title>t</title>
1766 <item><title>d</title><link>data:text/html,<script>1</script></link><guid>d1</guid></item>
1767 </channel></rss>"#;
1768 let parsed2 = feed_rs::parser::parse(xml2.as_bytes()).expect("parse");
1769 let e2 = normalize_entry(&parsed2.entries[0]);
1770 assert_eq!(e2.url, None);
1771
1772 // A normal https link survives.
1773 let xml3 = r#"<?xml version="1.0"?><rss version="2.0"><channel>
1774 <title>t</title>
1775 <item><title>ok</title><link>https://ok.example/post</link><guid>ok1</guid></item>
1776 </channel></rss>"#;
1777 let parsed3 = feed_rs::parser::parse(xml3.as_bytes()).expect("parse");
1778 let e3 = normalize_entry(&parsed3.entries[0]);
1779 assert_eq!(e3.url.as_deref(), Some("https://ok.example/post"));
1780 }
1781
1782 #[test]
1783 fn classify_privacy_flags_secret_urls_across_providers() {
1784 // --- Known providers: newsletters ---
1785 // Substack private feed path.
1786 assert!(
1787 classify_feed_privacy("https://author.substack.com/feed/private/deadbeefcafe1234")
1788 .is_private()
1789 );
1790 // Patreon ?auth= member feed.
1791 assert!(classify_feed_privacy(
1792 "https://www.patreon.com/rss/author?auth=Zm9vYmFyc2VjcmV0dG9rZW4"
1793 )
1794 .is_private());
1795 // Ghost members feed via ?uuid=.
1796 assert!(classify_feed_privacy(
1797 "https://blog.ghost.io/rss/?uuid=1f2e3d4c-5b6a-7089-90ab-cdef01234567"
1798 )
1799 .is_private());
1800
1801 // --- Known providers: private podcasts ---
1802 // Supporting Cast tokened podcast feed.
1803 assert!(classify_feed_privacy(
1804 "https://feeds.supportingcast.fm/show/abcdef0123456789abcdef01"
1805 )
1806 .is_private());
1807 // Supercast private podcast (host alone is enough).
1808 assert!(classify_feed_privacy("https://feeds.supercast.com/12345/rss").is_private());
1809
1810 // --- Generic, provider-agnostic heuristic ---
1811 // Named credential query params with an opaque value.
1812 assert!(
1813 classify_feed_privacy("https://example.com/feed?token=Zm9vYmFyc2VjcmV0").is_private()
1814 );
1815 assert!(
1816 classify_feed_privacy("https://example.com/feed?key=Zm9vYmFyc2VjcmV0").is_private()
1817 );
1818 assert!(
1819 classify_feed_privacy("https://example.com/feed?secret=Zm9vYmFyc2VjcmV0").is_private()
1820 );
1821 // Userinfo credentials in the authority.
1822 assert!(classify_feed_privacy("https://user:pass@example.com/feed").is_private());
1823 // A `/private/` path segment on an unknown host.
1824 assert!(classify_feed_privacy("https://blog.example.com/private/rss").is_private());
1825 // `/members/` path convention.
1826 assert!(classify_feed_privacy("https://news.example.com/members/feed.xml").is_private());
1827 // A high-entropy opaque token embedded in the path with no telltale name.
1828 assert!(
1829 classify_feed_privacy("https://feeds.example.com/aB3xK9zQ7mP2rT5wL8nD4vF6")
1830 .is_private()
1831 );
1832 // A bare UUID path segment (many tokened feeds).
1833 assert!(classify_feed_privacy(
1834 "https://feeds.example.com/1f2e3d4c-5b6a-7089-90ab-cdef01234567"
1835 )
1836 .is_private());
1837 }
1838
1839 /// The dominant real-world private-podcast shape delivers the token as a
1840 /// FILENAME (`<token>.rss` / `<token>.xml`) or affixed inside a larger
1841 /// segment (`feed-<uuid>`). Named providers are caught by their host rule;
1842 /// these are UNKNOWN-provider CDNs that must still be caught by the generic
1843 /// backstop, so the secret is never fetched or stored.
1844 #[test]
1845 fn classify_privacy_catches_tokened_filenames_on_unknown_hosts() {
1846 // **A stem only the extension-strip branch can see.** Every other case
1847 // here is also caught by the sub-part scan (branch 3) or the UUID scan,
1848 // so deleting the strip left the suite green — `FEED_EXTENSIONS` was
1849 // effectively dead. This stem's `-`-separated parts are each too short
1850 // to look like a secret on their own, and the whole segment fails on
1851 // the `.` — only stripping `.rss` and re-testing the 26-char stem sees
1852 // it. That is exactly the shape a hyphen-bearing base64url token
1853 // filename takes.
1854 assert!(
1855 classify_feed_privacy("https://cdn.example/feeds/aB3xK9pQ-7mZ2vN8w-Qr5tYuW.rss")
1856 .is_private(),
1857 "a token stem visible only after stripping the extension was not caught"
1858 );
1859 // hex-32 token as an .xml filename.
1860 assert!(classify_feed_privacy(
1861 "https://cdn.somepod.io/f/a1b2c3d4e5f60718293a4b5c6d7e8f90.xml"
1862 )
1863 .is_private());
1864 // hex-32 token as a .rss filename on an unknown CDN.
1865 assert!(classify_feed_privacy(
1866 "https://dcs.megaphone.example/network/a1b2c3d4e5f60718293a4b5c6d7e8f90.rss"
1867 )
1868 .is_private());
1869 // UUID + .xml filename.
1870 assert!(classify_feed_privacy(
1871 "https://brandnew.example/feed/1f2e3d4c-5b6a-7089-90ab-cdef01234567.xml"
1872 )
1873 .is_private());
1874 // UUID affixed with a prefix (`feed-<uuid>`) — split can't see it, the
1875 // UUID-substring scan must.
1876 assert!(classify_feed_privacy(
1877 "https://x.example/feed-1f2e3d4c-5b6a-7089-90ab-cdef01234567"
1878 )
1879 .is_private());
1880 // UUID + .rss suffix.
1881 assert!(classify_feed_privacy(
1882 "https://x.example/1f2e3d4c-5b6a-7089-90ab-cdef01234567.rss"
1883 )
1884 .is_private());
1885 // hex-16 token as an .xml filename.
1886 assert!(classify_feed_privacy("https://x.example/feed/9f8e7d6c5b4a3928.xml").is_private());
1887 // A base64url token with `=` padding as a clean path segment.
1888 assert!(
1889 classify_feed_privacy("https://cdn.pod.io/f/YWJjZGVmZ2hpamtsbW5vcHFyc3R1dnc=")
1890 .is_private()
1891 );
1892 }
1893
1894 /// YouTube channel/playlist RSS feeds are FULLY PUBLIC (the id is a public
1895 /// handle, not a secret) and are the standard way to subscribe to a channel —
1896 /// they must NOT be false-blocked by the generic entropy heuristic.
1897 #[test]
1898 fn classify_privacy_allows_public_youtube_feeds() {
1899 assert_eq!(
1900 classify_feed_privacy(
1901 "https://www.youtube.com/feeds/videos.xml?channel_id=UC-lHJZR3Gqxm24_Vd_AJ5Yw"
1902 ),
1903 FeedPrivacy::Public
1904 );
1905 assert_eq!(
1906 classify_feed_privacy(
1907 "https://www.youtube.com/feeds/videos.xml?playlist_id=PLFgquLnL59alCl_2TQvOiD5Vgm1hCaGSI"
1908 ),
1909 FeedPrivacy::Public
1910 );
1911 // Bare host form too.
1912 assert_eq!(
1913 classify_feed_privacy(
1914 "https://youtube.com/feeds/videos.xml?channel_id=UC-lHJZR3Gqxm24_Vd_AJ5Yw"
1915 ),
1916 FeedPrivacy::Public
1917 );
1918 // The allowlist is narrow: a `token=` on the YouTube feeds path still
1919 // classifies private (can't smuggle a credential through the allowlist).
1920 assert!(classify_feed_privacy(
1921 "https://www.youtube.com/feeds/videos.xml?token=Zm9vYmFyc2VjcmV0dG9rZW4"
1922 )
1923 .is_private());
1924 }
1925
1926 #[test]
1927 fn classify_privacy_leaves_normal_public_feeds_public() {
1928 // Plain feed documents.
1929 assert_eq!(
1930 classify_feed_privacy("https://example.com/feed.xml"),
1931 FeedPrivacy::Public
1932 );
1933 assert_eq!(
1934 classify_feed_privacy("https://blog.example.com/rss"),
1935 FeedPrivacy::Public
1936 );
1937 assert_eq!(
1938 classify_feed_privacy("https://blog.example.com/rss.xml"),
1939 FeedPrivacy::Public
1940 );
1941 // A Substack PUBLIC feed (`/feed`, not `/feed/private/`) stays public.
1942 assert_eq!(
1943 classify_feed_privacy("https://author.substack.com/feed"),
1944 FeedPrivacy::Public
1945 );
1946 // A WordPress `/feed` endpoint.
1947 assert_eq!(
1948 classify_feed_privacy("https://wordpress.example.com/feed/"),
1949 FeedPrivacy::Public
1950 );
1951 // A plain Atom feed.
1952 assert_eq!(
1953 classify_feed_privacy("https://example.org/atom.xml"),
1954 FeedPrivacy::Public
1955 );
1956 // A long, hyphenated slug must NOT be mistaken for an embedded secret.
1957 assert_eq!(
1958 classify_feed_privacy("https://example.com/2026/07/my-first-long-blog-post-title/feed"),
1959 FeedPrivacy::Public
1960 );
1961 // A benign query key that merely contains "key" as a substring is fine.
1962 assert_eq!(
1963 classify_feed_privacy("https://example.com/feed?keyword=rust"),
1964 FeedPrivacy::Public
1965 );
1966 // A short, non-opaque value on a named key (e.g. an enum) is not a secret.
1967 assert_eq!(
1968 classify_feed_privacy("https://example.com/feed?p=2"),
1969 FeedPrivacy::Public
1970 );
1971 // An empty credential value is not a secret.
1972 assert_eq!(
1973 classify_feed_privacy("https://example.com/feed?token="),
1974 FeedPrivacy::Public
1975 );
1976 // A hyphenated slug ending in a feed extension must NOT be seen as a
1977 // tokened filename (the stem is short dictionary words, not a blob).
1978 assert_eq!(
1979 classify_feed_privacy("https://example.com/my-first-long-blog-post.xml"),
1980 FeedPrivacy::Public
1981 );
1982 // A short hex episode id in an .xml filename (< 16 chars) is not a secret.
1983 assert_eq!(
1984 classify_feed_privacy("https://example.com/episodes/ab12cd.xml"),
1985 FeedPrivacy::Public
1986 );
1987 // A dotted host-style filename slug stays public.
1988 assert_eq!(
1989 classify_feed_privacy("https://example.com/category/tech-news/feed.xml"),
1990 FeedPrivacy::Public
1991 );
1992 // Unparseable URL: treated as Public (add path rejects it downstream).
1993 assert_eq!(classify_feed_privacy("not a url"), FeedPrivacy::Public);
1994 }
1995
1996 /// **The detail is bounded where it is CONSTRUCTED, not only where it is
1997 /// stored.**
1998 ///
1999 /// Review found that widening `PollOutcome::Failed` with this field opened a
2000 /// second sink nobody looked at: `web.rs`'s `add_subscription` logs
2001 /// `?outcome` at INFO on a user-facing request path, so the whole
2002 /// untruncated anyhow chain — redirect-hop URLs, the SSRF guard's refusal
2003 /// text naming a resolved internal address — went to the access log.
2004 ///
2005 /// Bounding inside `bump_feed_errors` protected the database and nothing
2006 /// else. Bounding at construction protects every sink, including the ones
2007 /// added later.
2008 #[test]
2009 fn a_failure_detail_is_bounded_at_construction() {
2010 let huge = "x".repeat(10_000);
2011 let outcome = PollOutcome::Failed {
2012 backoff: BACKOFF_BASE,
2013 kind: FailureKind::Fetch,
2014 detail: failure_detail(&huge),
2015 };
2016 let PollOutcome::Failed { detail, .. } = &outcome else {
2017 panic!("wrong variant");
2018 };
2019 assert!(
2020 detail.chars().count() <= MAX_FAILURE_DETAIL_CHARS,
2021 "detail was {} chars",
2022 detail.chars().count(),
2023 );
2024 // And the Debug rendering — which is what actually reached the log — is
2025 // bounded with it.
2026 assert!(format!("{outcome:?}").len() < 1_000);
2027 }
2028
2029 /// **Every failure kind has its own label, and they round-trip.**
2030 ///
2031 /// Review found that collapsing all four `as_str` arms to `"fetch"` left
2032 /// the whole suite green: every test of these columns passed string
2033 /// literals, so nothing tied a variant to its label. A histogram whose
2034 /// buckets all say the same thing is worse than no histogram — it reports a
2035 /// single confident cause for four different failures.
2036 ///
2037 /// Asserted over `ALL` rather than a hand-written list, so adding a variant
2038 /// without a label fails here instead of silently sharing one.
2039 #[test]
2040 fn every_failure_kind_has_a_distinct_round_tripping_label() {
2041 let mut seen = std::collections::BTreeSet::new();
2042 for kind in FailureKind::ALL {
2043 let label = kind.as_str();
2044 assert!(
2045 seen.insert(label),
2046 "{label:?} is used by more than one FailureKind",
2047 );
2048 assert_eq!(
2049 FailureKind::parse(label),
2050 Some(kind),
2051 "{label:?} does not read back as the kind that wrote it",
2052 );
2053 }
2054 assert_eq!(seen.len(), FailureKind::ALL.len());
2055 // A label from a newer build is not attributed to a cause this one
2056 // knows — the `metrics::Backend::parse` contract.
2057 assert_eq!(FailureKind::parse("quota"), None);
2058 }
2059
2060 /// **Escalation reaches `settle_poll`.** `backoff_for` grows with the
2061 /// count and is tested alone; nothing asserted that the poll path passes
2062 /// the COUNT in. `backoff_for(1)` in its place left the whole suite green
2063 /// — a permanently dead feed retrying forever at the first-failure floor,
2064 /// which the comment on that line says must not happen.
2065 #[tokio::test]
2066 async fn backoff_escalates_with_consecutive_failures() -> anyhow::Result<()> {
2067 let pool = crate::store::init_url("sqlite::memory:").await?;
2068 let url = "https://dead.example/feed.xml";
2069 crate::store::upsert_feed(
2070 &pool,
2071 &crate::store::NewFeed {
2072 url: url.to_string(),
2073 ..Default::default()
2074 },
2075 )
2076 .await?;
2077 for _ in 0..5 {
2078 crate::store::bump_feed_errors(&pool, url, FailureKind::Fetch, "down").await?;
2079 }
2080 let before = chrono::Utc::now();
2081 settle_poll(
2082 &pool,
2083 url,
2084 &PollOutcome::Failed {
2085 backoff: Duration::from_secs(300),
2086 kind: FailureKind::Fetch,
2087 detail: "still down".to_string(),
2088 },
2089 Duration::from_secs(3600),
2090 )
2091 .await;
2092 let next: String = sqlx::query_scalar("SELECT next_poll FROM feeds WHERE url = ?1")
2093 .bind(url)
2094 .fetch_one(&pool)
2095 .await?;
2096 let next = chrono::DateTime::parse_from_rfc3339(&next)?.with_timezone(&chrono::Utc);
2097 let delay = (next - before).num_seconds();
2098 let expected = backoff_for(6).as_secs() as i64;
2099 assert!(
2100 (delay - expected).abs() <= 60,
2101 "sixth failure scheduled {delay}s out; escalation says {expected}s"
2102 );
2103 assert!(
2104 delay > backoff_for(1).as_secs() as i64 + 60,
2105 "the sixth failure landed on the first-failure floor"
2106 );
2107 Ok(())
2108 }
2109
2110 #[test]
2111 fn backoff_grows_and_is_capped() {
2112 assert_eq!(backoff_for(1), BACKOFF_BASE);
2113 assert!(backoff_for(2) > backoff_for(1));
2114 assert_eq!(backoff_for(100), BACKOFF_MAX);
2115 }
2116
2117 /// **Storable and pollable are ONE decision.**
2118 ///
2119 /// Review found the sequencing error this closes: making `at://` storable
2120 /// while nothing can poll it does not leave the feature dormant, it creates
2121 /// permanent failures that the cause histogram then publishes as
2122 /// unreachable publishers — the exact conflation it exists to end.
2123 #[test]
2124 fn an_at_uri_is_not_storable_while_standard_site_is_off() {
2125 let uri = "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab";
2126 assert!(
2127 !is_storable_feed_url(uri, false),
2128 "stored a feed nothing can poll"
2129 );
2130 assert!(is_storable_feed_url(uri, true));
2131 assert!(is_storable_feed_url("https://example.com/feed.xml", false));
2132 assert!(is_storable_feed_url("https://example.com/feed.xml", true));
2133 }
2134
2135 /// **A non-canonical scheme spelling is recognised and refused.** URL
2136 /// schemes are case-insensitive, so `At://` names the same thing as
2137 /// `at://` — but `feeds.url` is UNIQUE, so accepting both is two rows for
2138 /// one publication. Recognised (not passed through to the generic checks
2139 /// as if it were an ordinary URL), then refused for the spelling.
2140 #[test]
2141 fn a_non_canonical_at_uri_spelling_is_recognised_and_refused() {
2142 for odd in [
2143 "At://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
2144 "AT://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
2145 ] {
2146 assert!(!is_storable_feed_url(odd, true), "stored {odd:?}");
2147 // Fails CLOSED: it is an at-URI this reader will not store, not an
2148 // unparseable string that the `Err(_) => Public` arm waves through.
2149 assert!(
2150 classify_feed_privacy(odd).is_private(),
2151 "{odd:?} was declared publishable"
2152 );
2153 }
2154 }
2155
2156 /// **The handle form is not storable — the DID form is the identity.**
2157 ///
2158 /// `feeds.url` is UNIQUE; a handle and its DID would be two rows for one
2159 /// publication, and a handle can change hands. Every spelling is refused,
2160 /// canonical or not; resolving one to a DID is the input path's job.
2161 #[test]
2162 fn a_handle_form_publication_uri_is_not_storable() {
2163 for authority in [
2164 "alice.example.com",
2165 "EXAMPLE.COM",
2166 "169.254.169.254",
2167 "pds.internal",
2168 "printer.local",
2169 "host:8080",
2170 "-.-",
2171 "a b.c",
2172 ] {
2173 let uri = format!("at://{authority}/site.standard.publication/3lab");
2174 assert!(
2175 !is_storable_feed_url(&uri, true),
2176 "accepted authority {authority:?}"
2177 );
2178 }
2179 }
2180
2181 /// A control character or space in the at-URI is refused: `scheduler.rs`
2182 /// logs `%feed.url` with Display, and `feeds.url` is UNIQUE.
2183 #[test]
2184 fn an_at_uri_with_control_characters_is_not_storable() {
2185 for bad in [
2186 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab\n",
2187 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab ",
2188 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3l\tab",
2189 ] {
2190 assert!(!is_storable_feed_url(bad, true), "accepted {bad:?}");
2191 }
2192 }
2193
2194 /// **The `at://` exemption is a REGRESSION unless it is narrow.**
2195 ///
2196 /// Fan-out review found the arm I added was a bare prefix match, so *any*
2197 /// attacker-chosen string starting `at://` was declared safe to publish —
2198 /// skipping the userinfo check, the known-provider table, the private-path
2199 /// markers, the secret-query keys and the entropy heuristics. Measured
2200 /// against `main`, these three went from `Private` to `Public`.
2201 ///
2202 /// That matters because `rename_subscription` caches the URL AND rewrites
2203 /// the user's PUBLIC PDS record, with `classify_feed_privacy` as its only
2204 /// gate.
2205 #[test]
2206 fn a_credential_bearing_at_uri_is_still_private() {
2207 for hostile in [
2208 "at://user:pass@private.example.com/feed/private/TOKEN?apikey=deadbeefdeadbeef",
2209 "at://patreon.com/rss/12345?auth=deadbeefdeadbeefdeadbeef",
2210 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab?apikey=sekrit",
2211 ] {
2212 assert!(
2213 matches!(classify_feed_privacy(hostile), FeedPrivacy::Private(_)),
2214 "declared public: {hostile}"
2215 );
2216 }
2217 }
2218
2219 /// An rkey is `[A-Za-z0-9._:~-]` per atproto. Without that, a query string
2220 /// or path fragment smuggled into the rkey satisfies the three-segment
2221 /// check — which is what the exemption above keys off.
2222 #[test]
2223 fn an_rkey_outside_the_atproto_charset_is_not_storable() {
2224 for bad in [
2225 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab?apikey=sekrit",
2226 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab#frag",
2227 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab%2Fevil",
2228 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/caf\u{e9}",
2229 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab\u{202e}x",
2230 ] {
2231 assert!(!is_storable_feed_url(bad, true), "accepted rkey in {bad:?}");
2232 }
2233 // The legitimate charset still passes.
2234 for good in [
2235 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h",
2236 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/a.b_c~d-e",
2237 ] {
2238 assert!(is_storable_feed_url(good, true), "refused {good:?}");
2239 }
2240 }
2241
2242 /// **A real rkey is a TID, and a TID looks exactly like a secret.**
2243 ///
2244 /// Without an explicit `at://` arm, `classify_feed_privacy` runs the generic
2245 /// "high-entropy token in path" heuristic over the rkey. Measured: a
2246 /// realistic 16-char rkey on a handle-form at-URI is classified PRIVATE and
2247 /// the subscription REFUSED. The DID form escaped only because it fails to
2248 /// parse as a `Url` at all — so the bug was invisible from that side.
2249 ///
2250 /// The first version of this test used the rkey `3lab`, which is too short
2251 /// to trip the heuristic, so it passed with and without the fix.
2252 #[test]
2253 fn a_realistic_at_uri_rkey_is_not_mistaken_for_a_secret() {
2254 for uri in [
2255 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h",
2256 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/aB3xK9pQ7mZ2vN8w",
2257 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h",
2258 ] {
2259 assert_eq!(
2260 classify_feed_privacy(uri),
2261 FeedPrivacy::Public,
2262 "a publication rkey was mistaken for a credential: {uri}"
2263 );
2264 }
2265 }
2266
2267 /// **The DID form is the one that matters, and the one `Url::parse` cannot
2268 /// read.**
2269 ///
2270 /// `Url::parse("at://did:plc:…/…")` fails with *invalid port number* — the
2271 /// colons in the DID are taken as a port separator. So the obvious
2272 /// implementation, adding `"at"` to the `matches!` on `u.scheme()`, silently
2273 /// rejects every DID-based at-URI while appearing to work: the handle form
2274 /// (`at://alice.example.com/…`) parses fine and would pass such a test.
2275 ///
2276 /// All 19 at-URI rows in production are the DID form. A test written with a
2277 /// handle would have passed against an implementation that cannot store a
2278 /// single one of them.
2279 #[test]
2280 fn a_did_form_publication_uri_is_storable() {
2281 assert!(is_storable_feed_url(
2282 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
2283 true
2284 ));
2285 }
2286
2287 /// **An allowlist entry, not a loosening.** `at://` is accepted for exactly
2288 /// one foreign collection. Any other collection is somebody else's lexicon
2289 /// arriving through a path (`resolve_subscriptions`, OPML import) that takes
2290 /// records from outside with no add-path to reject them.
2291 #[test]
2292 fn an_at_uri_for_another_collection_is_not_storable() {
2293 assert!(!is_storable_feed_url(
2294 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/community.lexicon.rss.subscription/3lab",
2295 true
2296 ));
2297 assert!(!is_storable_feed_url(
2298 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/app.bsky.feed.post/3lab",
2299 true
2300 ));
2301 }
2302
2303 /// The malformed shapes, each of which a naive `split('/')` would accept.
2304 #[test]
2305 fn a_malformed_at_uri_is_not_storable() {
2306 for bad in [
2307 "at://",
2308 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc",
2309 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication",
2310 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/",
2311 "at:///site.standard.publication/3lab",
2312 "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab/extra",
2313 "at://not-a-did-or-handle/site.standard.publication/3lab",
2314 "at://did:plc:TOOSHORT/site.standard.publication/3lab",
2315 ] {
2316 assert!(!is_storable_feed_url(bad, true), "accepted {bad:?}");
2317 }
2318 }
2319
2320 /// **The reason the function exists, unchanged.** Mutating the new branch to
2321 /// accept any scheme makes this fail while the at-URI tests keep passing —
2322 /// that asymmetry is what says the change was an allowlist entry.
2323 #[test]
2324 fn the_refused_schemes_are_still_refused() {
2325 for bad in [
2326 "javascript:alert(1)",
2327 "file:///etc/passwd",
2328 "data:text/html,<script>",
2329 "ftp://example.com/feed.xml",
2330 "at:did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
2331 ] {
2332 assert!(!is_storable_feed_url(bad, true), "accepted {bad:?}");
2333 }
2334 // A hostless http(s) URL is a PARSE error, not a parsed URL with no
2335 // host — `https:///feed.xml` even parses as host `feed.xml`. What
2336 // refuses these is the `Err` arm, so that is what this pins.
2337 for hostless in ["http://", "https://?q=1", "http:///"] {
2338 assert!(
2339 !is_storable_feed_url(hostless, true),
2340 "a hostless URL {hostless:?} was storable"
2341 );
2342 }
2343 assert!(is_storable_feed_url("https://example.com/feed.xml", true));
2344 assert!(is_storable_feed_url("http://example.com/feed.xml", true));
2345 }
2346}