use std::time::Duration;
use anyhow::{Context, Result};
use chrono::{DateTime, SecondsFormat, Utc};
use feed_rs::model::{Entry as RawEntry, Feed as RawFeed, Link as RawLink, Text};
use reqwest::header::{ETAG, IF_MODIFIED_SINCE, IF_NONE_MATCH, LAST_MODIFIED};
use reqwest::{Client, StatusCode};
use sqlx::SqlitePool;
use url::Url;
use crate::store::{self, Feed, NewEntry, NewFeed};
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum FeedPrivacy {
Public,
Private(String),
}
impl FeedPrivacy {
pub fn is_private(&self) -> bool {
matches!(self, FeedPrivacy::Private(_))
}
}
const SECRET_QUERY_KEYS: &[&str] = &[
"token", "key", "auth", "secret", "k", "sig", "hash", "access", "apikey", "api_key", "uuid",
"id", "u", "s", "p", "private", "password", "pw",
];
const PRIVATE_PATH_MARKERS: &[&str] = &[
"/private/",
"/feed/private/",
"/rss/private/",
"/private-feed/",
"/members/",
"/member/",
"/subscriber/",
];
struct KnownProvider {
host_contains: &'static str,
marker: Option<&'static str>,
reason: &'static str,
}
const KNOWN_PROVIDERS: &[KnownProvider] = &[
KnownProvider {
host_contains: "substack.com",
marker: Some("/feed/private/"),
reason: "Substack private feed",
},
KnownProvider {
host_contains: "patreon.com",
marker: Some("auth="),
reason: "Patreon member feed",
},
KnownProvider {
host_contains: "ghost.io",
marker: Some("uuid="),
reason: "Ghost members feed",
},
KnownProvider {
host_contains: "buttondown.email",
marker: Some("token"),
reason: "Buttondown premium feed",
},
KnownProvider {
host_contains: "buttondown.com",
marker: Some("token"),
reason: "Buttondown premium feed",
},
KnownProvider {
host_contains: "beehiiv.com",
marker: Some("token"),
reason: "Beehiiv premium feed",
},
KnownProvider {
host_contains: "memberful.com",
marker: None,
reason: "Memberful members feed",
},
KnownProvider {
host_contains: "pico.link",
marker: None,
reason: "Pico member feed",
},
KnownProvider {
host_contains: "steadyhq.com",
marker: None,
reason: "Steady member feed",
},
KnownProvider {
host_contains: "supercast.com",
marker: None,
reason: "Supercast private podcast",
},
KnownProvider {
host_contains: "supercast.tech",
marker: None,
reason: "Supercast private podcast",
},
KnownProvider {
host_contains: "supportingcast.fm",
marker: None,
reason: "Supporting Cast private podcast",
},
KnownProvider {
host_contains: "redcircle.com",
marker: Some("private"),
reason: "RedCircle private podcast",
},
KnownProvider {
host_contains: "megaphone.fm",
marker: Some("token"),
reason: "Megaphone private podcast",
},
KnownProvider {
host_contains: "acast.com",
marker: Some("token"),
reason: "Acast+ private podcast",
},
KnownProvider {
host_contains: "omny.fm",
marker: Some("token"),
reason: "Omny private podcast",
},
KnownProvider {
host_contains: "podcasts.apple.com",
marker: Some("token"),
reason: "Apple subscriber podcast",
},
KnownProvider {
host_contains: "spotify.com",
marker: Some("token"),
reason: "Spotify subscriber podcast",
},
];
pub fn is_storable_feed_url(url: &str, allow_at_uri: bool) -> bool {
if let Some(rest) = crate::atproto::strip_at_prefix(url) {
if !url.starts_with(crate::atproto::AT_URI_PREFIX) {
return false;
}
return allow_at_uri && is_storable_publication_uri(rest);
}
match Url::parse(url) {
Ok(u) => {
matches!(u.scheme(), "http" | "https")
}
Err(_) => false,
}
}
fn is_storable_publication_uri(rest: &str) -> bool {
let mut parts = rest.split('/');
let (Some(authority), Some(collection), Some(rkey)) =
(parts.next(), parts.next(), parts.next())
else {
return false;
};
parts.next().is_none()
&& collection == crate::lexicon::nsid::STANDARD_PUBLICATION
&& crate::atproto::is_valid_rkey(rkey)
&& is_storable_at_authority(authority)
}
fn is_storable_at_authority(authority: &str) -> bool {
crate::oauth::identity::is_atproto_did(authority)
}
pub fn classify_feed_privacy(url: &str) -> FeedPrivacy {
if let Some(rest) = crate::atproto::strip_at_prefix(url) {
if !url.starts_with(crate::atproto::AT_URI_PREFIX) {
return FeedPrivacy::Private("non-canonical at:// scheme spelling".to_string());
}
if is_storable_publication_uri(rest) {
return FeedPrivacy::Public;
}
return FeedPrivacy::Private("not a well-formed at:// publication URI".to_string());
}
let parsed = match Url::parse(url) {
Ok(u) => u,
Err(_) => return FeedPrivacy::Public,
};
if !parsed.username().is_empty() || parsed.password().is_some() {
return FeedPrivacy::Private("credentials in URL userinfo".to_string());
}
let path_lower = parsed.path().to_ascii_lowercase();
let query_lower = parsed.query().unwrap_or("").to_ascii_lowercase();
let host_lower = parsed.host_str().unwrap_or("").to_ascii_lowercase();
if is_public_youtube_feed(&host_lower, &path_lower, &parsed) {
return FeedPrivacy::Public;
}
for kp in KNOWN_PROVIDERS {
if host_lower.contains(kp.host_contains) {
let marker_ok = match kp.marker {
None => true,
Some(m) => {
let m = m.to_ascii_lowercase();
path_lower.contains(&m) || query_lower.contains(&m)
}
};
if marker_ok {
return FeedPrivacy::Private(kp.reason.to_string());
}
}
}
for marker in PRIVATE_PATH_MARKERS {
if path_lower.contains(marker) {
return FeedPrivacy::Private(format!("private feed path `{marker}`"));
}
}
for (k, v) in parsed.query_pairs() {
let key = k.as_ref().to_ascii_lowercase();
if SECRET_QUERY_KEYS.iter().any(|sk| *sk == key) && value_is_opaque(v.as_ref()) {
return FeedPrivacy::Private(format!("credential query parameter `{key}`"));
}
}
for seg in parsed.path().split('/').filter(|s| !s.is_empty()) {
if segment_hides_secret(seg) {
return FeedPrivacy::Private("high-entropy token in path".to_string());
}
}
for (_, v) in parsed.query_pairs() {
if looks_like_embedded_secret(v.as_ref()) {
return FeedPrivacy::Private("high-entropy token in query".to_string());
}
}
FeedPrivacy::Public
}
fn value_is_opaque(v: &str) -> bool {
if v.is_empty() {
return false;
}
if is_uuid(v) {
return true;
}
v.len() >= 8
}
fn looks_like_embedded_secret(s: &str) -> bool {
if is_uuid(s) {
return true;
}
if s.len() >= 16 && s.chars().all(|c| c.is_ascii_hexdigit()) {
return true;
}
if s.len() < 16 {
return false;
}
let separators = s
.bytes()
.filter(|b| *b == b'-' || *b == b'.' || *b == b' ')
.count();
if separators >= 3 {
return false;
}
let token_chars = s
.chars()
.filter(|c| c.is_ascii_alphanumeric() || matches!(c, '_' | '-' | '='))
.count();
if token_chars < s.chars().count() {
return false;
}
let has_alpha = s.chars().any(|c| c.is_ascii_alphabetic());
let has_digit = s.chars().any(|c| c.is_ascii_digit());
if !(has_alpha && has_digit) {
return false;
}
let mut seen = std::collections::HashSet::new();
for c in s.chars() {
seen.insert(c.to_ascii_lowercase());
}
seen.len() >= 10
}
const FEED_EXTENSIONS: &[&str] = &["rss", "xml", "atom", "json", "rss20"];
fn segment_hides_secret(seg: &str) -> bool {
if looks_like_embedded_secret(seg) {
return true;
}
if let Some((stem, ext)) = seg.rsplit_once('.') {
if FEED_EXTENSIONS.contains(&ext.to_ascii_lowercase().as_str())
&& looks_like_embedded_secret(stem)
{
return true;
}
}
if contains_uuid(seg) {
return true;
}
if seg.contains(['.', '_', '-']) {
for part in seg.split(['.', '_', '-']).filter(|p| !p.is_empty()) {
if looks_like_embedded_secret(part) {
return true;
}
}
}
false
}
fn contains_uuid(s: &str) -> bool {
const UUID_LEN: usize = 36; let bytes = s.as_bytes();
if bytes.len() < UUID_LEN {
return false;
}
(0..=bytes.len() - UUID_LEN).any(|i| s.get(i..i + UUID_LEN).map(is_uuid).unwrap_or(false))
}
fn is_public_youtube_feed(host_lower: &str, path_lower: &str, parsed: &Url) -> bool {
let host_ok = host_lower == "youtube.com"
|| host_lower == "www.youtube.com"
|| host_lower.ends_with(".youtube.com");
if !host_ok || !path_lower.starts_with("/feeds/videos.xml") {
return false;
}
parsed.query_pairs().all(|(k, _)| {
let k = k.as_ref().to_ascii_lowercase();
k == "channel_id" || k == "playlist_id" || k == "user"
})
}
fn is_uuid(s: &str) -> bool {
let groups = [8usize, 4, 4, 4, 12];
let parts: Vec<&str> = s.split('-').collect();
if parts.len() != groups.len() {
return false;
}
parts
.iter()
.zip(groups.iter())
.all(|(p, &n)| p.len() == n && p.chars().all(|c| c.is_ascii_hexdigit()))
}
const FETCH_TIMEOUT: Duration = Duration::from_secs(30);
const READ_TIMEOUT: Duration = Duration::from_secs(15);
const BACKOFF_BASE: Duration = Duration::from_secs(300);
const BACKOFF_MAX: Duration = Duration::from_secs(24 * 3600);
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum PollOutcome {
Updated { new_entries: u64 },
NotModified,
Failed {
backoff: Duration,
kind: FailureKind,
detail: String,
},
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum FailureKind {
Fetch,
Status,
Body,
Parse,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum FeedKind {
Rss,
Publication,
Unsupported,
}
impl FeedKind {
pub const POLLABLE: &'static [FeedKind] = &[FeedKind::Rss, FeedKind::Publication];
pub const AGED: &'static [FeedKind] = &[FeedKind::Rss];
pub fn as_str(self) -> &'static str {
match self {
FeedKind::Rss => "rss",
FeedKind::Publication => "publication",
FeedKind::Unsupported => "unsupported",
}
}
pub fn parse(raw: &str) -> Option<Self> {
match raw {
"rss" => Some(FeedKind::Rss),
"publication" => Some(FeedKind::Publication),
"unsupported" => Some(FeedKind::Unsupported),
_ => None,
}
}
pub fn of(url: &str) -> Self {
match crate::atproto::strip_at_prefix(url) {
Some(rest) if url.starts_with("at://") && is_storable_publication_uri(rest) => {
FeedKind::Publication
}
Some(_) => FeedKind::Unsupported,
None => FeedKind::Rss,
}
}
}
pub async fn settle_poll(
pool: &sqlx::SqlitePool,
url: &str,
outcome: &PollOutcome,
cadence: Duration,
) {
let delay = match outcome {
PollOutcome::Updated { .. } | PollOutcome::NotModified => {
if let Err(err) = crate::store::reset_feed_errors(pool, url).await {
tracing::warn!(feed = %url, %err, "failed to reset feed error count");
}
cadence
}
PollOutcome::Failed {
backoff,
kind,
detail,
} => {
match crate::store::bump_feed_errors(pool, url, *kind, detail).await {
Ok(count) => backoff_for(count.max(1) as u32),
Err(err) => {
tracing::warn!(feed = %url, %err, "failed to bump feed error count; using floor backoff");
*backoff
}
}
}
};
if let Err(err) = crate::store::set_next_poll(pool, url, delay).await {
tracing::error!(feed = %url, %err, "failed to persist next_poll");
}
}
pub const MAX_FAILURE_DETAIL_CHARS: usize = 300;
pub fn failure_detail(err: impl std::fmt::Display) -> String {
let s = err.to_string();
if s.chars().count() <= MAX_FAILURE_DETAIL_CHARS {
return s;
}
s.chars().take(MAX_FAILURE_DETAIL_CHARS).collect()
}
impl FailureKind {
pub fn as_str(self) -> &'static str {
match self {
Self::Fetch => "fetch",
Self::Status => "status",
Self::Body => "body",
Self::Parse => "parse",
}
}
pub fn parse(raw: &str) -> Option<Self> {
match raw {
"fetch" => Some(Self::Fetch),
"status" => Some(Self::Status),
"body" => Some(Self::Body),
"parse" => Some(Self::Parse),
_ => None,
}
}
pub const ALL: [Self; 4] = [Self::Fetch, Self::Status, Self::Body, Self::Parse];
}
pub fn build_client() -> Result<Client> {
Client::builder()
.user_agent(crate::USER_AGENT)
.timeout(FETCH_TIMEOUT)
.read_timeout(READ_TIMEOUT)
.no_proxy()
.redirect(reqwest::redirect::Policy::none())
.build()
.context("failed to build feed HTTP client")
}
pub fn backoff_for(consecutive_errors: u32) -> Duration {
let n = consecutive_errors.max(1);
let factor = 1u64.checked_shl(n.saturating_sub(1)).unwrap_or(u64::MAX);
let secs = BACKOFF_BASE
.as_secs()
.saturating_mul(factor)
.min(BACKOFF_MAX.as_secs());
Duration::from_secs(secs)
}
pub async fn poll_feed_by_kind(
pool: &SqlitePool,
client: &Client,
config: &crate::config::Config,
feed: &Feed,
) -> Result<PollOutcome> {
match FeedKind::of(&feed.url) {
FeedKind::Rss => poll_feed(pool, client, feed, config.max_entries_per_feed).await,
FeedKind::Publication => poll_publication(pool, client, config, feed).await,
FeedKind::Unsupported => Ok(PollOutcome::Failed {
backoff: backoff_for(1),
kind: FailureKind::Parse,
detail: failure_detail(format!("{} is not a feed this reader can poll", feed.url)),
}),
}
}
fn publication_failure_kind(err: &anyhow::Error) -> FailureKind {
use crate::atproto::{AtProtoError, DidResolutionCause};
for cause in err.chain() {
if cause.is::<crate::standard_site::NotAPublication>() || cause.is::<serde_json::Error>() {
return FailureKind::Parse;
}
match cause.downcast_ref::<AtProtoError>() {
Some(AtProtoError::Xrpc { .. }) => return FailureKind::Status,
Some(AtProtoError::DidResolution { cause, .. }) => {
return match cause {
DidResolutionCause::Status => FailureKind::Status,
DidResolutionCause::UnsupportedMethod | DidResolutionCause::NoPdsEndpoint => {
FailureKind::Parse
}
DidResolutionCause::NotAPublicTarget => FailureKind::Fetch,
}
}
_ => {}
}
}
FailureKind::Fetch
}
async fn poll_publication(
pool: &SqlitePool,
client: &Client,
config: &crate::config::Config,
feed: &Feed,
) -> Result<PollOutcome> {
poll_publication_group(pool, client, config, std::slice::from_ref(feed))
.await
.pop()
.unwrap_or_else(|| Err(anyhow::anyhow!("no outcome for {}", feed.url)))
}
pub async fn poll_publication_group(
pool: &SqlitePool,
client: &Client,
config: &crate::config::Config,
feeds: &[Feed],
) -> Vec<Result<PollOutcome>> {
let failed = |kind: FailureKind, detail: String| -> Result<PollOutcome> {
Ok(PollOutcome::Failed {
backoff: backoff_for(1),
kind,
detail: failure_detail(detail),
})
};
let uris: Vec<Option<crate::standard_site::AtUri>> = feeds
.iter()
.map(|f| crate::standard_site::AtUri::parse(&f.url))
.collect();
let Some(did) = uris.iter().flatten().next().map(|u| u.authority.clone()) else {
return feeds
.iter()
.map(|f| {
failed(
FailureKind::Parse,
format!("{} is not a readable at:// URI", f.url),
)
})
.collect();
};
let readable: Vec<usize> = (0..feeds.len())
.filter(|&i| {
uris[i].as_ref().is_some_and(|u| {
u.authority == did && u.collection == crate::lexicon::nsid::STANDARD_PUBLICATION
})
})
.collect();
let rkeys: Vec<String> = readable
.iter()
.map(|&i| uris[i].as_ref().unwrap().rkey.clone())
.collect();
let fetched = tokio::time::timeout(
config.publication_read_deadline,
crate::standard_site::fetch_repo(client, &config.oauth.plc_directory, &did, &rkeys),
)
.await;
let mut reads: Vec<Option<anyhow::Result<crate::standard_site::PublicationRead>>> =
feeds.iter().map(|_| None).collect();
let repo_failure = match fetched {
Ok(Ok(per)) => {
for (slot, read) in readable.iter().zip(per) {
reads[*slot] = Some(read);
}
None
}
Ok(Err(err)) => Some((publication_failure_kind(&err), format!("{err:#}"))),
Err(_) => Some((
FailureKind::Fetch,
format!(
"the read did not finish within {:?}",
config.publication_read_deadline
),
)),
};
let (retention_days, retention_hard_days) = config.retention_for(FeedKind::Publication);
let mut out = Vec::with_capacity(feeds.len());
for (i, feed) in feeds.iter().enumerate() {
let outcome = match (reads[i].take(), &repo_failure) {
(Some(Ok(read)), _) => {
crate::standard_site::store_publication(
pool,
&feed.url,
read,
config.max_entries_per_feed,
retention_days,
retention_hard_days,
)
.await
}
(Some(Err(err)), _) => failed(publication_failure_kind(&err), format!("{err:#}")),
(None, Some((kind, detail))) if readable.contains(&i) => failed(*kind, detail.clone()),
(None, _) => failed(
FailureKind::Parse,
format!("{} is not a publication in {did}", feed.url),
),
};
out.push(outcome);
}
out
}
pub async fn poll_feed(
pool: &SqlitePool,
client: &Client,
feed: &Feed,
max_entries_per_feed: i64,
) -> Result<PollOutcome> {
let mut extra: Vec<(reqwest::header::HeaderName, reqwest::header::HeaderValue)> = Vec::new();
if let Some(etag) = feed.etag.as_deref() {
if let Ok(v) = reqwest::header::HeaderValue::from_str(etag) {
extra.push((IF_NONE_MATCH, v));
}
}
if let Some(lm) = feed.last_modified.as_deref() {
if let Ok(v) = reqwest::header::HeaderValue::from_str(lm) {
extra.push((IF_MODIFIED_SINCE, v));
}
}
let resp = match crate::net::guarded_get(client, &feed.url, &extra).await {
Ok(r) => r,
Err(e) => {
tracing::warn!(feed = %feed.url, error = %e, "feed fetch failed (or blocked by SSRF guard)");
return Ok(PollOutcome::Failed {
backoff: backoff_for(1),
kind: FailureKind::Fetch,
detail: failure_detail(format!("{e:#}")),
});
}
};
let status = resp.status();
if status == StatusCode::NOT_MODIFIED {
tracing::debug!(feed = %feed.url, "feed not modified (304)");
touch_polled(pool, &feed.url, None, None)
.await
.with_context(|| format!("touch_polled after 304 for {}", feed.url))?;
return Ok(PollOutcome::NotModified);
}
if !status.is_success() {
tracing::warn!(feed = %feed.url, %status, "feed returned non-success status");
return Ok(PollOutcome::Failed {
backoff: backoff_for(1),
kind: FailureKind::Status,
detail: failure_detail(status),
});
}
let new_etag = header_str(resp.headers().get(ETAG));
let new_last_modified = header_str(resp.headers().get(LAST_MODIFIED));
let body = match crate::net::read_capped(resp).await {
Ok(b) => b,
Err(e) => {
tracing::warn!(feed = %feed.url, error = %e, "feed body rejected (too large / read error)");
return Ok(PollOutcome::Failed {
backoff: backoff_for(1),
kind: FailureKind::Body,
detail: failure_detail(format!("{e:#}")),
});
}
};
let parsed = match parse_feed(&body[..]) {
Ok(f) => f,
Err(e) => {
tracing::warn!(feed = %feed.url, error = %e, "malformed feed; skipping");
return Ok(PollOutcome::Failed {
backoff: backoff_for(1),
kind: FailureKind::Parse,
detail: failure_detail(format!("{e:#}")),
});
}
};
let (title, site_url) = feed_metadata(&parsed);
let new_feed = NewFeed {
url: feed.url.clone(),
title,
site_url,
etag: new_etag,
last_modified: new_last_modified,
last_polled: Some(now_rfc3339()),
next_poll: None, };
let entries: Vec<NewEntry> = parsed.entries.iter().map(normalize_entry).collect();
let feed_id = store::upsert_feed(pool, &new_feed)
.await
.with_context(|| format!("upsert_feed for {}", feed.url))?;
let n = store::insert_entries(pool, feed_id, &entries, max_entries_per_feed)
.await
.with_context(|| format!("insert_entries for {}", feed.url))?;
tracing::info!(feed = %feed.url, entries = n, "feed polled");
Ok(PollOutcome::Updated { new_entries: n })
}
async fn touch_polled(
pool: &SqlitePool,
url: &str,
etag: Option<String>,
last_modified: Option<String>,
) -> Result<()> {
let nf = NewFeed {
url: url.to_string(),
etag,
last_modified,
last_polled: Some(now_rfc3339()),
..Default::default()
};
store::upsert_feed(pool, &nf).await?;
Ok(())
}
fn feed_metadata(parsed: &RawFeed) -> (Option<String>, Option<String>) {
let title = parsed
.title
.as_ref()
.map(|t| bound_text(text_plain(t), MAX_TITLE_BYTES));
let site_url = parsed
.links
.iter()
.find(|l| {
l.rel.as_deref() == Some("alternate")
|| (l.rel.is_none()
&& l.media_type.as_deref() != Some("application/rss+xml")
&& l.media_type.as_deref() != Some("application/atom+xml"))
})
.or_else(|| {
parsed
.links
.iter()
.find(|l| l.rel.as_deref() != Some("self"))
})
.or_else(|| parsed.links.first())
.map(|l| bound_text(l.href.clone(), MAX_URL_BYTES));
(title, site_url)
}
fn parse_feed(body: &[u8]) -> Result<RawFeed> {
let parsed = std::panic::catch_unwind(|| {
feed_rs::parser::Builder::new()
.id_generator(entry_id)
.build()
.parse(body)
});
match parsed {
Ok(result) => Ok(result?),
Err(panic) => {
let why = panic
.downcast_ref::<String>()
.map(String::as_str)
.or_else(|| panic.downcast_ref::<&str>().copied())
.unwrap_or("non-string panic payload");
anyhow::bail!("feed parser panicked: {why}")
}
}
}
fn entry_id(links: &[RawLink], title: &Option<Text>, _uri: Option<&str>) -> String {
match links.iter().find(|l| is_primary_link(l)) {
Some(link) => feed_rs::parser::generate_id_from_link_and_title(link, title),
None => String::new(),
}
}
fn is_primary_link(l: &RawLink) -> bool {
l.target.is_none()
}
fn normalize_entry(e: &RawEntry) -> NewEntry {
let url = entry_link(e);
let content_html = e
.content
.as_ref()
.and_then(|c| c.body.as_deref())
.or_else(|| e.summary.as_ref().map(|t| t.content.as_str()))
.map(|raw| sanitize_html_bounded(raw, MAX_CONTENT_HTML_BYTES));
let guid = if !e.id.trim().is_empty() {
e.id.trim().to_string()
} else if let Some(link) = raw_entry_link(e) {
link
} else {
stable_guid(e)
};
NewEntry {
guid: bound_guid(guid),
url: url.map(|u| bound_text(u, MAX_URL_BYTES)),
title: e
.title
.as_ref()
.map(|t| bound_text(text_plain(t), MAX_TITLE_BYTES)),
author: entry_author(e).map(|a| bound_text(a, MAX_AUTHOR_BYTES)),
published: entry_time(e),
content_html,
fetched_at: None, }
}
fn raw_entry_link(e: &RawEntry) -> Option<String> {
let mut links = e.links.iter().filter(|l| is_primary_link(l));
links
.clone()
.find(|l| l.rel.as_deref() == Some("alternate") || l.rel.is_none())
.or_else(|| links.next())
.map(|l| l.href.clone())
}
fn entry_link(e: &RawEntry) -> Option<String> {
raw_entry_link(e).and_then(|href| crate::net::safe_link(&href))
}
fn entry_author(e: &RawEntry) -> Option<String> {
e.authors.iter().find_map(|p| {
let name = p.name.as_deref()?.trim();
let name = [('(', ')'), ('<', '>'), ('[', ']')]
.iter()
.find_map(|&(open, close)| name.strip_prefix(open)?.strip_suffix(close))
.unwrap_or(name)
.trim();
(!name.is_empty()).then(|| name.to_string())
})
}
fn entry_time(e: &RawEntry) -> Option<String> {
let ceiling = Utc::now() + chrono::Duration::days(MAX_FUTURE_PUBLISHED_DAYS);
e.published
.filter(|d| *d <= ceiling)
.or_else(|| e.updated.filter(|d| *d <= ceiling))
.map(fmt_time)
}
pub(crate) const MAX_FUTURE_PUBLISHED_DAYS: i64 = 2;
fn text_plain(t: &Text) -> String {
t.content.trim().to_string()
}
pub(crate) fn sanitize_html(raw: &str) -> String {
ammonia::clean(raw)
}
pub(crate) fn plain_text_to_html(raw: &str) -> String {
let escaped = raw
.replace('&', "&")
.replace('<', "<")
.replace('>', ">");
escaped.replace('\n', "<br>")
}
pub(crate) const MAX_TITLE_BYTES: usize = 5_000;
pub(crate) const MAX_AUTHOR_BYTES: usize = 1_000;
pub(crate) const MAX_URL_BYTES: usize = 8_192;
pub(crate) const MAX_CONTENT_HTML_BYTES: usize = 2 * 1024 * 1024;
pub(crate) const MAX_GUID_BYTES: usize = 2_048;
fn floor_char_boundary(s: &str, at: usize) -> usize {
if at >= s.len() {
return s.len();
}
(0..=at).rev().find(|&i| s.is_char_boundary(i)).unwrap_or(0)
}
pub(crate) fn bound_text(mut s: String, max: usize) -> String {
let cut = floor_char_boundary(&s, max);
s.truncate(cut);
s
}
pub(crate) fn plain_text_to_html_bounded(raw: &str, max: usize) -> String {
let mut size = 0usize;
let mut cut = raw.len();
for (i, c) in raw.char_indices() {
let escaped = match c {
'&' => 5,
'<' | '>' | '\n' => 4,
c => c.len_utf8(),
};
if size + escaped > max {
cut = i;
break;
}
size += escaped;
}
plain_text_to_html(&raw[..cut])
}
pub(crate) fn sanitize_html_bounded(raw: &str, max: usize) -> String {
let clean = sanitize_html(raw);
if clean.len() <= max {
return clean;
}
let margin = (max / 64).max(64);
let first = floor_char_boundary(&clean, max.saturating_sub(margin));
let again = sanitize_html(&clean[..first]);
if again.len() <= max {
return again;
}
let (mut lo, mut hi) = (0usize, first);
let mut best = String::new();
let resolution = (max / 1024).max(1);
for _ in 0..SANITIZE_BOUND_ATTEMPTS {
if hi - lo <= resolution {
break;
}
let mut cut = floor_char_boundary(&clean, lo + (hi - lo) / 2);
if cut <= lo {
cut = ceil_char_boundary(&clean, lo + 1);
if cut >= hi {
break;
}
}
let out = sanitize_html(&clean[..cut]);
if out.len() <= max {
lo = cut;
best = out;
} else {
hi = cut;
}
}
best
}
fn ceil_char_boundary(s: &str, at: usize) -> usize {
(at..=s.len())
.find(|&i| s.is_char_boundary(i))
.unwrap_or(s.len())
}
const SANITIZE_BOUND_ATTEMPTS: usize = 14;
pub(crate) fn bound_guid(guid: String) -> String {
if guid.len() <= MAX_GUID_BYTES {
return guid;
}
use std::hash::{Hash, Hasher};
let mut h = dedup_hasher();
guid.hash(&mut h);
format!("featherreader:long-guid:{:016x}", h.finish())
}
fn dedup_hasher() -> siphasher::sip::SipHasher13 {
siphasher::sip::SipHasher13::new_with_keys(0, 0)
}
pub(crate) fn fmt_time(dt: DateTime<Utc>) -> String {
dt.to_rfc3339_opts(SecondsFormat::Secs, true)
}
fn now_rfc3339() -> String {
Utc::now().to_rfc3339_opts(SecondsFormat::Secs, true)
}
fn stable_guid(e: &RawEntry) -> String {
use std::hash::{Hash, Hasher};
let mut h = dedup_hasher();
e.title.as_ref().map(|t| t.content.as_str()).hash(&mut h);
e.links
.iter()
.find(|l| is_primary_link(l))
.map(|l| l.href.as_str())
.hash(&mut h);
e.summary.as_ref().map(|s| s.content.as_str()).hash(&mut h);
format!("featherreader:synthetic:{:016x}", h.finish())
}
fn header_str(v: Option<&reqwest::header::HeaderValue>) -> Option<String> {
v.and_then(|h| h.to_str().ok()).map(str::to_string)
}
pub fn discover_feed(site_html: &str, base: Option<&Url>) -> Option<Url> {
for tag in link_tags(site_html) {
let rel = attr(&tag, "rel").unwrap_or_default().to_ascii_lowercase();
let typ = attr(&tag, "type").unwrap_or_default().to_ascii_lowercase();
let is_feed_type = typ.contains("application/rss+xml")
|| typ.contains("application/atom+xml")
|| typ.contains("application/feed+json")
|| typ.contains("application/json");
let rel_ok = rel.split_whitespace().any(|r| r == "alternate") || rel.is_empty();
if is_feed_type && rel_ok {
if let Some(href) = attr(&tag, "href") {
let href = href.trim();
if href.is_empty() {
continue;
}
let resolved = match Url::parse(href) {
Ok(u) => Some(u),
Err(_) => base.and_then(|b| b.join(href).ok()),
};
match resolved {
Some(u) if matches!(u.scheme(), "http" | "https") => return Some(u),
_ => continue,
}
}
}
}
None
}
fn link_tags(html: &str) -> Vec<String> {
let mut out = Vec::new();
let bytes = html.as_bytes();
let lower = html.to_ascii_lowercase();
let mut search_from = 0usize;
while let Some(rel_idx) = lower[search_from..].find("<link") {
let start = search_from + rel_idx;
let after = bytes.get(start + 5).copied();
let boundary = matches!(after, Some(b) if b == b' ' || b == b'\t' || b == b'\n' || b == b'\r' || b == b'>' || b == b'/');
if !boundary {
search_from = start + 5;
continue;
}
if let Some(end_rel) = html[start..].find('>') {
let end = start + end_rel;
out.push(html[start..=end].to_string());
search_from = end + 1;
} else {
break;
}
}
out
}
fn attr(tag: &str, name: &str) -> Option<String> {
let lower = tag.to_ascii_lowercase();
let needle = format!("{name}=");
let mut from = 0usize;
while let Some(rel) = lower[from..].find(&needle) {
let name_start = from + rel;
let ok_prefix = name_start == 0
|| matches!(
tag.as_bytes().get(name_start - 1),
Some(b' ') | Some(b'\t') | Some(b'\n') | Some(b'\r') | Some(b'<')
);
let val_start = name_start + needle.len();
if !ok_prefix {
from = val_start;
continue;
}
let rest = &tag[val_start..];
let quote = rest.chars().next();
let value = match quote {
Some('"') => rest[1..].split('"').next(),
Some('\'') => rest[1..].split('\'').next(),
_ => rest
.split(|c: char| c.is_whitespace() || c == '>' || c == '/')
.next(),
};
return value.map(str::to_string);
}
None
}
#[cfg(test)]
mod tests {
use super::*;
const RSS_SAMPLE: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0">
<channel>
<title>Example RSS Feed</title>
<link>https://example.com/</link>
<description>An example feed for tests</description>
<item>
<title>First post</title>
<link>https://example.com/first</link>
<guid>https://example.com/first</guid>
<author>alice@example.com (Alice)</author>
<pubDate>Fri, 10 Jul 2026 08:00:00 GMT</pubDate>
<description><![CDATA[<p>Hello <b>world</b>.</p><script>alert('xss')</script><img src="x" onerror="alert(1)">]]></description>
</item>
<item>
<title>Second post</title>
<link>https://example.com/second</link>
<guid>guid-second</guid>
<pubDate>Sat, 11 Jul 2026 08:00:00 GMT</pubDate>
<description><![CDATA[<a href="javascript:alert(1)">click</a><a href="https://ok.example/">ok</a>]]></description>
</item>
</channel>
</rss>"#;
const ATOM_SAMPLE: &str = r#"<?xml version="1.0" encoding="utf-8"?>
<feed xmlns="http://www.w3.org/2005/Atom">
<title>Example Atom Feed</title>
<link rel="alternate" href="https://atom.example.com/"/>
<link rel="self" href="https://atom.example.com/feed.xml"/>
<id>urn:uuid:feed-1</id>
<updated>2026-07-11T08:00:00Z</updated>
<entry>
<title>Atom entry</title>
<id>urn:uuid:entry-1</id>
<link rel="alternate" href="https://atom.example.com/a"/>
<author><name>Bob</name></author>
<updated>2026-07-11T08:00:00Z</updated>
<content type="html"><![CDATA[<p>Safe <em>text</em>.</p><script>steal()</script><iframe src="evil"></iframe>]]></content>
</entry>
</feed>"#;
const ATOM_SELF_FIRST: &str = r#"<?xml version="1.0" encoding="utf-8"?>
<feed xmlns="http://www.w3.org/2005/Atom">
<title>Example Atom Feed</title>
<link rel="self" href="https://atom.example.com/feed.xml"/>
<link rel="alternate" href="https://atom.example.com/"/>
<id>urn:uuid:feed-1</id>
<updated>2026-07-11T08:00:00Z</updated>
<entry>
<title>Atom entry</title>
<id>urn:uuid:entry-1</id>
<link rel="alternate" href="https://atom.example.com/a"/>
<author><name>Bob</name></author>
<updated>2026-07-11T08:00:00Z</updated>
<content type="html"><![CDATA[<p>Safe <em>text</em>.</p><script>steal()</script><iframe src="evil"></iframe>]]></content>
</entry>
</feed>"#;
fn rss_with_fields(title: &str, link: &str, author: &str, body: &str, guid: &str) -> String {
format!(
r#"<?xml version="1.0"?><rss version="2.0" xmlns:dc="http://purl.org/dc/elements/1.1/"><channel>
<title>{title}</title><link>https://example.com/{link}</link>
<item><title>{title}</title><link>https://example.com/{link}</link><guid>{guid}</guid>
<dc:creator>{author}</dc:creator><description><![CDATA[{body}]]></description></item>
</channel></rss>"#
)
}
#[test]
fn bound_text_cuts_on_a_character_boundary() {
let cut = bound_text("é".repeat(100), 51);
assert!(cut.len() <= 51, "not bounded: {} bytes", cut.len());
assert_eq!(cut, "é".repeat(25), "cut too short or mid-character");
assert_eq!(
bound_text("short".into(), 51),
"short",
"a short value changed"
);
}
#[test]
fn rendered_content_fits_even_when_rendering_grows_it() {
let out = plain_text_to_html_bounded(&"&".repeat(1_000), 100);
assert!(
out.len() <= 100,
"escaped output not bounded: {} bytes",
out.len()
);
assert!(!out.is_empty(), "bounded to nothing");
assert_eq!(
out.matches("&").count() * 5,
out.len(),
"cut mid-entity: {out}"
);
}
#[test]
fn an_rss_items_text_fields_are_bounded() {
let big = "x".repeat(100_000);
let xml = rss_with_fields(&big, &big, &big, "body", "id-1");
let parsed = parse_feed(xml.as_bytes()).unwrap();
let e = normalize_entry(&parsed.entries[0]);
assert!(e.title.as_ref().unwrap().len() <= MAX_TITLE_BYTES, "title");
assert!(e.url.as_ref().unwrap().len() <= MAX_URL_BYTES, "url");
assert!(
e.author.as_ref().unwrap().len() <= MAX_AUTHOR_BYTES,
"author"
);
let (title, site) = feed_metadata(&parsed);
assert!(title.unwrap().len() <= MAX_TITLE_BYTES, "feed title");
assert!(site.unwrap().len() <= MAX_URL_BYTES, "feed site url");
}
#[test]
fn an_rss_body_is_bounded_and_still_well_formed() {
let body = format!("<p>{}<b>tail</b></p>", "a".repeat(MAX_CONTENT_HTML_BYTES));
let xml = rss_with_fields("t", "l", "a", &body, "id-2");
let parsed = parse_feed(xml.as_bytes()).unwrap();
let html = normalize_entry(&parsed.entries[0]).content_html.unwrap();
assert!(
html.len() <= MAX_CONTENT_HTML_BYTES,
"body not bounded: {}",
html.len()
);
assert_eq!(
sanitize_html(&html),
html,
"the stored body is not well-formed sanitized HTML"
);
}
#[test]
fn content_the_sanitizer_strips_does_not_count_against_the_bound() {
const MAX: usize = 64 * 1024;
let body = format!(
r#"<p><img src="data:image/png;base64,{}"></p><p>the article</p>"#,
"A".repeat(MAX + 16 * 1024)
);
let html = sanitize_html_bounded(&body, MAX);
assert!(
html.contains("the article"),
"the article was cut away: {} bytes kept",
html.len()
);
assert!(html.len() <= MAX);
}
#[test]
fn content_cut_beside_stripped_bytes_still_keeps_what_fits() {
const MAX: usize = 64 * 1024;
let body = format!("<p>{}</p><!--{}", "a".repeat(MAX + 8), "x".repeat(16 * MAX));
let html = sanitize_html_bounded(&body, MAX);
assert!(html.len() <= MAX);
assert!(
html.len() > MAX - MAX / 16,
"kept far less than fits: {}",
html.len()
);
}
#[test]
fn a_stripped_prefix_does_not_leave_an_empty_body() {
const MAX: usize = 64 * 1024;
let body = format!(
r#"<p><img src="data:image/png;base64,{}"></p><p>{}</p>"#,
"A".repeat(3 * MAX),
"&".repeat(MAX)
);
let html = sanitize_html_bounded(&body, MAX);
assert!(html.len() <= MAX);
assert!(
html.len() > MAX - MAX / 16,
"nothing like what fits was kept: {} bytes",
html.len()
);
}
#[test]
fn a_stripped_multibyte_prefix_does_not_leave_an_empty_body() {
const MAX: usize = 64 * 1024;
let body = format!(
"<!--{}--><p>{}</p>",
"漢".repeat(MAX),
"&".repeat(MAX * 3 / 10)
);
let html = sanitize_html_bounded(&body, MAX);
assert!(html.len() <= MAX);
assert!(
html.len() > MAX - MAX / 16,
"kept {} of ~{MAX} that fits",
html.len()
);
}
#[test]
fn nested_markup_is_cut_not_emptied() {
const MAX: usize = 64 * 1024;
let html = sanitize_html_bounded(&"<span>".repeat(16_384), MAX);
assert!(html.len() <= MAX);
assert!(
html.len() > MAX / 3,
"kept {} of ~{MAX} that fits",
html.len()
);
let mixed = format!("{}{}", "t".repeat(MAX * 3 / 4), "<span>".repeat(MAX / 8));
let html = sanitize_html_bounded(&mixed, MAX);
assert!(html.len() <= MAX);
assert!(
html.len() > MAX - MAX / 16,
"kept {} of ~{MAX} that fits",
html.len()
);
}
#[test]
fn the_plain_text_bound_is_exact() {
const MAX: usize = 64 * 1024;
let raw = format!("{}{}", "漢".repeat(MAX / 4), "&".repeat(MAX / 10));
let out = plain_text_to_html_bounded(&raw, MAX);
assert!(out.len() <= MAX);
assert!(out.len() > MAX - 5, "kept {} of {MAX}", out.len());
assert!(
raw.starts_with(&out.replace("&", "&")),
"not a prefix of the input"
);
}
#[test]
fn an_overlong_rss_guid_becomes_a_stable_short_one() {
let long = "g".repeat(10_000);
let xml = rss_with_fields("t", "l", "a", "b", &long);
let parsed = parse_feed(xml.as_bytes()).unwrap();
let a = normalize_entry(&parsed.entries[0]).guid;
let b = normalize_entry(&parsed.entries[0]).guid;
assert!(a.len() <= MAX_GUID_BYTES, "guid not bounded: {}", a.len());
assert_eq!(
a, b,
"the stand-in is not stable, so the entry would duplicate"
);
let other = rss_with_fields("t", "l", "a", "b", &format!("{long}h"));
let other = parse_feed(other.as_bytes()).unwrap();
assert_ne!(
a,
normalize_entry(&other.entries[0]).guid,
"two ids collapsed into one"
);
}
#[test]
fn an_ordinary_long_article_is_untouched() {
let body = format!("<p>{}</p>", "word ".repeat(18_000));
let xml = rss_with_fields(
"A normal title",
"post",
"Author",
&body,
"https://example.com/post",
);
let parsed = parse_feed(xml.as_bytes()).unwrap();
let e = normalize_entry(&parsed.entries[0]);
assert_eq!(e.content_html.unwrap(), sanitize_html(&body));
assert_eq!(e.title.as_deref(), Some("A normal title"));
assert_eq!(e.guid, "https://example.com/post");
}
#[test]
fn a_future_dated_rss_item_is_stored_undated_rather_than_dated_in_2999() {
let future = r#"<?xml version="1.0"?>
<rss version="2.0"><channel><title>Clock</title><link>https://clock.example/</link>
<item><title>From the future</title><link>https://clock.example/1</link>
<guid>https://clock.example/1</guid>
<pubDate>Sat, 01 Jan 2999 00:00:00 GMT</pubDate></item>
</channel></rss>"#;
let parsed = parse_feed(future.as_bytes()).expect("should parse");
assert!(
parsed.entries[0].published.is_some(),
"the fixture's pubDate did not parse, so this test proves nothing",
);
let e = normalize_entry(&parsed.entries[0]);
assert_eq!(
e.published, None,
"a year-2999 date was stored, which makes the row unsweepable, \
un-evictable and permanently first in the reading list",
);
let past = future.replace("01 Jan 2999", "01 Jan 2020");
let parsed = parse_feed(past.as_bytes()).expect("should parse");
let e = normalize_entry(&parsed.entries[0]);
assert!(
e.published
.as_deref()
.is_some_and(|p| p.starts_with("2020")),
"an ordinary past date was discarded: {:?}",
e.published,
);
let soon = (Utc::now() + chrono::Duration::hours(14)).to_rfc2822();
let near = future.replace("Sat, 01 Jan 2999 00:00:00 GMT", &soon);
let parsed = parse_feed(near.as_bytes()).expect("should parse");
assert!(
parsed.entries[0].published.is_some(),
"the mislabelled-date fixture did not parse",
);
let e = normalize_entry(&parsed.entries[0]);
assert!(
e.published.is_some(),
"a date 14 hours ahead — the widest real UTC offset — was refused, \
so a timezone-mislabelled article loses its date entirely",
);
let far = (Utc::now() + chrono::Duration::days(30)).to_rfc2822();
let month = future.replace("Sat, 01 Jan 2999 00:00:00 GMT", &far);
let parsed = parse_feed(month.as_bytes()).expect("should parse");
let e = normalize_entry(&parsed.entries[0]);
assert_eq!(
e.published, None,
"a date a month ahead was kept, so the row leads the list for a month",
);
let both = r#"<?xml version="1.0"?>
<feed xmlns="http://www.w3.org/2005/Atom"><title>Clock</title>
<entry><title>Mixed</title><id>https://clock.example/2</id>
<link href="https://clock.example/2"/>
<published>2999-01-01T00:00:00Z</published>
<updated>2020-06-01T00:00:00Z</updated></entry></feed>"#;
let parsed = parse_feed(both.as_bytes()).expect("should parse");
assert!(
parsed.entries[0].published.is_some() && parsed.entries[0].updated.is_some(),
"the fixture needs BOTH dates parsed for this case to mean anything",
);
let e = normalize_entry(&parsed.entries[0]);
assert!(
e.published
.as_deref()
.is_some_and(|p| p.starts_with("2020")),
"a credible `updated` was discarded along with a bogus `published`, \
leaving the entry undated: {:?}",
e.published,
);
}
#[test]
fn rss_parses_and_sanitizes() {
let parsed = parse_feed(RSS_SAMPLE.as_bytes()).expect("RSS should parse");
assert_eq!(
parsed.title.as_ref().map(text_plain).as_deref(),
Some("Example RSS Feed")
);
assert_eq!(parsed.entries.len(), 2);
let (title, site) = feed_metadata(&parsed);
assert_eq!(title.as_deref(), Some("Example RSS Feed"));
assert_eq!(site.as_deref(), Some("https://example.com/"));
let e0 = normalize_entry(&parsed.entries[0]);
assert_eq!(e0.guid, "https://example.com/first");
assert_eq!(e0.title.as_deref(), Some("First post"));
assert_eq!(e0.url.as_deref(), Some("https://example.com/first"));
assert!(e0.published.is_some());
let html0 = e0.content_html.expect("content present");
assert!(html0.contains("Hello"));
assert!(html0.contains("<b>world</b>") || html0.contains("<b>"));
assert!(!html0.to_ascii_lowercase().contains("<script"));
assert!(!html0.to_ascii_lowercase().contains("onerror"));
assert!(!html0.to_ascii_lowercase().contains("alert"));
let e1 = normalize_entry(&parsed.entries[1]);
assert_eq!(e1.guid, "guid-second");
let html1 = e1.content_html.expect("content present");
assert!(!html1.to_ascii_lowercase().contains("javascript:"));
assert!(html1.contains("https://ok.example/"));
}
#[test]
fn atom_prefers_alternate_over_a_self_link_listed_first() {
let parsed = parse_feed(ATOM_SELF_FIRST.as_bytes()).expect("Atom should parse");
let (title, site) = feed_metadata(&parsed);
assert_eq!(title.as_deref(), Some("Example Atom Feed"));
assert_eq!(site.as_deref(), Some("https://atom.example.com/"));
}
#[test]
fn atom_parses_and_sanitizes() {
let parsed = parse_feed(ATOM_SAMPLE.as_bytes()).expect("Atom should parse");
let (title, site) = feed_metadata(&parsed);
assert_eq!(title.as_deref(), Some("Example Atom Feed"));
assert_eq!(site.as_deref(), Some("https://atom.example.com/"));
assert_eq!(parsed.entries.len(), 1);
let e = normalize_entry(&parsed.entries[0]);
assert_eq!(e.guid, "urn:uuid:entry-1");
assert_eq!(e.title.as_deref(), Some("Atom entry"));
assert_eq!(e.author.as_deref(), Some("Bob"));
assert_eq!(e.url.as_deref(), Some("https://atom.example.com/a"));
let html = e.content_html.expect("content present");
assert!(html.contains("Safe"));
assert!(!html.to_ascii_lowercase().contains("<script"));
assert!(!html.to_ascii_lowercase().contains("<iframe"));
}
#[test]
fn an_rkey_must_obey_atprotos_length_and_dot_rules() {
let uri = |rkey: &str| {
format!("at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/{rkey}")
};
assert!(
!is_storable_feed_url(&uri("."), true),
"`.` is a reserved rkey"
);
assert!(
!is_storable_feed_url(&uri(".."), true),
"`..` is a reserved rkey"
);
assert!(
!is_storable_feed_url(&uri(&"a".repeat(513)), true),
"an rkey over 512 bytes was accepted"
);
assert!(
is_storable_feed_url(&uri(&"a".repeat(512)), true),
"an rkey of exactly 512 bytes is valid"
);
assert!(is_storable_feed_url(&uri("3lab2c4d5e6f7g8h"), true));
}
#[test]
fn each_known_provider_is_caught_by_its_own_row() {
for (url, reason) in [
(
"https://author.substack.com/feed/private/x",
"Substack private feed",
),
(
"https://www.patreon.com/rss/creator?auth=ab",
"Patreon member feed",
),
("https://blog.ghost.io/rss/?uuid=x", "Ghost members feed"),
(
"https://buttondown.email/me/rss?token=x",
"Buttondown premium feed",
),
(
"https://buttondown.com/me/rss?token=x",
"Buttondown premium feed",
),
(
"https://rss.beehiiv.com/feeds/x.xml?token=x",
"Beehiiv premium feed",
),
(
"https://example.memberful.com/feed",
"Memberful members feed",
),
("https://example.pico.link/feed", "Pico member feed"),
("https://steadyhq.com/rss/example", "Steady member feed"),
(
"https://example.supercast.com/feed",
"Supercast private podcast",
),
(
"https://example.supercast.tech/feed",
"Supercast private podcast",
),
(
"https://example.supportingcast.fm/feed",
"Supporting Cast private podcast",
),
(
"https://feeds.redcircle.com/x?private=1",
"RedCircle private podcast",
),
(
"https://feeds.megaphone.fm/x?token=x",
"Megaphone private podcast",
),
(
"https://feeds.acast.com/public/shows/x?token=x",
"Acast+ private podcast",
),
(
"https://omny.fm/shows/x/playlists/podcast.rss?token=x",
"Omny private podcast",
),
(
"https://podcasts.apple.com/feed/x?token=x",
"Apple subscriber podcast",
),
(
"https://anchor.spotify.com/s/x/podcast/rss?token=x",
"Spotify subscriber podcast",
),
] {
match classify_feed_privacy(url) {
FeedPrivacy::Private(r) => {
assert_eq!(r, reason, "{url} was refused by another rule")
}
FeedPrivacy::Public => panic!("{url} was not refused at all"),
}
}
}
#[test]
fn plain_text_is_escaped_rather_than_swallowed() {
assert_eq!(
plain_text_to_html("Vec<String> is a type"),
"Vec<String> is a type"
);
assert_eq!(plain_text_to_html("if x<y then z"), "if x<y then z");
assert_eq!(plain_text_to_html("a & b"), "a & b");
assert_eq!(plain_text_to_html("one\ntwo"), "one<br>two");
let hostile = plain_text_to_html("<script>alert(1)</script>");
assert!(!hostile.contains("<script"), "{hostile}");
}
#[test]
fn discover_finds_rss_link() {
let html = r#"<!doctype html><html><head>
<title>Blog</title>
<link rel="stylesheet" href="/style.css">
<link rel="alternate" type="application/rss+xml" title="RSS" href="/feed.xml">
</head><body>hi</body></html>"#;
let base = Url::parse("https://blog.example.com/").unwrap();
let found = discover_feed(html, Some(&base)).expect("should discover feed");
assert_eq!(found.as_str(), "https://blog.example.com/feed.xml");
}
#[test]
fn discover_finds_atom_absolute_link() {
let html = r#"<head><link rel="alternate" type="application/atom+xml" href="https://x.example/atom"></head>"#;
let found = discover_feed(html, None).expect("should discover absolute feed");
assert_eq!(found.as_str(), "https://x.example/atom");
}
#[test]
fn discover_skips_a_non_http_alternate() {
let at_link = r#"<link rel="alternate" type="application/rss+xml" href="at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h">"#;
assert!(
discover_feed(&format!("<head>{at_link}</head>"), None).is_none(),
"an at:// alternate was handed back as a feed URL"
);
let ftp_link =
r#"<link rel="alternate" type="application/atom+xml" href="ftp://x.example/atom">"#;
assert!(discover_feed(&format!("<head>{ftp_link}</head>"), None).is_none());
let real =
r#"<link rel="alternate" type="application/atom+xml" href="https://x.example/atom">"#;
let found = discover_feed(&format!("<head>{at_link}{real}</head>"), None)
.expect("the http(s) link after a skipped one must still be found");
assert_eq!(found.as_str(), "https://x.example/atom");
}
#[test]
fn discover_returns_none_without_feed_link() {
let html =
r#"<head><link rel="stylesheet" href="/s.css"><link rel="icon" href="/f.ico"></head>"#;
assert!(discover_feed(html, None).is_none());
}
#[test]
fn synthetic_guid_is_stable_and_dedups() {
let xml = r#"<?xml version="1.0"?><rss version="2.0"><channel>
<title>t</title>
<item><title>only a title</title><description>body</description></item>
</channel></rss>"#;
let mut parsed = parse_feed(xml.as_bytes()).expect("parse");
parsed.entries[0].id.clear();
parsed.entries[0].links.clear();
let g1 = normalize_entry(&parsed.entries[0]).guid;
let g2 = normalize_entry(&parsed.entries[0]).guid;
assert_eq!(g1, g2);
assert!(g1.starts_with("featherreader:synthetic:"));
}
#[test]
fn a_generated_entry_id_is_the_one_feed_rs_2_4_produced() {
let xml = r#"<?xml version="1.0"?><rss version="2.0"><channel><title>t</title><item><title>No guid here</title><link>https://n.example/post</link></item></channel></rss>"#;
let e = normalize_entry(&parse_feed(xml.as_bytes()).expect("parse").entries[0]);
assert_eq!(e.guid, "5813b43a0512aaef2750311bf4d978a");
assert_eq!(e.url.as_deref(), Some("https://n.example/post"));
}
#[test]
fn a_comments_link_listed_first_is_neither_the_permalink_nor_the_id() {
let xml = r#"<?xml version="1.0"?>
<rss version="2.0" xmlns:wfw="http://wellformedweb.org/CommentAPI/"><channel><title>c</title><link>https://c.example/</link>
<item><title>Comments listed first, no guid</title><comments>https://c.example/1#comments</comments><link>https://c.example/1</link><wfw:commentRss>https://c.example/1/feed</wfw:commentRss></item>
<item><title>Comments first, guid present</title><comments>https://c.example/6#comments</comments><link>https://c.example/6</link><guid>c6</guid></item>
</channel></rss>"#;
let parsed = parse_feed(xml.as_bytes()).expect("parse");
let e0 = normalize_entry(&parsed.entries[0]);
assert_eq!(
e0.guid, "cd0017f2746ee934cf45ca0125100796",
"the generated id moved, so this item would be stored twice"
);
assert_eq!(e0.url.as_deref(), Some("https://c.example/1"));
let e1 = normalize_entry(&parsed.entries[1]);
assert_eq!(e1.guid, "c6");
assert_eq!(e1.url.as_deref(), Some("https://c.example/6"));
}
#[test]
fn an_atom_comment_feed_link_is_neither_the_permalink_nor_the_id() {
let xml = r#"<?xml version="1.0" encoding="utf-8"?>
<feed xmlns="http://www.w3.org/2005/Atom" xmlns:wfw="http://wellformedweb.org/CommentAPI/"><title>a</title>
<entry><title>commentRss before link, no id</title><wfw:commentRss>https://a.example/1/feed</wfw:commentRss><link href="https://a.example/1"/><updated>2020-06-01T00:00:00Z</updated></entry>
</feed>"#;
let e = normalize_entry(&parse_feed(xml.as_bytes()).expect("parse").entries[0]);
assert_eq!(e.guid, "5148a3d11836efc42f82a7b25a38383d");
assert_eq!(e.url.as_deref(), Some("https://a.example/1"));
}
#[test]
fn an_item_without_guid_or_permalink_dedups_across_polls() {
let xml = r#"<?xml version="1.0"?><rss version="2.0"><channel><title>t</title>
<item><title>only a title</title><description>body</description></item>
<item><title>Only a comments link</title><comments>https://c.example/2#comments</comments></item>
</channel></rss>"#;
let first = parse_feed(xml.as_bytes()).expect("parse");
let second = parse_feed(xml.as_bytes()).expect("parse");
for i in 0..2 {
let a = normalize_entry(&first.entries[i]);
let b = normalize_entry(&second.entries[i]);
assert_eq!(
a.guid, b.guid,
"entry {i}'s guid changed between two parses"
);
assert!(a.guid.starts_with("featherreader:synthetic:"), "{}", a.guid);
assert_eq!(a.url, None, "entry {i}");
}
}
#[test]
fn the_author_is_the_name_not_the_element_or_the_address() {
let xml = r#"<?xml version="1.0"?>
<rss version="2.0" xmlns:dc="http://purl.org/dc/elements/1.1/"><channel><title>p</title>
<item><title>a1</title><guid>a1</guid><author>alice@example.com (Alice Example)</author></item>
<item><title>a2</title><guid>a2</guid><author>bob@example.com</author></item>
<item><title>a3</title><guid>a3</guid><author>Carol</author></item>
<item><title>a4</title><guid>a4</guid><dc:creator>Dave <dave@example.com></dc:creator></item>
<item><title>a5</title><guid>a5</guid><author>erin@example.com ()</author></item>
</channel></rss>"#;
let parsed = parse_feed(xml.as_bytes()).expect("parse");
let authors: Vec<Option<String>> = parsed
.entries
.iter()
.map(|e| normalize_entry(e).author)
.collect();
assert_eq!(
authors,
vec![
Some("Alice Example".to_string()),
None,
Some("Carol".to_string()),
Some("Dave".to_string()),
None,
]
);
let atom = r#"<?xml version="1.0" encoding="utf-8"?>
<feed xmlns="http://www.w3.org/2005/Atom"><title>a</title>
<entry><id>x1</id><title>t</title><author><name></name></author></entry>
<entry><id>x2</id><title>t</title><author><name>Bob</name></author></entry>
</feed>"#;
let parsed = parse_feed(atom.as_bytes()).expect("parse");
assert_eq!(normalize_entry(&parsed.entries[0]).author, None);
assert_eq!(
normalize_entry(&parsed.entries[1]).author.as_deref(),
Some("Bob")
);
}
#[test]
fn the_byline_is_the_first_author_with_a_name() {
let xml = r#"<?xml version="1.0"?>
<rss version="2.0" xmlns:dc="http://purl.org/dc/elements/1.1/"><channel><title>p</title>
<item><title>b1</title><guid>b1</guid><author>bob@example.com</author><dc:creator>Bob Smith</dc:creator></item>
<item><title>b2</title><guid>b2</guid><author>erin@example.com ()</author><dc:creator>Erin</dc:creator></item>
<item><title>b3</title><guid>b3</guid><author>only@example.com</author></item>
</channel></rss>"#;
let parsed = parse_feed(xml.as_bytes()).expect("parse");
let authors: Vec<Option<String>> = parsed
.entries
.iter()
.map(|e| normalize_entry(e).author)
.collect();
assert_eq!(
authors,
vec![
Some("Bob Smith".to_string()),
Some("Erin".to_string()),
None
]
);
}
const PANICKING_AUTHORS: [&str; 3] = [
"jose@example.com(José)",
"«zoe@example.com» Zoë Long Name Here",
"Zoë Long Name «zoe@example.com»",
];
fn documents_with_author(v: &str) -> Vec<(&'static str, String)> {
vec![
(
"rss author",
format!(
r#"<?xml version="1.0"?><rss version="2.0"><channel><title>t</title><item><title>x</title><guid>g</guid><author>{v}</author></item></channel></rss>"#
),
),
(
"rss dc:creator",
format!(
r#"<?xml version="1.0"?><rss version="2.0" xmlns:dc="http://purl.org/dc/elements/1.1/"><channel><title>t</title><item><title>x</title><guid>g</guid><dc:creator>{v}</dc:creator></item></channel></rss>"#
),
),
(
"rss managingEditor",
format!(
r#"<?xml version="1.0"?><rss version="2.0"><channel><title>t</title><managingEditor>{v}</managingEditor><item><title>x</title><guid>g</guid></item></channel></rss>"#
),
),
(
"rss webMaster",
format!(
r#"<?xml version="1.0"?><rss version="2.0"><channel><title>t</title><webMaster>{v}</webMaster><item><title>x</title><guid>g</guid></item></channel></rss>"#
),
),
(
"rss1 dc:creator",
format!(
r#"<?xml version="1.0"?><rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" xmlns="http://purl.org/rss/1.0/" xmlns:dc="http://purl.org/dc/elements/1.1/"><channel><title>t</title></channel><item><title>x</title><link>https://e.example/1</link><dc:creator>{v}</dc:creator></item></rdf:RDF>"#
),
),
(
"json author",
format!(
r#"{{"version":"https://jsonfeed.org/version/1.1","title":"t","items":[{{"id":"1","content_text":"x","authors":[{{"name":"{v}"}}]}}]}}"#
),
),
]
}
#[test]
fn a_feed_rs_panic_is_returned_as_an_error() {
for v in PANICKING_AUTHORS {
for (slot, doc) in documents_with_author(v) {
let err = match parse_feed(doc.as_bytes()) {
Ok(_) => {
panic!("{slot} {v:?}: parsed; the fixture no longer reaches the panic")
}
Err(e) => format!("{e:#}"),
};
assert!(err.contains("panicked"), "{slot} {v:?}: {err}");
}
}
for (slot, doc) in documents_with_author("jose@example.com (Jose)") {
assert!(parse_feed(doc.as_bytes()).is_ok(), "{slot}");
}
}
#[tokio::test]
async fn a_poll_of_a_feed_that_panics_the_parser_is_a_parse_failure() {
let doc = &documents_with_author(PANICKING_AUTHORS[0])[0].1;
let base = crate::net::tests::serve_body(doc.as_bytes().to_vec()).await;
let port: u16 = base
.trim_end_matches('/')
.rsplit(':')
.next()
.unwrap()
.parse()
.unwrap();
crate::net::test_host_override(
"panicking-author.test",
std::net::SocketAddr::from(([127, 0, 0, 1], port)),
);
let url = format!("http://panicking-author.test:{port}/feed.xml");
let pool = crate::store::init_url("sqlite::memory:").await.unwrap();
crate::store::upsert_feed(
&pool,
&crate::store::NewFeed {
url: url.clone(),
..Default::default()
},
)
.await
.unwrap();
let feed = crate::store::get_feed_by_url(&pool, &url)
.await
.unwrap()
.unwrap();
let client = build_client().unwrap();
let outcome = poll_feed(&pool, &client, &feed, 0)
.await
.expect("a parser panic surfaced as a store error");
match outcome {
PollOutcome::Failed {
kind: FailureKind::Parse,
detail,
..
} => assert!(detail.contains("panicked"), "{detail}"),
other => panic!("expected a parse failure, got {other:?}"),
}
}
#[test]
fn a_synthetic_guid_is_the_value_already_stored() {
let xml = r#"<?xml version="1.0"?><rss version="2.0"><channel><title>t</title>
<item><title>only a title</title><description>body</description></item>
<item><description>no title either</description></item>
<item><title>Tïtle wíth ünïcode</title></item>
</channel></rss>"#;
let parsed = parse_feed(xml.as_bytes()).expect("parse");
let guids: Vec<String> = parsed
.entries
.iter()
.map(|e| normalize_entry(e).guid)
.collect();
assert_eq!(
guids,
vec![
"featherreader:synthetic:964034cf9a24b551".to_string(),
"featherreader:synthetic:baa7c19198c76b30".into(),
"featherreader:synthetic:34b72cc7fa0a59ac".into(),
]
);
}
#[test]
fn a_long_guid_is_the_value_already_stored() {
let a = bound_guid("g".repeat(MAX_GUID_BYTES + 1));
let b = bound_guid(format!(
"https://long.example/{}",
"ü".repeat(MAX_GUID_BYTES)
));
assert_eq!(
(a.as_str(), b.as_str()),
(
"featherreader:long-guid:966404db20ef8f05",
"featherreader:long-guid:a1d631602455ace4",
)
);
}
#[test]
fn entry_link_scheme_allowlist_neutralizes_javascript() {
let xml = r#"<?xml version="1.0"?><rss version="2.0"><channel>
<title>t</title>
<item>
<title>evil</title>
<link>javascript:alert(document.domain)</link>
<guid>evil-1</guid>
</item>
</channel></rss>"#;
let parsed = parse_feed(xml.as_bytes()).expect("parse");
let e = normalize_entry(&parsed.entries[0]);
assert_eq!(e.url, None);
assert_eq!(e.guid, "evil-1");
let xml2 = r#"<?xml version="1.0"?><rss version="2.0"><channel>
<title>t</title>
<item><title>d</title><link>data:text/html,<script>1</script></link><guid>d1</guid></item>
</channel></rss>"#;
let parsed2 = parse_feed(xml2.as_bytes()).expect("parse");
let e2 = normalize_entry(&parsed2.entries[0]);
assert_eq!(e2.url, None);
let xml3 = r#"<?xml version="1.0"?><rss version="2.0"><channel>
<title>t</title>
<item><title>ok</title><link>https://ok.example/post</link><guid>ok1</guid></item>
</channel></rss>"#;
let parsed3 = parse_feed(xml3.as_bytes()).expect("parse");
let e3 = normalize_entry(&parsed3.entries[0]);
assert_eq!(e3.url.as_deref(), Some("https://ok.example/post"));
}
#[test]
fn classify_privacy_flags_secret_urls_across_providers() {
assert!(
classify_feed_privacy("https://author.substack.com/feed/private/deadbeefcafe1234")
.is_private()
);
assert!(classify_feed_privacy(
"https://www.patreon.com/rss/author?auth=Zm9vYmFyc2VjcmV0dG9rZW4"
)
.is_private());
assert!(classify_feed_privacy(
"https://blog.ghost.io/rss/?uuid=1f2e3d4c-5b6a-7089-90ab-cdef01234567"
)
.is_private());
assert!(classify_feed_privacy(
"https://feeds.supportingcast.fm/show/abcdef0123456789abcdef01"
)
.is_private());
assert!(classify_feed_privacy("https://feeds.supercast.com/12345/rss").is_private());
assert!(
classify_feed_privacy("https://example.com/feed?token=Zm9vYmFyc2VjcmV0").is_private()
);
assert!(
classify_feed_privacy("https://example.com/feed?key=Zm9vYmFyc2VjcmV0").is_private()
);
assert!(
classify_feed_privacy("https://example.com/feed?secret=Zm9vYmFyc2VjcmV0").is_private()
);
assert!(classify_feed_privacy("https://user:pass@example.com/feed").is_private());
assert!(classify_feed_privacy("https://blog.example.com/private/rss").is_private());
assert!(classify_feed_privacy("https://news.example.com/members/feed.xml").is_private());
assert!(
classify_feed_privacy("https://feeds.example.com/aB3xK9zQ7mP2rT5wL8nD4vF6")
.is_private()
);
assert!(classify_feed_privacy(
"https://feeds.example.com/1f2e3d4c-5b6a-7089-90ab-cdef01234567"
)
.is_private());
}
#[test]
fn classify_privacy_catches_tokened_filenames_on_unknown_hosts() {
assert!(
classify_feed_privacy("https://cdn.example/feeds/aB3xK9pQ-7mZ2vN8w-Qr5tYuW.rss")
.is_private(),
"a token stem visible only after stripping the extension was not caught"
);
assert!(classify_feed_privacy(
"https://cdn.somepod.io/f/a1b2c3d4e5f60718293a4b5c6d7e8f90.xml"
)
.is_private());
assert!(classify_feed_privacy(
"https://dcs.megaphone.example/network/a1b2c3d4e5f60718293a4b5c6d7e8f90.rss"
)
.is_private());
assert!(classify_feed_privacy(
"https://brandnew.example/feed/1f2e3d4c-5b6a-7089-90ab-cdef01234567.xml"
)
.is_private());
assert!(classify_feed_privacy(
"https://x.example/feed-1f2e3d4c-5b6a-7089-90ab-cdef01234567"
)
.is_private());
assert!(classify_feed_privacy(
"https://x.example/1f2e3d4c-5b6a-7089-90ab-cdef01234567.rss"
)
.is_private());
assert!(classify_feed_privacy("https://x.example/feed/9f8e7d6c5b4a3928.xml").is_private());
assert!(
classify_feed_privacy("https://cdn.pod.io/f/YWJjZGVmZ2hpamtsbW5vcHFyc3R1dnc=")
.is_private()
);
}
#[test]
fn classify_privacy_allows_public_youtube_feeds() {
assert_eq!(
classify_feed_privacy(
"https://www.youtube.com/feeds/videos.xml?channel_id=UC-lHJZR3Gqxm24_Vd_AJ5Yw"
),
FeedPrivacy::Public
);
assert_eq!(
classify_feed_privacy(
"https://www.youtube.com/feeds/videos.xml?playlist_id=PLFgquLnL59alCl_2TQvOiD5Vgm1hCaGSI"
),
FeedPrivacy::Public
);
assert_eq!(
classify_feed_privacy(
"https://youtube.com/feeds/videos.xml?channel_id=UC-lHJZR3Gqxm24_Vd_AJ5Yw"
),
FeedPrivacy::Public
);
assert!(classify_feed_privacy(
"https://www.youtube.com/feeds/videos.xml?token=Zm9vYmFyc2VjcmV0dG9rZW4"
)
.is_private());
}
#[test]
fn classify_privacy_leaves_normal_public_feeds_public() {
assert_eq!(
classify_feed_privacy("https://example.com/feed.xml"),
FeedPrivacy::Public
);
assert_eq!(
classify_feed_privacy("https://blog.example.com/rss"),
FeedPrivacy::Public
);
assert_eq!(
classify_feed_privacy("https://blog.example.com/rss.xml"),
FeedPrivacy::Public
);
assert_eq!(
classify_feed_privacy("https://author.substack.com/feed"),
FeedPrivacy::Public
);
assert_eq!(
classify_feed_privacy("https://wordpress.example.com/feed/"),
FeedPrivacy::Public
);
assert_eq!(
classify_feed_privacy("https://example.org/atom.xml"),
FeedPrivacy::Public
);
assert_eq!(
classify_feed_privacy("https://example.com/2026/07/my-first-long-blog-post-title/feed"),
FeedPrivacy::Public
);
assert_eq!(
classify_feed_privacy("https://example.com/feed?keyword=rust"),
FeedPrivacy::Public
);
assert_eq!(
classify_feed_privacy("https://example.com/feed?p=2"),
FeedPrivacy::Public
);
assert_eq!(
classify_feed_privacy("https://example.com/feed?token="),
FeedPrivacy::Public
);
assert_eq!(
classify_feed_privacy("https://example.com/my-first-long-blog-post.xml"),
FeedPrivacy::Public
);
assert_eq!(
classify_feed_privacy("https://example.com/episodes/ab12cd.xml"),
FeedPrivacy::Public
);
assert_eq!(
classify_feed_privacy("https://example.com/category/tech-news/feed.xml"),
FeedPrivacy::Public
);
assert_eq!(classify_feed_privacy("not a url"), FeedPrivacy::Public);
}
#[test]
fn a_failure_detail_is_bounded_at_construction() {
let huge = "x".repeat(10_000);
let outcome = PollOutcome::Failed {
backoff: BACKOFF_BASE,
kind: FailureKind::Fetch,
detail: failure_detail(&huge),
};
let PollOutcome::Failed { detail, .. } = &outcome else {
panic!("wrong variant");
};
assert!(
detail.chars().count() <= MAX_FAILURE_DETAIL_CHARS,
"detail was {} chars",
detail.chars().count(),
);
assert!(format!("{outcome:?}").len() < 1_000);
}
#[test]
fn every_failure_kind_has_a_distinct_round_tripping_label() {
let mut seen = std::collections::BTreeSet::new();
for kind in FailureKind::ALL {
let label = kind.as_str();
assert!(
seen.insert(label),
"{label:?} is used by more than one FailureKind",
);
assert_eq!(
FailureKind::parse(label),
Some(kind),
"{label:?} does not read back as the kind that wrote it",
);
}
assert_eq!(seen.len(), FailureKind::ALL.len());
assert_eq!(FailureKind::parse("quota"), None);
}
#[tokio::test]
async fn backoff_escalates_with_consecutive_failures() -> anyhow::Result<()> {
let pool = crate::store::init_url("sqlite::memory:").await?;
let url = "https://dead.example/feed.xml";
crate::store::upsert_feed(
&pool,
&crate::store::NewFeed {
url: url.to_string(),
..Default::default()
},
)
.await?;
for _ in 0..5 {
crate::store::bump_feed_errors(&pool, url, FailureKind::Fetch, "down").await?;
}
let before = chrono::Utc::now();
settle_poll(
&pool,
url,
&PollOutcome::Failed {
backoff: Duration::from_secs(300),
kind: FailureKind::Fetch,
detail: "still down".to_string(),
},
Duration::from_secs(3600),
)
.await;
let next: String = sqlx::query_scalar("SELECT next_poll FROM feeds WHERE url = ?1")
.bind(url)
.fetch_one(&pool)
.await?;
let next = chrono::DateTime::parse_from_rfc3339(&next)?.with_timezone(&chrono::Utc);
let delay = (next - before).num_seconds();
let expected = backoff_for(6).as_secs() as i64;
assert!(
(delay - expected).abs() <= 60,
"sixth failure scheduled {delay}s out; escalation says {expected}s"
);
assert!(
delay > backoff_for(1).as_secs() as i64 + 60,
"the sixth failure landed on the first-failure floor"
);
Ok(())
}
#[test]
fn backoff_grows_and_is_capped() {
assert_eq!(backoff_for(1), BACKOFF_BASE);
assert!(backoff_for(2) > backoff_for(1));
assert_eq!(backoff_for(100), BACKOFF_MAX);
}
#[test]
fn an_at_uri_is_not_storable_while_standard_site_is_off() {
let uri = "at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab";
assert!(
!is_storable_feed_url(uri, false),
"stored a feed nothing can poll"
);
assert!(is_storable_feed_url(uri, true));
assert!(is_storable_feed_url("https://example.com/feed.xml", false));
assert!(is_storable_feed_url("https://example.com/feed.xml", true));
}
#[test]
fn only_a_storable_publication_uri_is_a_publication() {
let did = "did:plc:ohutz6x5acjmpuulp3x7wxxc";
let pubn = crate::lexicon::nsid::STANDARD_PUBLICATION;
assert_eq!(
FeedKind::of(&format!("at://{did}/{pubn}/3lab")),
FeedKind::Publication
);
for unsupported in [
format!("at://{did}/app.bsky.feed.post/3lab"),
format!("At://{did}/{pubn}/3lab"),
format!("at://alice.example.com/{pubn}/3lab"),
format!("at://did:plc:short/{pubn}/3lab"),
] {
assert_eq!(
FeedKind::of(&unsupported),
FeedKind::Unsupported,
"{unsupported}"
);
}
assert_eq!(FeedKind::of("https://example.com/feed.xml"), FeedKind::Rss);
assert!(!FeedKind::POLLABLE.contains(&FeedKind::Unsupported));
assert_eq!(FeedKind::parse("unsupported"), Some(FeedKind::Unsupported));
}
#[test]
fn a_non_canonical_at_uri_spelling_is_recognised_and_refused() {
for odd in [
"At://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
"AT://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
] {
assert!(!is_storable_feed_url(odd, true), "stored {odd:?}");
assert!(
classify_feed_privacy(odd).is_private(),
"{odd:?} was declared publishable"
);
}
}
#[test]
fn a_handle_form_publication_uri_is_not_storable() {
for authority in [
"alice.example.com",
"EXAMPLE.COM",
"169.254.169.254",
"pds.internal",
"printer.local",
"host:8080",
"-.-",
"a b.c",
] {
let uri = format!("at://{authority}/site.standard.publication/3lab");
assert!(
!is_storable_feed_url(&uri, true),
"accepted authority {authority:?}"
);
}
}
#[test]
fn an_at_uri_with_control_characters_is_not_storable() {
for bad in [
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab\n",
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab ",
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3l\tab",
] {
assert!(!is_storable_feed_url(bad, true), "accepted {bad:?}");
}
}
#[test]
fn a_credential_bearing_at_uri_is_still_private() {
for hostile in [
"at://user:pass@private.example.com/feed/private/TOKEN?apikey=deadbeefdeadbeef",
"at://patreon.com/rss/12345?auth=deadbeefdeadbeefdeadbeef",
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab?apikey=sekrit",
] {
assert!(
matches!(classify_feed_privacy(hostile), FeedPrivacy::Private(_)),
"declared public: {hostile}"
);
}
}
#[test]
fn an_rkey_outside_the_atproto_charset_is_not_storable() {
for bad in [
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab?apikey=sekrit",
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab#frag",
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab%2Fevil",
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/caf\u{e9}",
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab\u{202e}x",
] {
assert!(!is_storable_feed_url(bad, true), "accepted rkey in {bad:?}");
}
for good in [
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h",
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/a.b_c~d-e",
] {
assert!(is_storable_feed_url(good, true), "refused {good:?}");
}
}
#[test]
fn a_realistic_at_uri_rkey_is_not_mistaken_for_a_secret() {
for uri in [
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h",
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/aB3xK9pQ7mZ2vN8w",
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab2c4d5e6f7g8h",
] {
assert_eq!(
classify_feed_privacy(uri),
FeedPrivacy::Public,
"a publication rkey was mistaken for a credential: {uri}"
);
}
}
#[test]
fn a_did_form_publication_uri_is_storable() {
assert!(is_storable_feed_url(
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
true
));
}
#[test]
fn an_at_uri_for_another_collection_is_not_storable() {
assert!(!is_storable_feed_url(
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/community.lexicon.rss.subscription/3lab",
true
));
assert!(!is_storable_feed_url(
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/app.bsky.feed.post/3lab",
true
));
}
#[test]
fn a_malformed_at_uri_is_not_storable() {
for bad in [
"at://",
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc",
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication",
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/",
"at:///site.standard.publication/3lab",
"at://did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab/extra",
"at://not-a-did-or-handle/site.standard.publication/3lab",
"at://did:plc:TOOSHORT/site.standard.publication/3lab",
] {
assert!(!is_storable_feed_url(bad, true), "accepted {bad:?}");
}
}
#[test]
fn the_refused_schemes_are_still_refused() {
for bad in [
"javascript:alert(1)",
"file:///etc/passwd",
"data:text/html,<script>",
"ftp://example.com/feed.xml",
"at:did:plc:ohutz6x5acjmpuulp3x7wxxc/site.standard.publication/3lab",
] {
assert!(!is_storable_feed_url(bad, true), "accepted {bad:?}");
}
for hostless in ["http://", "https://?q=1", "http:///"] {
assert!(
!is_storable_feed_url(hostless, true),
"a hostless URL {hostless:?} was storable"
);
}
assert!(is_storable_feed_url("https://example.com/feed.xml", true));
assert!(is_storable_feed_url("http://example.com/feed.xml", true));
}
}