Skip to main content

mfp_core/catalog/
feed.rs

1//! Retrieval and parsing of the RSS feed, which is the sole authority on which episodes
2//! exist.
3//!
4//! Every episode corresponds to an `<item>` carrying an `<enclosure>` of type
5//! `audio/mpeg`, and the enclosure URL is the episode's identity. A retrieval failure and
6//! a parse failure are distinguishable, and neither yields a partial list.
7
8use quick_xml::XmlVersion;
9use quick_xml::escape::resolve_predefined_entity;
10use quick_xml::events::Event;
11use quick_xml::reader::Reader;
12
13use crate::error::{Error, Result};
14use crate::model::Episode;
15
16/// Fails rather than returning a partial list when the feed is unreachable, returns a
17/// non-success status, or cannot be parsed as RSS.
18pub async fn fetch(client: &reqwest::Client) -> Result<Vec<Episode>> {
19    let response = super::request(client, super::FEED_URL)
20        .send()
21        .await
22        .map_err(|error| unreachable(&error.to_string()))?;
23
24    let status = response.status();
25    if !status.is_success() {
26        return Err(unreachable(&format!("It returned HTTP {status}")));
27    }
28
29    let body = response
30        .text()
31        .await
32        .map_err(|error| unreachable(&error.to_string()))?;
33
34    parse(&body)
35}
36
37fn unreachable(reason: &str) -> Error {
38    Error::CatalogUnavailable(format!(
39        "The feed at {} is unreachable: {reason}",
40        super::FEED_URL
41    ))
42}
43
44fn malformed(reason: &str) -> Error {
45    Error::Internal(format!(
46        "The feed at {} is not a valid RSS document: {reason}",
47        super::FEED_URL
48    ))
49}
50
51/// Parses an RSS document into episodes, keeping only items with an `audio/mpeg`
52/// enclosure and preserving feed order.
53pub fn parse(xml: &str) -> Result<Vec<Episode>> {
54    // text is not trimmed by the reader: an entity arrives as its own event, so trimming
55    // each run would silently eat the spaces around it
56    let mut reader = Reader::from_str(xml);
57
58    let mut episodes = Vec::new();
59    let mut saw_rss = false;
60    let mut item: Option<PartialItem> = None;
61    let mut text = String::new();
62    let mut path: Vec<String> = Vec::new();
63
64    loop {
65        let event = reader
66            .read_event()
67            .map_err(|error| malformed(&error.to_string()))?;
68
69        match event {
70            Event::Eof => break,
71            Event::Start(start) => {
72                let name = local_name(start.name().as_ref()).to_owned();
73                if name == "rss" || name == "channel" {
74                    saw_rss = true;
75                }
76                if name == "item" {
77                    item = Some(PartialItem::default());
78                }
79                path.push(name);
80                text.clear();
81            }
82            Event::Empty(empty) => {
83                // enclosure is the only empty tag in the feed carrying data we need
84                if local_name(empty.name().as_ref()) == "enclosure"
85                    && let Some(partial) = item.as_mut()
86                {
87                    let mut url = None;
88                    let mut length = None;
89                    let mut mime = None;
90                    for attribute in empty.attributes().flatten() {
91                        // `value` is the raw attribute text; RSS writes `&` in an
92                        // enclosure URL as `&amp;`, and that URL is both the download
93                        // target and the episode's identity
94                        let Ok(value) = attribute.normalized_value(XmlVersion::Implicit1_0) else {
95                            continue;
96                        };
97                        match local_name(attribute.key.as_ref()) {
98                            "url" => url = Some(value.into_owned()),
99                            "length" => length = value.parse::<u64>().ok(),
100                            "type" => mime = Some(value.into_owned()),
101                            _ => {}
102                        }
103                    }
104                    if mime.as_deref() == Some("audio/mpeg")
105                        && let Some(url) = url
106                    {
107                        partial.enclosure_url = Some(url);
108                        partial.byte_len = length.unwrap_or(0);
109                    }
110                }
111            }
112            Event::Text(chunk) => text.push_str(&chunk.xml10_content()),
113            Event::CData(chunk) => text.push_str(&chunk.xml10_content()),
114            Event::GeneralRef(reference) => {
115                if let Some(character) = reference
116                    .resolve_char_ref()
117                    .map_err(|error| malformed(&error.to_string()))?
118                {
119                    text.push(character);
120                } else if let Some(resolved) = resolve_predefined_entity(&reference) {
121                    text.push_str(resolved);
122                }
123            }
124            Event::End(end) => {
125                let name = local_name(end.name().as_ref()).to_owned();
126                path.pop();
127                // only the direct children of an <item> carry episode data, so text from
128                // the channel-level elements of the same name is ignored
129                let in_item = path.last().map(String::as_str) == Some("item");
130                if in_item && let Some(partial) = item.as_mut() {
131                    match name.as_str() {
132                        "title" => partial.title = Some(text.trim().to_owned()),
133                        "link" => partial.link = Some(text.trim().to_owned()),
134                        "duration" => partial.duration_secs = parse_duration(&text),
135                        "pubDate" => partial.published_at = parse_pub_date(&text),
136                        _ => {}
137                    }
138                }
139                if name == "item"
140                    && let Some(episode) = item.take().and_then(PartialItem::into_episode)
141                {
142                    episodes.push(episode);
143                }
144                text.clear();
145            }
146            _ => {}
147        }
148    }
149
150    if !saw_rss {
151        return Err(malformed("It has no <rss> or <channel> element"));
152    }
153    // an error page rendered as RSS parses, and a catalog built from it would be cached
154    // over the good one for the whole of the cache's lifetime
155    if episodes.is_empty() {
156        return Err(malformed("It lists no episode with an audio enclosure"));
157    }
158
159    Ok(episodes)
160}
161
162/// Strips any namespace prefix, so `itunes:duration` reads as `duration`.
163fn local_name(raw: &str) -> &str {
164    match raw.rsplit_once(':') {
165        Some((_, local)) => local,
166        None => raw,
167    }
168}
169
170#[derive(Debug, Default, Clone)]
171struct PartialItem {
172    title: Option<String>,
173    link: Option<String>,
174    enclosure_url: Option<String>,
175    byte_len: u64,
176    duration_secs: Option<u64>,
177    published_at: Option<i64>,
178}
179
180impl PartialItem {
181    /// An item without an `audio/mpeg` enclosure is not an episode and is dropped; every
182    /// other field falls back to a neutral value rather than dropping an episode the feed
183    /// does list.
184    fn into_episode(self) -> Option<Episode> {
185        Some(Episode {
186            title: self.title.unwrap_or_default(),
187            link: self.link.unwrap_or_default(),
188            enclosure_url: self.enclosure_url?,
189            byte_len: self.byte_len,
190            duration_secs: self.duration_secs.unwrap_or(0),
191            published_at: self.published_at.unwrap_or(0),
192            slug: None,
193            bundle_title: None,
194            order: None,
195            tracklist: None,
196            body: None,
197            links: None,
198            special: false,
199        })
200    }
201}
202
203/// Parses `itunes:duration`, which the feed writes as `H:MM:SS` but which the format also
204/// permits as `MM:SS` or as whole seconds.
205fn parse_duration(raw: &str) -> Option<u64> {
206    let mut total = 0u64;
207    for part in raw.trim().split(':') {
208        total = total
209            .checked_mul(60)?
210            .checked_add(part.trim().parse().ok()?)?;
211    }
212    Some(total)
213}
214
215/// The years a `pubDate` may name.
216///
217/// Anything outside overflows the arithmetic below, which wraps in a release build and
218/// panics in a debug one, and no real feed carries it.
219const PUB_DATE_YEARS: std::ops::RangeInclusive<i64> = 1..=9999;
220
221/// Parses an RFC 2822 `pubDate` into whole seconds since the Unix epoch.
222fn parse_pub_date(raw: &str) -> Option<i64> {
223    let rest = match raw.trim().split_once(',') {
224        Some((_weekday, rest)) => rest,
225        None => raw.trim(),
226    };
227    let mut fields = rest.split_whitespace();
228
229    let day: i64 = fields.next()?.parse().ok()?;
230    let month = month_number(fields.next()?)?;
231    let year: i64 = fields.next()?.parse().ok()?;
232    if !PUB_DATE_YEARS.contains(&year) || !(1..=31).contains(&day) {
233        return None;
234    }
235
236    let mut clock = fields.next()?.split(':');
237    let hour: i64 = clock.next()?.parse().ok()?;
238    let minute: i64 = clock.next()?.parse().ok()?;
239    let second: i64 = clock.next().unwrap_or("0").parse().ok()?;
240    if !(0..=23).contains(&hour) || !(0..=59).contains(&minute) || !(0..=60).contains(&second) {
241        return None;
242    }
243
244    let offset = fields.next().and_then(zone_offset_secs).unwrap_or(0);
245
246    Some(days_from_civil(year, month, day) * 86_400 + hour * 3_600 + minute * 60 + second - offset)
247}
248
249fn month_number(name: &str) -> Option<i64> {
250    const MONTHS: [&str; 12] = [
251        "Jan", "Feb", "Mar", "Apr", "May", "Jun", "Jul", "Aug", "Sep", "Oct", "Nov", "Dec",
252    ];
253    MONTHS
254        .iter()
255        .position(|month| name.eq_ignore_ascii_case(month))
256        .map(|index| index as i64 + 1)
257}
258
259/// Seconds east of UTC for a `+HHMM`, `-HHMM`, or named zone.
260///
261/// RFC 5322 section 4.3 gives the North American zone names numeric offsets; everything
262/// else it leaves unknown, which reads as UTC.
263fn zone_offset_secs(zone: &str) -> Option<i64> {
264    let (sign, digits) = match zone.split_at_checked(1)? {
265        ("+", digits) => (1, digits),
266        ("-", digits) => (-1, digits),
267        _ => return Some(named_zone_offset_secs(zone)),
268    };
269    if digits.len() != 4 || !digits.bytes().all(|byte| byte.is_ascii_digit()) {
270        return Some(0);
271    }
272    let hours: i64 = digits[..2].parse().ok()?;
273    let minutes: i64 = digits[2..].parse().ok()?;
274    Some(sign * (hours * 3_600 + minutes * 60))
275}
276
277/// Days between 1970-01-01 and the given civil date, by Howard Hinnant's algorithm.
278/// Seconds east of UTC for one of the zone names RFC 5322 section 4.3 defines, and zero
279/// for anything else.
280fn named_zone_offset_secs(zone: &str) -> i64 {
281    let hours = match zone.to_ascii_uppercase().as_str() {
282        "EDT" => -4,
283        "EST" | "CDT" => -5,
284        "CST" | "MDT" => -6,
285        "MST" | "PDT" => -7,
286        "PST" => -8,
287        _ => 0,
288    };
289    hours * 3_600
290}
291
292fn days_from_civil(year: i64, month: i64, day: i64) -> i64 {
293    let year = if month <= 2 { year - 1 } else { year };
294    let era = if year >= 0 { year } else { year - 399 } / 400;
295    let year_of_era = year - era * 400;
296    let day_of_year = (153 * (if month > 2 { month - 3 } else { month + 9 }) + 2) / 5 + day - 1;
297    let day_of_era = year_of_era * 365 + year_of_era / 4 - year_of_era / 100 + day_of_year;
298    era * 146_097 + day_of_era - 719_468
299}
300
301#[cfg(test)]
302mod tests {
303    use super::*;
304
305    const FEED: &str = include_str!("../../tests/fixtures/rss.xml");
306
307    #[test]
308    fn the_checked_in_feed_parses_into_every_episode() {
309        let episodes = parse(FEED).unwrap();
310        assert_eq!(episodes.len(), 79);
311    }
312
313    #[test]
314    fn every_episode_carries_its_guaranteed_fields() {
315        for episode in parse(FEED).unwrap() {
316            assert!(!episode.title.is_empty(), "{episode:?}");
317            assert!(!episode.link.is_empty(), "{episode:?}");
318            assert!(
319                episode.enclosure_url.ends_with(".mp3"),
320                "{}",
321                episode.enclosure_url
322            );
323            assert!(episode.byte_len > 0, "{episode:?}");
324            assert!(episode.duration_secs > 0, "{episode:?}");
325            assert!(episode.published_at > 0, "{episode:?}");
326        }
327    }
328
329    #[test]
330    fn every_episode_carries_only_feed_fields_before_enrichment() {
331        for episode in parse(FEED).unwrap() {
332            assert!(episode.slug.is_none());
333            assert!(episode.order.is_none());
334            assert!(episode.tracklist.is_none());
335            assert!(episode.body.is_none());
336            assert!(episode.links.is_none());
337        }
338    }
339
340    #[test]
341    fn enclosure_urls_are_unique_so_they_can_identify_an_episode() {
342        let episodes = parse(FEED).unwrap();
343        let mut urls: Vec<&str> = episodes
344            .iter()
345            .map(|episode| episode.enclosure_url.as_str())
346            .collect();
347        urls.sort_unstable();
348        urls.dedup();
349        assert_eq!(urls.len(), episodes.len());
350    }
351
352    #[test]
353    fn feed_order_is_preserved() {
354        let episodes = parse(FEED).unwrap();
355        assert_eq!(episodes[0].title, "Episode 79: Corticyte");
356        assert_eq!(episodes[78].title, "Episode 01: Datassette");
357        assert_eq!(
358            episodes[0].enclosure_url,
359            "https://datashat.net/music_for_programming_79-corticyte.mp3"
360        );
361        assert_eq!(episodes[0].byte_len, 441_077_163);
362        assert_eq!(episodes[0].duration_secs, 4 * 3_600);
363    }
364
365    #[test]
366    fn a_body_that_is_not_rss_is_a_parse_failure() {
367        let error = parse("<html><body>not a feed</body></html>").unwrap_err();
368        assert!(error.to_string().contains("not a valid RSS document"));
369    }
370
371    #[test]
372    fn a_body_that_is_not_xml_is_a_parse_failure() {
373        assert!(parse("<rss><channel><item></rss>").is_err());
374        assert!(parse("").is_err());
375    }
376
377    #[test]
378    fn a_parse_failure_is_distinguishable_from_a_network_failure() {
379        let parse_failure = parse("<html></html>").unwrap_err();
380        let network_failure = unreachable("connection refused");
381        assert_ne!(parse_failure.code(), network_failure.code());
382        assert_eq!(
383            network_failure.code(),
384            crate::error::ErrorCode::CatalogUnavailable
385        );
386    }
387
388    #[test]
389    fn a_non_success_status_names_the_feed_as_unreachable() {
390        let error = unreachable("it returned HTTP 503 Service Unavailable");
391
392        assert_eq!(error.code(), crate::error::ErrorCode::CatalogUnavailable);
393        assert!(error.to_string().contains(super::super::FEED_URL));
394        assert!(error.to_string().contains("unreachable"));
395        assert!(error.to_string().contains("503"));
396    }
397
398    #[test]
399    fn items_without_an_audio_enclosure_are_not_episodes() {
400        let xml = r#"<rss><channel>
401            <item><title>No audio</title><link>a</link>
402                <enclosure url="a.jpg" length="1" type="image/jpeg"/></item>
403            <item><title>No enclosure at all</title><link>b</link></item>
404            <item><title>Audio</title><link>c</link>
405                <pubDate>Tue, 22 Feb 2011 17:17:58 GMT</pubDate>
406                <itunes:duration>1:02:16</itunes:duration>
407                <enclosure url="c.mp3" length="7" type="audio/mpeg"/></item>
408        </channel></rss>"#;
409
410        let episodes = parse(xml).unwrap();
411
412        assert_eq!(episodes.len(), 1);
413        assert_eq!(episodes[0].enclosure_url, "c.mp3");
414        assert_eq!(episodes[0].byte_len, 7);
415        assert_eq!(episodes[0].duration_secs, 3_736);
416    }
417
418    #[test]
419    fn channel_level_elements_do_not_leak_into_an_episode() {
420        let xml = r#"<rss><channel>
421            <title>Music For Programming</title>
422            <link>https://musicforprogramming.net/</link>
423            <itunes:duration>9:99:99</itunes:duration>
424            <item><title>Episode 01</title><link>ep</link>
425                <enclosure url="c.mp3" length="7" type="audio/mpeg"/></item>
426        </channel></rss>"#;
427
428        let episodes = parse(xml).unwrap();
429
430        assert_eq!(episodes[0].title, "Episode 01");
431        assert_eq!(episodes[0].link, "ep");
432        assert_eq!(episodes[0].duration_secs, 0);
433    }
434
435    #[test]
436    fn entities_in_a_title_are_decoded() {
437        let xml = r#"<rss><channel><item>
438            <title>A &amp; B &#66;</title><link>x</link>
439            <enclosure url="c.mp3" length="1" type="audio/mpeg"/>
440        </item></channel></rss>"#;
441
442        assert_eq!(parse(xml).unwrap()[0].title, "A & B B");
443    }
444
445    /// A five-digit year overflowed the day arithmetic, which panics a debug build and
446    /// wraps a release one into a date that sorts as nonsense.
447    #[test]
448    fn a_date_too_far_out_to_compute_is_refused_rather_than_overflowing() {
449        assert_eq!(
450            parse_pub_date("Mon, 01 Jan 999999999999999 00:00:00 GMT"),
451            None
452        );
453        assert_eq!(
454            parse_pub_date("Mon, 01 Jan -999999999999 00:00:00 GMT"),
455            None
456        );
457        assert_eq!(parse_pub_date("Mon, 99 Jan 2024 00:00:00 GMT"), None);
458        assert_eq!(parse_pub_date("Mon, 01 Jan 2024 99:00:00 GMT"), None);
459    }
460
461    /// The item is dropped rather than the whole feed: the enclosure is what makes an
462    /// episode, and a date that will not parse falls back to the epoch.
463    #[test]
464    fn an_item_whose_date_cannot_be_computed_still_yields_an_episode() {
465        let xml = r#"<rss><channel><item>
466            <title>Episode 01</title><link>ep</link>
467            <pubDate>Mon, 01 Jan 999999999999999 00:00:00 GMT</pubDate>
468            <enclosure url="c.mp3" length="7" type="audio/mpeg"/>
469        </item></channel></rss>"#;
470
471        let episodes = parse(xml).unwrap();
472
473        assert_eq!(episodes.len(), 1);
474        assert_eq!(episodes[0].published_at, 0);
475    }
476
477    #[test]
478    fn a_feed_listing_no_episode_is_a_parse_failure() {
479        let error = parse("<rss><channel><title>Music For Programming</title></channel></rss>")
480            .unwrap_err();
481
482        assert_eq!(error.code(), crate::error::ErrorCode::Internal);
483        assert!(error.to_string().contains("lists no episode"));
484    }
485
486    /// RSS writes `&` in a URL as `&amp;`, and the enclosure URL is both the download
487    /// target and the episode's identity.
488    #[test]
489    fn an_escaped_enclosure_url_is_decoded() {
490        let xml = r#"<rss><channel><item>
491            <title>Episode 01</title><link>ep</link>
492            <enclosure url="https://datashat.net/one.mp3?a=1&amp;b=2" length="7" type="audio/mpeg"/>
493        </item></channel></rss>"#;
494
495        assert_eq!(
496            parse(xml).unwrap()[0].enclosure_url,
497            "https://datashat.net/one.mp3?a=1&b=2"
498        );
499    }
500
501    #[test]
502    fn the_zone_names_rfc_5322_defines_are_not_read_as_utc() {
503        assert_eq!(
504            parse_pub_date("Tue, 22 Feb 2011 12:17:58 EST"),
505            parse_pub_date("Tue, 22 Feb 2011 17:17:58 GMT")
506        );
507        assert_eq!(
508            parse_pub_date("Tue, 22 Feb 2011 09:17:58 PST"),
509            parse_pub_date("Tue, 22 Feb 2011 17:17:58 GMT")
510        );
511        // an unknown name stays UTC, which is what an unknown zone means
512        assert_eq!(
513            parse_pub_date("Tue, 22 Feb 2011 17:17:58 XYZ"),
514            parse_pub_date("Tue, 22 Feb 2011 17:17:58 GMT")
515        );
516    }
517
518    #[test]
519    fn durations_parse_in_every_permitted_shape() {
520        assert_eq!(parse_duration("4:00:00"), Some(14_400));
521        assert_eq!(parse_duration("51:15"), Some(3_075));
522        assert_eq!(parse_duration("90"), Some(90));
523        assert_eq!(parse_duration("not a duration"), None);
524        assert_eq!(parse_duration(""), None);
525    }
526
527    #[test]
528    fn pub_dates_parse_to_the_unix_epoch() {
529        assert_eq!(parse_pub_date("Thu, 01 Jan 1970 00:00:00 GMT"), Some(0));
530        assert_eq!(
531            parse_pub_date("Tue, 22 Feb 2011 17:17:58 GMT"),
532            Some(1_298_395_078)
533        );
534        assert_eq!(
535            parse_pub_date("Mon, 24 Aug 2026 17:18:00 GMT"),
536            Some(1_787_591_880)
537        );
538        assert_eq!(
539            parse_pub_date("24 Aug 2026 17:18:00 +0000"),
540            parse_pub_date("Mon, 24 Aug 2026 12:18:00 -0500")
541        );
542        assert_eq!(parse_pub_date("nonsense"), None);
543    }
544}