1use quick_xml::XmlVersion;
9use quick_xml::escape::resolve_predefined_entity;
10use quick_xml::events::Event;
11use quick_xml::reader::Reader;
12
13use crate::error::{Error, Result};
14use crate::model::Episode;
15
16pub async fn fetch(client: &reqwest::Client) -> Result<Vec<Episode>> {
19 let response = super::request(client, super::FEED_URL)
20 .send()
21 .await
22 .map_err(|error| unreachable(&error.to_string()))?;
23
24 let status = response.status();
25 if !status.is_success() {
26 return Err(unreachable(&format!("It returned HTTP {status}")));
27 }
28
29 let body = response
30 .text()
31 .await
32 .map_err(|error| unreachable(&error.to_string()))?;
33
34 parse(&body)
35}
36
37fn unreachable(reason: &str) -> Error {
38 Error::CatalogUnavailable(format!(
39 "The feed at {} is unreachable: {reason}",
40 super::FEED_URL
41 ))
42}
43
44fn malformed(reason: &str) -> Error {
45 Error::Internal(format!(
46 "The feed at {} is not a valid RSS document: {reason}",
47 super::FEED_URL
48 ))
49}
50
51pub fn parse(xml: &str) -> Result<Vec<Episode>> {
54 let mut reader = Reader::from_str(xml);
57
58 let mut episodes = Vec::new();
59 let mut saw_rss = false;
60 let mut item: Option<PartialItem> = None;
61 let mut text = String::new();
62 let mut path: Vec<String> = Vec::new();
63
64 loop {
65 let event = reader
66 .read_event()
67 .map_err(|error| malformed(&error.to_string()))?;
68
69 match event {
70 Event::Eof => break,
71 Event::Start(start) => {
72 let name = local_name(start.name().as_ref()).to_owned();
73 if name == "rss" || name == "channel" {
74 saw_rss = true;
75 }
76 if name == "item" {
77 item = Some(PartialItem::default());
78 }
79 path.push(name);
80 text.clear();
81 }
82 Event::Empty(empty) => {
83 if local_name(empty.name().as_ref()) == "enclosure"
85 && let Some(partial) = item.as_mut()
86 {
87 let mut url = None;
88 let mut length = None;
89 let mut mime = None;
90 for attribute in empty.attributes().flatten() {
91 let Ok(value) = attribute.normalized_value(XmlVersion::Implicit1_0) else {
95 continue;
96 };
97 match local_name(attribute.key.as_ref()) {
98 "url" => url = Some(value.into_owned()),
99 "length" => length = value.parse::<u64>().ok(),
100 "type" => mime = Some(value.into_owned()),
101 _ => {}
102 }
103 }
104 if mime.as_deref() == Some("audio/mpeg")
105 && let Some(url) = url
106 {
107 partial.enclosure_url = Some(url);
108 partial.byte_len = length.unwrap_or(0);
109 }
110 }
111 }
112 Event::Text(chunk) => text.push_str(&chunk.xml10_content()),
113 Event::CData(chunk) => text.push_str(&chunk.xml10_content()),
114 Event::GeneralRef(reference) => {
115 if let Some(character) = reference
116 .resolve_char_ref()
117 .map_err(|error| malformed(&error.to_string()))?
118 {
119 text.push(character);
120 } else if let Some(resolved) = resolve_predefined_entity(&reference) {
121 text.push_str(resolved);
122 }
123 }
124 Event::End(end) => {
125 let name = local_name(end.name().as_ref()).to_owned();
126 path.pop();
127 let in_item = path.last().map(String::as_str) == Some("item");
130 if in_item && let Some(partial) = item.as_mut() {
131 match name.as_str() {
132 "title" => partial.title = Some(text.trim().to_owned()),
133 "link" => partial.link = Some(text.trim().to_owned()),
134 "duration" => partial.duration_secs = parse_duration(&text),
135 "pubDate" => partial.published_at = parse_pub_date(&text),
136 _ => {}
137 }
138 }
139 if name == "item"
140 && let Some(episode) = item.take().and_then(PartialItem::into_episode)
141 {
142 episodes.push(episode);
143 }
144 text.clear();
145 }
146 _ => {}
147 }
148 }
149
150 if !saw_rss {
151 return Err(malformed("It has no <rss> or <channel> element"));
152 }
153 if episodes.is_empty() {
156 return Err(malformed("It lists no episode with an audio enclosure"));
157 }
158
159 Ok(episodes)
160}
161
162fn local_name(raw: &str) -> &str {
164 match raw.rsplit_once(':') {
165 Some((_, local)) => local,
166 None => raw,
167 }
168}
169
170#[derive(Debug, Default, Clone)]
171struct PartialItem {
172 title: Option<String>,
173 link: Option<String>,
174 enclosure_url: Option<String>,
175 byte_len: u64,
176 duration_secs: Option<u64>,
177 published_at: Option<i64>,
178}
179
180impl PartialItem {
181 fn into_episode(self) -> Option<Episode> {
185 Some(Episode {
186 title: self.title.unwrap_or_default(),
187 link: self.link.unwrap_or_default(),
188 enclosure_url: self.enclosure_url?,
189 byte_len: self.byte_len,
190 duration_secs: self.duration_secs.unwrap_or(0),
191 published_at: self.published_at.unwrap_or(0),
192 slug: None,
193 bundle_title: None,
194 order: None,
195 tracklist: None,
196 body: None,
197 links: None,
198 special: false,
199 })
200 }
201}
202
203fn parse_duration(raw: &str) -> Option<u64> {
206 let mut total = 0u64;
207 for part in raw.trim().split(':') {
208 total = total
209 .checked_mul(60)?
210 .checked_add(part.trim().parse().ok()?)?;
211 }
212 Some(total)
213}
214
215const PUB_DATE_YEARS: std::ops::RangeInclusive<i64> = 1..=9999;
220
221fn parse_pub_date(raw: &str) -> Option<i64> {
223 let rest = match raw.trim().split_once(',') {
224 Some((_weekday, rest)) => rest,
225 None => raw.trim(),
226 };
227 let mut fields = rest.split_whitespace();
228
229 let day: i64 = fields.next()?.parse().ok()?;
230 let month = month_number(fields.next()?)?;
231 let year: i64 = fields.next()?.parse().ok()?;
232 if !PUB_DATE_YEARS.contains(&year) || !(1..=31).contains(&day) {
233 return None;
234 }
235
236 let mut clock = fields.next()?.split(':');
237 let hour: i64 = clock.next()?.parse().ok()?;
238 let minute: i64 = clock.next()?.parse().ok()?;
239 let second: i64 = clock.next().unwrap_or("0").parse().ok()?;
240 if !(0..=23).contains(&hour) || !(0..=59).contains(&minute) || !(0..=60).contains(&second) {
241 return None;
242 }
243
244 let offset = fields.next().and_then(zone_offset_secs).unwrap_or(0);
245
246 Some(days_from_civil(year, month, day) * 86_400 + hour * 3_600 + minute * 60 + second - offset)
247}
248
249fn month_number(name: &str) -> Option<i64> {
250 const MONTHS: [&str; 12] = [
251 "Jan", "Feb", "Mar", "Apr", "May", "Jun", "Jul", "Aug", "Sep", "Oct", "Nov", "Dec",
252 ];
253 MONTHS
254 .iter()
255 .position(|month| name.eq_ignore_ascii_case(month))
256 .map(|index| index as i64 + 1)
257}
258
259fn zone_offset_secs(zone: &str) -> Option<i64> {
264 let (sign, digits) = match zone.split_at_checked(1)? {
265 ("+", digits) => (1, digits),
266 ("-", digits) => (-1, digits),
267 _ => return Some(named_zone_offset_secs(zone)),
268 };
269 if digits.len() != 4 || !digits.bytes().all(|byte| byte.is_ascii_digit()) {
270 return Some(0);
271 }
272 let hours: i64 = digits[..2].parse().ok()?;
273 let minutes: i64 = digits[2..].parse().ok()?;
274 Some(sign * (hours * 3_600 + minutes * 60))
275}
276
277fn named_zone_offset_secs(zone: &str) -> i64 {
281 let hours = match zone.to_ascii_uppercase().as_str() {
282 "EDT" => -4,
283 "EST" | "CDT" => -5,
284 "CST" | "MDT" => -6,
285 "MST" | "PDT" => -7,
286 "PST" => -8,
287 _ => 0,
288 };
289 hours * 3_600
290}
291
292fn days_from_civil(year: i64, month: i64, day: i64) -> i64 {
293 let year = if month <= 2 { year - 1 } else { year };
294 let era = if year >= 0 { year } else { year - 399 } / 400;
295 let year_of_era = year - era * 400;
296 let day_of_year = (153 * (if month > 2 { month - 3 } else { month + 9 }) + 2) / 5 + day - 1;
297 let day_of_era = year_of_era * 365 + year_of_era / 4 - year_of_era / 100 + day_of_year;
298 era * 146_097 + day_of_era - 719_468
299}
300
301#[cfg(test)]
302mod tests {
303 use super::*;
304
305 const FEED: &str = include_str!("../../tests/fixtures/rss.xml");
306
307 #[test]
308 fn the_checked_in_feed_parses_into_every_episode() {
309 let episodes = parse(FEED).unwrap();
310 assert_eq!(episodes.len(), 79);
311 }
312
313 #[test]
314 fn every_episode_carries_its_guaranteed_fields() {
315 for episode in parse(FEED).unwrap() {
316 assert!(!episode.title.is_empty(), "{episode:?}");
317 assert!(!episode.link.is_empty(), "{episode:?}");
318 assert!(
319 episode.enclosure_url.ends_with(".mp3"),
320 "{}",
321 episode.enclosure_url
322 );
323 assert!(episode.byte_len > 0, "{episode:?}");
324 assert!(episode.duration_secs > 0, "{episode:?}");
325 assert!(episode.published_at > 0, "{episode:?}");
326 }
327 }
328
329 #[test]
330 fn every_episode_carries_only_feed_fields_before_enrichment() {
331 for episode in parse(FEED).unwrap() {
332 assert!(episode.slug.is_none());
333 assert!(episode.order.is_none());
334 assert!(episode.tracklist.is_none());
335 assert!(episode.body.is_none());
336 assert!(episode.links.is_none());
337 }
338 }
339
340 #[test]
341 fn enclosure_urls_are_unique_so_they_can_identify_an_episode() {
342 let episodes = parse(FEED).unwrap();
343 let mut urls: Vec<&str> = episodes
344 .iter()
345 .map(|episode| episode.enclosure_url.as_str())
346 .collect();
347 urls.sort_unstable();
348 urls.dedup();
349 assert_eq!(urls.len(), episodes.len());
350 }
351
352 #[test]
353 fn feed_order_is_preserved() {
354 let episodes = parse(FEED).unwrap();
355 assert_eq!(episodes[0].title, "Episode 79: Corticyte");
356 assert_eq!(episodes[78].title, "Episode 01: Datassette");
357 assert_eq!(
358 episodes[0].enclosure_url,
359 "https://datashat.net/music_for_programming_79-corticyte.mp3"
360 );
361 assert_eq!(episodes[0].byte_len, 441_077_163);
362 assert_eq!(episodes[0].duration_secs, 4 * 3_600);
363 }
364
365 #[test]
366 fn a_body_that_is_not_rss_is_a_parse_failure() {
367 let error = parse("<html><body>not a feed</body></html>").unwrap_err();
368 assert!(error.to_string().contains("not a valid RSS document"));
369 }
370
371 #[test]
372 fn a_body_that_is_not_xml_is_a_parse_failure() {
373 assert!(parse("<rss><channel><item></rss>").is_err());
374 assert!(parse("").is_err());
375 }
376
377 #[test]
378 fn a_parse_failure_is_distinguishable_from_a_network_failure() {
379 let parse_failure = parse("<html></html>").unwrap_err();
380 let network_failure = unreachable("connection refused");
381 assert_ne!(parse_failure.code(), network_failure.code());
382 assert_eq!(
383 network_failure.code(),
384 crate::error::ErrorCode::CatalogUnavailable
385 );
386 }
387
388 #[test]
389 fn a_non_success_status_names_the_feed_as_unreachable() {
390 let error = unreachable("it returned HTTP 503 Service Unavailable");
391
392 assert_eq!(error.code(), crate::error::ErrorCode::CatalogUnavailable);
393 assert!(error.to_string().contains(super::super::FEED_URL));
394 assert!(error.to_string().contains("unreachable"));
395 assert!(error.to_string().contains("503"));
396 }
397
398 #[test]
399 fn items_without_an_audio_enclosure_are_not_episodes() {
400 let xml = r#"<rss><channel>
401 <item><title>No audio</title><link>a</link>
402 <enclosure url="a.jpg" length="1" type="image/jpeg"/></item>
403 <item><title>No enclosure at all</title><link>b</link></item>
404 <item><title>Audio</title><link>c</link>
405 <pubDate>Tue, 22 Feb 2011 17:17:58 GMT</pubDate>
406 <itunes:duration>1:02:16</itunes:duration>
407 <enclosure url="c.mp3" length="7" type="audio/mpeg"/></item>
408 </channel></rss>"#;
409
410 let episodes = parse(xml).unwrap();
411
412 assert_eq!(episodes.len(), 1);
413 assert_eq!(episodes[0].enclosure_url, "c.mp3");
414 assert_eq!(episodes[0].byte_len, 7);
415 assert_eq!(episodes[0].duration_secs, 3_736);
416 }
417
418 #[test]
419 fn channel_level_elements_do_not_leak_into_an_episode() {
420 let xml = r#"<rss><channel>
421 <title>Music For Programming</title>
422 <link>https://musicforprogramming.net/</link>
423 <itunes:duration>9:99:99</itunes:duration>
424 <item><title>Episode 01</title><link>ep</link>
425 <enclosure url="c.mp3" length="7" type="audio/mpeg"/></item>
426 </channel></rss>"#;
427
428 let episodes = parse(xml).unwrap();
429
430 assert_eq!(episodes[0].title, "Episode 01");
431 assert_eq!(episodes[0].link, "ep");
432 assert_eq!(episodes[0].duration_secs, 0);
433 }
434
435 #[test]
436 fn entities_in_a_title_are_decoded() {
437 let xml = r#"<rss><channel><item>
438 <title>A & B B</title><link>x</link>
439 <enclosure url="c.mp3" length="1" type="audio/mpeg"/>
440 </item></channel></rss>"#;
441
442 assert_eq!(parse(xml).unwrap()[0].title, "A & B B");
443 }
444
445 #[test]
448 fn a_date_too_far_out_to_compute_is_refused_rather_than_overflowing() {
449 assert_eq!(
450 parse_pub_date("Mon, 01 Jan 999999999999999 00:00:00 GMT"),
451 None
452 );
453 assert_eq!(
454 parse_pub_date("Mon, 01 Jan -999999999999 00:00:00 GMT"),
455 None
456 );
457 assert_eq!(parse_pub_date("Mon, 99 Jan 2024 00:00:00 GMT"), None);
458 assert_eq!(parse_pub_date("Mon, 01 Jan 2024 99:00:00 GMT"), None);
459 }
460
461 #[test]
464 fn an_item_whose_date_cannot_be_computed_still_yields_an_episode() {
465 let xml = r#"<rss><channel><item>
466 <title>Episode 01</title><link>ep</link>
467 <pubDate>Mon, 01 Jan 999999999999999 00:00:00 GMT</pubDate>
468 <enclosure url="c.mp3" length="7" type="audio/mpeg"/>
469 </item></channel></rss>"#;
470
471 let episodes = parse(xml).unwrap();
472
473 assert_eq!(episodes.len(), 1);
474 assert_eq!(episodes[0].published_at, 0);
475 }
476
477 #[test]
478 fn a_feed_listing_no_episode_is_a_parse_failure() {
479 let error = parse("<rss><channel><title>Music For Programming</title></channel></rss>")
480 .unwrap_err();
481
482 assert_eq!(error.code(), crate::error::ErrorCode::Internal);
483 assert!(error.to_string().contains("lists no episode"));
484 }
485
486 #[test]
489 fn an_escaped_enclosure_url_is_decoded() {
490 let xml = r#"<rss><channel><item>
491 <title>Episode 01</title><link>ep</link>
492 <enclosure url="https://datashat.net/one.mp3?a=1&b=2" length="7" type="audio/mpeg"/>
493 </item></channel></rss>"#;
494
495 assert_eq!(
496 parse(xml).unwrap()[0].enclosure_url,
497 "https://datashat.net/one.mp3?a=1&b=2"
498 );
499 }
500
501 #[test]
502 fn the_zone_names_rfc_5322_defines_are_not_read_as_utc() {
503 assert_eq!(
504 parse_pub_date("Tue, 22 Feb 2011 12:17:58 EST"),
505 parse_pub_date("Tue, 22 Feb 2011 17:17:58 GMT")
506 );
507 assert_eq!(
508 parse_pub_date("Tue, 22 Feb 2011 09:17:58 PST"),
509 parse_pub_date("Tue, 22 Feb 2011 17:17:58 GMT")
510 );
511 assert_eq!(
513 parse_pub_date("Tue, 22 Feb 2011 17:17:58 XYZ"),
514 parse_pub_date("Tue, 22 Feb 2011 17:17:58 GMT")
515 );
516 }
517
518 #[test]
519 fn durations_parse_in_every_permitted_shape() {
520 assert_eq!(parse_duration("4:00:00"), Some(14_400));
521 assert_eq!(parse_duration("51:15"), Some(3_075));
522 assert_eq!(parse_duration("90"), Some(90));
523 assert_eq!(parse_duration("not a duration"), None);
524 assert_eq!(parse_duration(""), None);
525 }
526
527 #[test]
528 fn pub_dates_parse_to_the_unix_epoch() {
529 assert_eq!(parse_pub_date("Thu, 01 Jan 1970 00:00:00 GMT"), Some(0));
530 assert_eq!(
531 parse_pub_date("Tue, 22 Feb 2011 17:17:58 GMT"),
532 Some(1_298_395_078)
533 );
534 assert_eq!(
535 parse_pub_date("Mon, 24 Aug 2026 17:18:00 GMT"),
536 Some(1_787_591_880)
537 );
538 assert_eq!(
539 parse_pub_date("24 Aug 2026 17:18:00 +0000"),
540 parse_pub_date("Mon, 24 Aug 2026 12:18:00 -0500")
541 );
542 assert_eq!(parse_pub_date("nonsense"), None);
543 }
544}