Skip to main content

webserver_base/feed/
mod.rs

1//! RSS 2.0, Atom 1.0 and JSON Feed 1.1 for a site's one content stream.
2//!
3//! A site gets exactly one feed, served in all three formats. Which entries it
4//! holds is passed in rather than derived: a blog is a
5//! [`dynamic_page_group`](crate::webserver::Pages::dynamic_page_group), and a
6//! dynamic page builds its template data per request, so there is nothing to
7//! harvest at boot.
8//!
9//! Three properties are load-bearing and easy to break:
10//!
11//! - **The documents are byte-stable.** `<lastBuildDate>` and `<updated>` come
12//!   from the newest entry, never from the clock, and ties break on path. A
13//!   timestamp or a map iteration order in the body would change the bytes on
14//!   every deploy, and every subscriber would re-download a feed that did not
15//!   change.
16//! - **Content is rewritten, not copied.** See [`rewrite`].
17//! - **Text is escaped by the writer.** `quick_xml::events::BytesText::new`
18//!   escapes; nothing here hand-rolls it. What escaping cannot fix — characters
19//!   outside XML's `Char` production — is stripped in [`sanitize`].
20
21mod atom;
22mod error;
23mod json;
24mod rewrite;
25mod rss;
26mod sanitize;
27
28use chrono::{DateTime, Utc};
29use tracing::{error, instrument};
30
31pub use error::FeedError;
32
33/// Where the RSS 2.0 document is served.
34///
35/// `/rss.xml` rather than `/feed.xml`: the latter names RSS on some sites and
36/// Atom on others, so it is a path that has to be looked up to be understood.
37pub const RSS_PATH: &str = "/rss.xml";
38
39/// Where the Atom 1.0 document is served.
40pub const ATOM_PATH: &str = "/atom.xml";
41
42/// Where the JSON Feed 1.1 document is served.
43pub const JSON_PATH: &str = "/feed.json";
44
45/// How many entries a feed carries, newest first.
46///
47/// The complete inventory is `sitemap.xml`'s job; a feed is the recency signal,
48/// and Google's own guidance for a feed-as-sitemap is that it holds only what
49/// changed recently. Twenty is roughly a year of a personal blog's cadence.
50pub const MAX_FEED_ENTRIES: usize = 20;
51
52/// How long a feed may be served from a cache before revalidating.
53///
54/// Feeds are the most-polled document a blog serves. Thirty minutes is shorter
55/// than every default poll interval that matters, so it costs no real freshness
56/// while letting an intermediary absorb duplicate polls.
57pub const FEED_MAX_AGE_SECONDS: u32 = 1800;
58
59/// Channel-level metadata, all of it derived from what the site already
60/// declares rather than configured a second time.
61#[derive(Debug, Clone, PartialEq, Eq)]
62pub struct FeedSite {
63    /// The site origin, without a trailing slash.
64    pub base_url: String,
65    /// Who wrote it.
66    pub author: String,
67    /// An RFC 5646 tag, e.g. `en-US`.
68    pub language: String,
69    /// Absolute URL of the large square icon, if the icon set produced one.
70    pub icon_url: Option<String>,
71    /// Absolute URL of the favicon, if one exists.
72    pub favicon_url: Option<String>,
73    /// The rights line, e.g. `© 1998–2026 Todd Everett Griffin`.
74    pub copyright: String,
75}
76
77/// A site's one feed.
78///
79/// A named-field struct rather than a builder on purpose: a forgotten
80/// `content_html` is a teaser feed nobody meant to ship, and here it has to be
81/// typed as `None` to be omitted.
82#[derive(Debug, Clone, PartialEq, Eq)]
83pub struct Feed {
84    /// What a reader shows in its subscription list. Prefer the page's
85    /// `display_name` shape — `Blog | Todd Everett Griffin` — because a bare
86    /// `Blog` is unidentifiable beside forty other subscriptions.
87    pub title: String,
88    /// One sentence describing the stream, not the site.
89    pub description: String,
90    /// The HTML page this feed mirrors, e.g. `/blog`.
91    pub page_url: String,
92    /// Any order; the feed is sorted newest-first and truncated to
93    /// [`MAX_FEED_ENTRIES`].
94    pub entries: Vec<FeedEntry>,
95}
96
97/// One item in a feed.
98#[derive(Debug, Clone, PartialEq, Eq)]
99pub struct FeedEntry {
100    /// Site-relative, e.g. `/blog/rust-tips`. Becomes both the link and the
101    /// permanent id, so changing it republishes the post to every subscriber.
102    pub path: String,
103    pub title: String,
104    /// Required: an entry with no summary is a bare headline in every reader
105    /// that shows previews.
106    pub summary: String,
107    /// The rendered post HTML. `None` ships a teaser.
108    pub content_html: Option<String>,
109    pub published: DateTime<Utc>,
110    /// Atom requires a per-entry `<updated>`; this falls back to `published`.
111    pub modified: Option<DateTime<Utc>>,
112    pub tags: Vec<String>,
113    /// The post's own card image, site-relative. Unlike an image inside the
114    /// content, this one is *declared*, so an unresolved path fails the boot.
115    pub image: Option<String>,
116}
117
118impl FeedEntry {
119    /// `modified`, or `published` when the post was never revised.
120    #[must_use]
121    pub fn updated(&self) -> DateTime<Utc> {
122        self.modified.unwrap_or(self.published)
123    }
124}
125
126/// One rendered feed document.
127#[derive(Debug, Clone, PartialEq, Eq)]
128pub struct FeedDocument {
129    path: &'static str,
130    content_type: &'static str,
131    body: String,
132    etag: String,
133}
134
135impl FeedDocument {
136    /// Where it is served.
137    #[must_use]
138    pub const fn path(&self) -> &'static str {
139        self.path
140    }
141
142    /// Its `Content-Type`.
143    #[must_use]
144    pub const fn content_type(&self) -> &'static str {
145        self.content_type
146    }
147
148    /// The document itself.
149    #[must_use]
150    pub fn body(&self) -> &str {
151        &self.body
152    }
153
154    /// A strong validator over the body, quoted and ready for the header.
155    #[must_use]
156    pub fn etag(&self) -> &str {
157        &self.etag
158    }
159}
160
161/// All three documents, plus what they share.
162#[derive(Debug, Clone, PartialEq, Eq)]
163pub struct FeedSet {
164    documents: Vec<FeedDocument>,
165    last_modified: DateTime<Utc>,
166}
167
168impl FeedSet {
169    /// Every document, in serving order.
170    #[must_use]
171    pub fn documents(&self) -> &[FeedDocument] {
172        &self.documents
173    }
174
175    /// The newest entry's date, for `Last-Modified`.
176    #[must_use]
177    pub const fn last_modified(&self) -> DateTime<Utc> {
178        self.last_modified
179    }
180}
181
182/// Channel-level strings, already stripped of characters XML cannot hold.
183///
184/// Sanitizing at one boundary rather than at each use is deliberate: the bug
185/// this prevents is *forgetting a field*, and a per-field call site is exactly
186/// where that happens.
187#[derive(Debug, Clone)]
188struct Channel {
189    base_url: String,
190    title: String,
191    description: String,
192    author: String,
193    copyright: String,
194    language: String,
195    home_url: String,
196    icon_url: Option<String>,
197    favicon_url: Option<String>,
198}
199
200/// An entry with its content already rewritten and its URLs already absolute,
201/// which is all three serializers need.
202#[derive(Debug, Clone)]
203struct PreparedEntry {
204    url: String,
205    title: String,
206    summary: String,
207    content_html: Option<String>,
208    published: DateTime<Utc>,
209    updated: DateTime<Utc>,
210    tags: Vec<String>,
211    image_url: Option<String>,
212}
213
214/// Builds all three feed documents.
215///
216/// `resolve` maps a manifest path to its content-hashed one and returns `None`
217/// for an asset the manifest does not know.
218///
219/// # Errors
220///
221/// [`FeedError`] for anything that makes the feed wrong rather than merely
222/// degraded: an entry missing a title, an unrooted path, `srcset` in content,
223/// an unresolved declared image, or a serialization failure.
224#[instrument(skip_all)]
225pub fn build_feeds<F>(site: &FeedSite, feed: &Feed, resolve: F) -> Result<FeedSet, FeedError>
226where
227    F: Fn(&str) -> Option<String>,
228{
229    for diagnostic in diagnostics(feed, Utc::now()) {
230        // Boots and serves, but subscribers get something wrong and nothing
231        // else will say so. `warn!` is never seen by anyone.
232        error!("{diagnostic}");
233    }
234
235    let channel: Channel = channel(site, feed);
236    let entries: Vec<PreparedEntry> = prepare(site, feed, &resolve)?;
237
238    let last_modified: DateTime<Utc> = entries
239        .iter()
240        .map(|entry| entry.updated)
241        .max()
242        // An empty feed still needs a coherent `<updated>`; the site's own
243        // epoch is meaningless, so fall back to the start of time rather than
244        // to the clock, which would break byte-stability.
245        .unwrap_or_else(|| DateTime::UNIX_EPOCH.to_utc());
246
247    let documents: Vec<FeedDocument> = vec![
248        document(
249            RSS_PATH,
250            "application/rss+xml; charset=utf-8",
251            rss::render(&channel, &entries, last_modified)?,
252        ),
253        document(
254            ATOM_PATH,
255            "application/atom+xml; charset=utf-8",
256            atom::render(&channel, &entries, last_modified)?,
257        ),
258        document(
259            JSON_PATH,
260            "application/feed+json",
261            json::render(&channel, &entries)?,
262        ),
263    ];
264
265    Ok(FeedSet {
266        documents,
267        last_modified,
268    })
269}
270
271/// Sanitizes every channel-level string once, and resolves the home link.
272fn channel(site: &FeedSite, feed: &Feed) -> Channel {
273    let clean = |text: &str| -> String { sanitize::strip_illegal(text).0 };
274
275    Channel {
276        base_url: String::from(site.base_url.trim_end_matches('/')),
277        title: clean(&feed.title),
278        description: clean(&feed.description),
279        author: clean(&site.author),
280        copyright: clean(&site.copyright),
281        language: clean(&site.language),
282        home_url: absolute(&site.base_url, &feed.page_url),
283        icon_url: site.icon_url.clone(),
284        favicon_url: site.favicon_url.clone(),
285    }
286}
287
288/// Wraps a rendered body with the validator the route will serve it under.
289fn document(path: &'static str, content_type: &'static str, body: String) -> FeedDocument {
290    let etag: String = format!("\"{:x}\"", md5::compute(body.as_bytes()));
291    FeedDocument {
292        path,
293        content_type,
294        body,
295        etag,
296    }
297}
298
299/// Sorts, truncates, validates and rewrites, in that order.
300fn prepare<F>(site: &FeedSite, feed: &Feed, resolve: &F) -> Result<Vec<PreparedEntry>, FeedError>
301where
302    F: Fn(&str) -> Option<String>,
303{
304    let mut ordered: Vec<&FeedEntry> = feed.entries.iter().collect();
305    // Path breaks ties so the bytes cannot depend on the caller's iteration
306    // order — a `HashMap` upstream would otherwise change the ETag at random.
307    ordered.sort_by(|left, right| {
308        right
309            .published
310            .cmp(&left.published)
311            .then_with(|| left.path.cmp(&right.path))
312    });
313    ordered.truncate(MAX_FEED_ENTRIES);
314
315    let mut prepared: Vec<PreparedEntry> = Vec::with_capacity(ordered.len());
316    let mut unresolved: Vec<String> = Vec::new();
317    let mut stripped: usize = 0;
318
319    for entry in ordered {
320        validate_entry(entry)?;
321
322        let url: String = absolute(&site.base_url, &entry.path);
323
324        let image_url: Option<String> = match &entry.image {
325            Some(image) if is_external(image) => Some(image.clone()),
326            Some(image) => {
327                let key: &str = image.trim_start_matches('/');
328                let hashed: String = resolve(key).ok_or_else(|| FeedError::UnresolvedImage {
329                    path: entry.path.clone(),
330                    image: image.clone(),
331                })?;
332                Some(absolute(&site.base_url, &hashed))
333            }
334            None => None,
335        };
336
337        let (title, removed) = sanitize::strip_illegal(&entry.title);
338        stripped += removed;
339        let (summary, removed) = sanitize::strip_illegal(&entry.summary);
340        stripped += removed;
341
342        let mut tags: Vec<String> = Vec::with_capacity(entry.tags.len());
343        for tag in &entry.tags {
344            let (tag, removed) = sanitize::strip_illegal(tag);
345            stripped += removed;
346            tags.push(tag);
347        }
348
349        let content_html: Option<String> = match &entry.content_html {
350            Some(content) => {
351                let (content, removed) = sanitize::strip_illegal(content);
352                stripped += removed;
353                let rewritten: rewrite::Rewritten =
354                    rewrite::rewrite_content(&content, &site.base_url, &url, resolve);
355                unresolved.extend(rewritten.unresolved);
356                Some(rewritten.html)
357            }
358            None => None,
359        };
360
361        prepared.push(PreparedEntry {
362            url,
363            title,
364            summary,
365            content_html,
366            published: entry.published,
367            updated: entry.updated(),
368            tags,
369            image_url,
370        });
371    }
372
373    if stripped > 0 {
374        error!(
375            "stripped {stripped} character(s) that XML cannot represent from the feed; the \
376             documents are valid, but something upstream is emitting control characters"
377        );
378    }
379    for path in &unresolved {
380        error!(
381            "feed content references `{path}`, which is not in the asset manifest; it is a broken \
382             image in every reader, and on the page itself"
383        );
384    }
385
386    Ok(prepared)
387}
388
389/// The boot failures: an entry in this state produces output nobody wants.
390fn validate_entry(entry: &FeedEntry) -> Result<(), FeedError> {
391    if entry.path.trim().is_empty() {
392        return Err(FeedError::IncompleteEntry {
393            path: entry.path.clone(),
394            field: "path",
395        });
396    }
397    if !entry.path.starts_with('/') {
398        return Err(FeedError::UnrootedPath {
399            path: entry.path.clone(),
400        });
401    }
402    if entry.title.trim().is_empty() {
403        return Err(FeedError::IncompleteEntry {
404            path: entry.path.clone(),
405            field: "title",
406        });
407    }
408    if entry.summary.trim().is_empty() {
409        return Err(FeedError::IncompleteEntry {
410            path: entry.path.clone(),
411            field: "summary",
412        });
413    }
414    if let Some(content) = &entry.content_html
415        && rewrite::contains_srcset(content)
416    {
417        return Err(FeedError::SrcsetUnsupported {
418            path: entry.path.clone(),
419        });
420    }
421    Ok(())
422}
423
424/// The `error!`-tier findings: the feed serves, but a human should look.
425///
426/// Pure and clock-injected so the reporting itself is testable.
427fn diagnostics(feed: &Feed, now: DateTime<Utc>) -> Vec<String> {
428    let mut findings: Vec<String> = Vec::new();
429
430    if feed.entries.is_empty() {
431        findings.push(String::from(
432            "the feed is configured but has no entries; subscribers get an empty document",
433        ));
434    }
435
436    let mut seen: Vec<&str> = Vec::new();
437    for entry in &feed.entries {
438        if seen.contains(&entry.path.as_str()) {
439            findings.push(format!(
440                "feed entries share the path `{}`; readers deduplicate by id, so one of those \
441                 posts never reaches a subscriber",
442                entry.path
443            ));
444        } else {
445            seen.push(&entry.path);
446        }
447
448        if entry.published > now {
449            findings.push(format!(
450                "feed entry `{}` is published in the future; it pins to the top of the feed and \
451                 some readers hide it entirely",
452                entry.path
453            ));
454        }
455
456        if let Some(modified) = entry.modified
457            && modified < entry.published
458        {
459            findings.push(format!(
460                "feed entry `{}` was modified before it was published",
461                entry.path
462            ));
463        }
464    }
465
466    findings
467}
468
469/// RFC 3339 to the second, with `Z` rather than `+00:00`.
470///
471/// Shared by Atom and JSON Feed so the two cannot disagree about an instant.
472fn rfc3339(moment: DateTime<Utc>) -> String {
473    moment.to_rfc3339_opts(chrono::SecondsFormat::Secs, true)
474}
475
476/// The MIME type an image URL implies, for the one place Atom wants it.
477///
478/// Extension-derived rather than probed: the file has already been proved to
479/// exist, and re-reading its header bytes here would be a second source of
480/// truth for something the manifest path already states.
481fn mime_for(url: &str) -> Option<&'static str> {
482    let extension: &str = url.rsplit('.').next()?;
483    match extension.to_ascii_lowercase().as_str() {
484        "png" => Some("image/png"),
485        "jpg" | "jpeg" => Some("image/jpeg"),
486        "webp" => Some("image/webp"),
487        "gif" => Some("image/gif"),
488        "avif" => Some("image/avif"),
489        "svg" => Some("image/svg+xml"),
490        _ => None,
491    }
492}
493
494/// Whether a reference already names its own origin.
495fn is_external(reference: &str) -> bool {
496    reference.starts_with("http://") || reference.starts_with("https://")
497}
498
499/// Joins a site origin and a path, tolerating a slash on either side or both.
500///
501/// Deliberately the same rule as [`crate::sitemap`]: a second, subtly different
502/// one is how a feed and a sitemap come to disagree about a URL.
503fn absolute(base_url: &str, path: &str) -> String {
504    if is_external(path) {
505        return String::from(path);
506    }
507    let base: &str = base_url.trim_end_matches('/');
508    let path: &str = path.trim_start_matches('/');
509    if path.is_empty() {
510        format!("{base}/")
511    } else {
512        format!("{base}/{path}")
513    }
514}
515
516#[cfg(test)]
517mod tests {
518    use std::collections::BTreeMap;
519
520    use chrono::{DateTime, TimeZone, Utc};
521    use serde_json::Value;
522
523    use super::{
524        ATOM_PATH, Feed, FeedDocument, FeedEntry, FeedError, FeedSet, FeedSite, JSON_PATH,
525        MAX_FEED_ENTRIES, RSS_PATH, build_feeds, diagnostics,
526    };
527
528    const BASE: &str = "https://www.example.com";
529
530    fn moment(year: i32, month: u32, day: u32) -> DateTime<Utc> {
531        Utc.with_ymd_and_hms(year, month, day, 12, 0, 0)
532            .single()
533            .expect("a real instant")
534    }
535
536    fn site() -> FeedSite {
537        FeedSite {
538            base_url: String::from(BASE),
539            author: String::from("Todd Everett Griffin"),
540            language: String::from("en-US"),
541            icon_url: Some(format!("{BASE}/icon-512.png")),
542            favicon_url: Some(format!("{BASE}/favicon.ico")),
543            copyright: String::from("© 1998–2026 Todd Everett Griffin"),
544        }
545    }
546
547    fn entry(slug: &str, published: DateTime<Utc>) -> FeedEntry {
548        FeedEntry {
549            path: format!("/blog/{slug}"),
550            title: format!("Post {slug}"),
551            summary: format!("A summary of {slug}."),
552            content_html: Some(format!("<p>The body of {slug}.</p>")),
553            published,
554            modified: None,
555            tags: vec![String::from("rust")],
556            image: None,
557        }
558    }
559
560    fn feed(entries: Vec<FeedEntry>) -> Feed {
561        Feed {
562            title: String::from("Blog | Todd Everett Griffin"),
563            description: String::from("Writing on Rust and WebGPU."),
564            page_url: String::from("/blog"),
565            entries,
566        }
567    }
568
569    fn resolver() -> impl Fn(&str) -> Option<String> {
570        let mut manifest: BTreeMap<String, String> = BTreeMap::new();
571        manifest.insert(
572            String::from("static/image/blog/a.png"),
573            String::from("static/image/blog/a.abc123.png"),
574        );
575        move |path: &str| manifest.get(path).cloned()
576    }
577
578    fn build(feed: &Feed) -> FeedSet {
579        build_feeds(&site(), feed, resolver()).expect("builds")
580    }
581
582    fn body(set: &FeedSet, path: &str) -> String {
583        set.documents()
584            .iter()
585            .find(|document| document.path() == path)
586            .map_or_else(
587                || panic!("no document at {path}"),
588                |document| String::from(document.body()),
589            )
590    }
591
592    // ---- well-formedness, proved by a parser that did not write the document
593
594    #[test]
595    fn every_xml_document_parses_under_a_strict_conformant_parser() {
596        // Deliberately not quick-xml: it wrote these, and it is a non-validating
597        // pull parser that accepts characters outside XML's `Char` production.
598        // A generator checked by its own parser waves through exactly the bug
599        // this test exists for.
600        let set: FeedSet = build(&feed(vec![
601            entry("first", moment(2024, 3, 10)),
602            entry("second", moment(2026, 1, 2)),
603        ]));
604
605        for path in [RSS_PATH, ATOM_PATH] {
606            let document: String = body(&set, path);
607            roxmltree::Document::parse(&document)
608                .unwrap_or_else(|error| panic!("{path} is not well-formed XML: {error}"));
609        }
610    }
611
612    #[test]
613    fn hostile_text_survives_a_round_trip_instead_of_breaking_the_document() {
614        // Every character that has ever broken a hand-rolled feed, in one post.
615        let mut hostile: FeedEntry = entry("hostile", moment(2026, 1, 2));
616        hostile.title = String::from("Tom & Jerry <script> \"quoted\" 'single' ]]> &amp;");
617        hostile.summary = String::from("5 < 6 && 7 > 6");
618        hostile.content_html = Some(String::from("<p>a &amp; b ]]&gt; c</p>"));
619
620        let set: FeedSet = build(&feed(vec![hostile]));
621
622        for path in [RSS_PATH, ATOM_PATH] {
623            let document: String = body(&set, path);
624            roxmltree::Document::parse(&document)
625                .unwrap_or_else(|error| panic!("{path} broke on hostile text: {error}"));
626        }
627
628        let channel: rss::Channel =
629            rss::Channel::read_from(body(&set, RSS_PATH).as_bytes()).expect("valid RSS");
630        let expected: &str = "Tom & Jerry <script> \"quoted\" 'single' ]]> &amp;";
631        let actual: &str = channel.items()[0].title().expect("a title");
632        assert_eq!(expected, actual);
633    }
634
635    #[test]
636    fn a_control_character_is_stripped_rather_than_escaped_because_no_escape_exists() {
637        let mut broken: FeedEntry = entry("broken", moment(2026, 1, 2));
638        broken.title = String::from("before\u{0B}after");
639        broken.tags = vec![String::from("ru\u{0C}st")];
640
641        let set: FeedSet = build(&feed(vec![broken]));
642
643        // roxmltree is the only oracle that catches this: quick-xml's reader
644        // and both syndication crates parse it happily.
645        for path in [RSS_PATH, ATOM_PATH] {
646            let document: String = body(&set, path);
647            roxmltree::Document::parse(&document)
648                .unwrap_or_else(|error| panic!("{path} kept a non-XML character: {error}"));
649        }
650
651        assert!(body(&set, RSS_PATH).contains("beforeafter"));
652        assert!(body(&set, RSS_PATH).contains("rust"));
653    }
654
655    // ---- spec conformance, proved by independent implementations
656
657    #[test]
658    fn the_rss_document_round_trips_through_an_independent_rss_implementation() {
659        let set: FeedSet = build(&feed(vec![entry("first", moment(2024, 3, 10))]));
660        let channel: rss::Channel =
661            rss::Channel::read_from(body(&set, RSS_PATH).as_bytes()).expect("valid RSS 2.0");
662
663        assert_eq!("Blog | Todd Everett Griffin", channel.title());
664        assert_eq!("https://www.example.com/blog", channel.link());
665        assert_eq!("Writing on Rust and WebGPU.", channel.description());
666        assert_eq!(Some("en-US"), channel.language());
667        assert_eq!(
668            Some("© 1998–2026 Todd Everett Griffin"),
669            channel.copyright()
670        );
671
672        let expected_items: usize = 1;
673        let actual_items: usize = channel.items().len();
674        assert_eq!(expected_items, actual_items);
675
676        let item: &rss::Item = &channel.items()[0];
677        assert_eq!(Some("Post first"), item.title());
678        assert_eq!(Some("https://www.example.com/blog/first"), item.link());
679        assert_eq!(Some("A summary of first."), item.description());
680        assert_eq!(
681            Some("https://www.example.com/blog/first"),
682            item.guid().map(rss::Guid::value)
683        );
684        assert_eq!(Some(true), item.guid().map(rss::Guid::is_permalink));
685        assert_eq!(Some("<p>The body of first.</p>"), item.content());
686    }
687
688    #[test]
689    fn the_atom_document_round_trips_through_an_independent_atom_implementation() {
690        let set: FeedSet = build(&feed(vec![entry("first", moment(2024, 3, 10))]));
691        let parsed: atom_syndication::Feed =
692            atom_syndication::Feed::read_from(body(&set, ATOM_PATH).as_bytes())
693                .expect("valid Atom 1.0");
694
695        assert_eq!("Blog | Todd Everett Griffin", parsed.title().as_str());
696        assert_eq!("https://www.example.com/atom.xml", parsed.id());
697        assert_eq!(
698            Some("Writing on Rust and WebGPU."),
699            parsed.subtitle().map(atom_syndication::Text::as_str)
700        );
701        assert_eq!(
702            vec![String::from("Todd Everett Griffin")],
703            parsed
704                .authors()
705                .iter()
706                .map(|author| author.name().to_string())
707                .collect::<Vec<String>>()
708        );
709
710        let entry: &atom_syndication::Entry = &parsed.entries()[0];
711        assert_eq!("https://www.example.com/blog/first", entry.id());
712        assert_eq!("Post first", entry.title().as_str());
713        assert_eq!(
714            Some("<p>The body of first.</p>"),
715            entry.content().and_then(atom_syndication::Content::value)
716        );
717        assert_eq!(
718            vec![String::from("rust")],
719            entry
720                .categories()
721                .iter()
722                .map(|category| category.term().to_string())
723                .collect::<Vec<String>>()
724        );
725    }
726
727    #[test]
728    fn the_json_document_carries_every_field_the_1_1_spec_requires() {
729        let set: FeedSet = build(&feed(vec![entry("first", moment(2024, 3, 10))]));
730        let parsed: Value = serde_json::from_str(&body(&set, JSON_PATH)).expect("valid JSON");
731
732        assert_eq!("https://jsonfeed.org/version/1.1", parsed["version"]);
733        assert_eq!("Blog | Todd Everett Griffin", parsed["title"]);
734        assert_eq!("https://www.example.com/blog", parsed["home_page_url"]);
735        assert_eq!("https://www.example.com/feed.json", parsed["feed_url"]);
736        assert_eq!("en-US", parsed["language"]);
737        assert_eq!("Todd Everett Griffin", parsed["authors"][0]["name"]);
738
739        let item: &Value = &parsed["items"][0];
740        assert_eq!("https://www.example.com/blog/first", item["id"]);
741        assert_eq!("https://www.example.com/blog/first", item["url"]);
742        assert_eq!("Post first", item["title"]);
743        assert_eq!("<p>The body of first.</p>", item["content_html"]);
744        assert_eq!("2024-03-10T12:00:00Z", item["date_published"]);
745    }
746
747    // ---- the properties the caching design depends on
748
749    #[test]
750    fn building_the_same_feed_twice_produces_identical_bytes() {
751        // This is what makes the ETag work. A timestamp anywhere in a body
752        // would re-download the feed for every subscriber on every deploy, and
753        // nothing else in the suite would notice.
754        let declaration: Feed = feed(vec![
755            entry("first", moment(2024, 3, 10)),
756            entry("second", moment(2026, 1, 2)),
757        ]);
758
759        let expected: FeedSet = build(&declaration);
760        let actual: FeedSet = build(&declaration);
761        assert_eq!(expected, actual);
762    }
763
764    #[test]
765    fn entries_sharing_an_instant_are_ordered_deterministically_not_by_caller_order() {
766        // A `HashMap` upstream hands entries over in a different order each
767        // run; without a tie-break the bytes — and so the ETag — would move.
768        let same: DateTime<Utc> = moment(2026, 1, 2);
769        let forward: FeedSet = build(&feed(vec![
770            entry("alpha", same),
771            entry("beta", same),
772            entry("gamma", same),
773        ]));
774        let reversed: FeedSet = build(&feed(vec![
775            entry("gamma", same),
776            entry("beta", same),
777            entry("alpha", same),
778        ]));
779
780        assert_eq!(forward, reversed);
781    }
782
783    #[test]
784    fn the_feed_date_is_the_newest_entry_rather_than_the_clock() {
785        let newest: DateTime<Utc> = moment(2026, 1, 2);
786        let set: FeedSet = build(&feed(vec![
787            entry("old", moment(2020, 1, 1)),
788            entry("new", newest),
789        ]));
790
791        assert_eq!(newest, set.last_modified());
792        assert!(body(&set, ATOM_PATH).contains("2026-01-02T12:00:00Z"));
793    }
794
795    #[test]
796    fn each_document_gets_its_own_strong_validator() {
797        let set: FeedSet = build(&feed(vec![entry("first", moment(2024, 3, 10))]));
798
799        let etags: Vec<&str> = set.documents().iter().map(FeedDocument::etag).collect();
800
801        let expected: usize = 3;
802        let actual: usize = etags.len();
803        assert_eq!(expected, actual);
804
805        for etag in &etags {
806            assert!(etag.starts_with('"') && etag.ends_with('"'), "{etag}");
807        }
808        assert_ne!(etags[0], etags[1]);
809        assert_ne!(etags[1], etags[2]);
810    }
811
812    // ---- ordering and truncation
813
814    #[test]
815    fn the_newest_entries_are_kept_not_the_first_ones_supplied() {
816        let entries: Vec<FeedEntry> = (1..=MAX_FEED_ENTRIES + 5)
817            .map(|day| {
818                entry(
819                    &format!("post-{day:02}"),
820                    Utc.with_ymd_and_hms(2026, 1, u32::try_from(day).expect("small"), 12, 0, 0)
821                        .single()
822                        .expect("a real instant"),
823                )
824            })
825            .collect();
826
827        let set: FeedSet = build(&feed(entries));
828        let channel: rss::Channel =
829            rss::Channel::read_from(body(&set, RSS_PATH).as_bytes()).expect("valid RSS");
830
831        let expected_count: usize = MAX_FEED_ENTRIES;
832        let actual_count: usize = channel.items().len();
833        assert_eq!(expected_count, actual_count);
834
835        // Newest first: day 25 down to day 6.
836        assert_eq!(Some("Post post-25"), channel.items()[0].title());
837        assert_eq!(
838            Some("Post post-06"),
839            channel.items()[MAX_FEED_ENTRIES - 1].title()
840        );
841    }
842
843    #[test]
844    fn a_post_never_revised_reports_its_publication_date_as_its_update() {
845        let set: FeedSet = build(&feed(vec![entry("first", moment(2024, 3, 10))]));
846        let parsed: atom_syndication::Feed =
847            atom_syndication::Feed::read_from(body(&set, ATOM_PATH).as_bytes())
848                .expect("valid Atom");
849
850        let entry: &atom_syndication::Entry = &parsed.entries()[0];
851        assert_eq!(
852            entry.published().map(chrono::DateTime::to_rfc3339),
853            Some(entry.updated().to_rfc3339())
854        );
855    }
856
857    #[test]
858    fn a_revision_date_reaches_the_documents_that_can_express_one() {
859        let mut revised: FeedEntry = entry("first", moment(2024, 3, 10));
860        revised.modified = Some(moment(2026, 1, 2));
861
862        let set: FeedSet = build(&feed(vec![revised]));
863
864        assert!(body(&set, ATOM_PATH).contains("<updated>2026-01-02T12:00:00Z</updated>"));
865        assert!(body(&set, JSON_PATH).contains("\"date_modified\": \"2026-01-02T12:00:00Z\""));
866    }
867
868    // ---- content rewriting, end to end
869
870    #[test]
871    fn post_content_reaches_a_reader_with_urls_that_resolve_off_this_origin() {
872        let mut illustrated: FeedEntry = entry("first", moment(2026, 1, 2));
873        illustrated.content_html = Some(String::from(
874            "<p><img src=\"/static/image/blog/a.png\"><a href=\"/blog/other\">more</a></p>",
875        ));
876
877        let set: FeedSet = build(&feed(vec![illustrated]));
878        let channel: rss::Channel =
879            rss::Channel::read_from(body(&set, RSS_PATH).as_bytes()).expect("valid RSS");
880
881        let expected: &str = "<p><img src=\"https://www.example.com/static/image/blog/a.abc123.png\">\
882             <a href=\"https://www.example.com/blog/other\">more</a></p>";
883        let actual: &str = channel.items()[0].content().expect("content");
884        assert_eq!(expected, actual);
885    }
886
887    #[test]
888    fn a_declared_entry_image_is_hashed_absolutised_and_reaches_all_three_documents() {
889        let mut illustrated: FeedEntry = entry("first", moment(2026, 1, 2));
890        illustrated.image = Some(String::from("/static/image/blog/a.png"));
891
892        let set: FeedSet = build(&feed(vec![illustrated]));
893        let hashed: &str = "https://www.example.com/static/image/blog/a.abc123.png";
894
895        assert!(body(&set, RSS_PATH).contains(hashed));
896        assert!(body(&set, ATOM_PATH).contains(hashed));
897        assert!(body(&set, JSON_PATH).contains(hashed));
898        // Atom's enclosure needs a type; it is derived from the extension.
899        assert!(body(&set, ATOM_PATH).contains("type=\"image/png\""));
900    }
901
902    // ---- boot failures
903
904    #[test]
905    fn an_entry_missing_a_title_stops_the_deploy_rather_than_shipping_a_blank_row() {
906        let mut blank: FeedEntry = entry("first", moment(2026, 1, 2));
907        blank.title = String::from("   ");
908
909        let error: FeedError =
910            build_feeds(&site(), &feed(vec![blank]), resolver()).expect_err("refuses");
911        assert!(matches!(
912            error,
913            FeedError::IncompleteEntry { ref path, field } if path == "/blog/first" && field == "title"
914        ));
915    }
916
917    #[test]
918    fn an_entry_missing_a_summary_stops_the_deploy() {
919        let mut blank: FeedEntry = entry("first", moment(2026, 1, 2));
920        blank.summary = String::new();
921
922        let error: FeedError =
923            build_feeds(&site(), &feed(vec![blank]), resolver()).expect_err("refuses");
924        assert!(matches!(
925            error,
926            FeedError::IncompleteEntry { field, .. } if field == "summary"
927        ));
928    }
929
930    #[test]
931    fn an_unrooted_path_is_refused_because_it_cannot_become_a_stable_id() {
932        let mut floating: FeedEntry = entry("first", moment(2026, 1, 2));
933        floating.path = String::from("blog/first");
934
935        let error: FeedError =
936            build_feeds(&site(), &feed(vec![floating]), resolver()).expect_err("refuses");
937        assert!(matches!(
938            error,
939            FeedError::UnrootedPath { ref path } if path == "blog/first"
940        ));
941    }
942
943    #[test]
944    fn srcset_is_refused_rather_than_half_rewritten() {
945        let mut responsive: FeedEntry = entry("first", moment(2026, 1, 2));
946        responsive.content_html = Some(String::from(
947            "<img srcset=\"/a.png 1x, /b.png 2x\" src=\"/a.png\">",
948        ));
949
950        let error: FeedError =
951            build_feeds(&site(), &feed(vec![responsive]), resolver()).expect_err("refuses");
952        assert!(matches!(
953            error,
954            FeedError::SrcsetUnsupported { ref path } if path == "/blog/first"
955        ));
956    }
957
958    #[test]
959    fn a_declared_image_absent_from_the_manifest_stops_the_deploy() {
960        let mut illustrated: FeedEntry = entry("first", moment(2026, 1, 2));
961        illustrated.image = Some(String::from("/static/image/blog/gone.png"));
962
963        let error: FeedError =
964            build_feeds(&site(), &feed(vec![illustrated]), resolver()).expect_err("refuses");
965        assert!(matches!(
966            error,
967            FeedError::UnresolvedImage { ref image, .. }
968                if image == "/static/image/blog/gone.png"
969        ));
970    }
971
972    // ---- the `error!`-tier findings
973
974    #[test]
975    fn a_configured_feed_with_no_entries_is_reported() {
976        let findings: Vec<String> = diagnostics(&feed(Vec::new()), moment(2026, 1, 2));
977
978        let expected: usize = 1;
979        let actual: usize = findings.len();
980        assert_eq!(expected, actual);
981        assert!(findings[0].contains("no entries"));
982    }
983
984    #[test]
985    fn two_entries_sharing_a_path_are_reported_because_one_post_would_vanish() {
986        let declaration: Feed = feed(vec![
987            entry("first", moment(2024, 3, 10)),
988            entry("first", moment(2026, 1, 2)),
989        ]);
990
991        let findings: Vec<String> = diagnostics(&declaration, moment(2026, 6, 1));
992
993        let expected: usize = 1;
994        let actual: usize = findings.len();
995        assert_eq!(expected, actual);
996        assert!(findings[0].contains("/blog/first"));
997    }
998
999    #[test]
1000    fn a_post_dated_in_the_future_is_reported_because_it_pins_to_the_top_forever() {
1001        let declaration: Feed = feed(vec![entry("first", moment(2027, 1, 1))]);
1002        let findings: Vec<String> = diagnostics(&declaration, moment(2026, 1, 2));
1003
1004        let expected: usize = 1;
1005        let actual: usize = findings.len();
1006        assert_eq!(expected, actual);
1007        assert!(findings[0].contains("future"));
1008    }
1009
1010    #[test]
1011    fn a_revision_predating_publication_is_reported() {
1012        let mut backwards: FeedEntry = entry("first", moment(2026, 1, 2));
1013        backwards.modified = Some(moment(2020, 1, 1));
1014
1015        let findings: Vec<String> = diagnostics(&feed(vec![backwards]), moment(2026, 6, 1));
1016
1017        let expected: usize = 1;
1018        let actual: usize = findings.len();
1019        assert_eq!(expected, actual);
1020        assert!(findings[0].contains("modified before"));
1021    }
1022
1023    #[test]
1024    fn a_healthy_feed_reports_nothing() {
1025        let declaration: Feed = feed(vec![
1026            entry("first", moment(2024, 3, 10)),
1027            entry("second", moment(2026, 1, 2)),
1028        ]);
1029
1030        let expected: Vec<String> = Vec::new();
1031        let actual: Vec<String> = diagnostics(&declaration, moment(2026, 6, 1));
1032        assert_eq!(expected, actual);
1033    }
1034
1035    // ---- the shape of what gets served
1036
1037    #[test]
1038    fn all_three_documents_are_produced_at_their_fixed_paths_with_their_own_media_types() {
1039        let set: FeedSet = build(&feed(vec![entry("first", moment(2026, 1, 2))]));
1040
1041        let expected: Vec<(&str, &str)> = vec![
1042            (RSS_PATH, "application/rss+xml; charset=utf-8"),
1043            (ATOM_PATH, "application/atom+xml; charset=utf-8"),
1044            (JSON_PATH, "application/feed+json"),
1045        ];
1046        let actual: Vec<(&str, &str)> = set
1047            .documents()
1048            .iter()
1049            .map(|document| (document.path(), document.content_type()))
1050            .collect();
1051        assert_eq!(expected, actual);
1052    }
1053}