use super::parse::{FeedMeta, ItemRow, ParsedDocument, parse_feed_document};
struct Expect {
dialect: Option<&'static str>,
declared: Option<&'static str>,
notes: Option<&'static [&'static str]>,
failure: Option<(&'static str, &'static str)>,
items: usize,
golden: Option<(usize, &'static str)>,
}
impl Expect {
const fn parses(dialect: &'static str, declared: &'static str, items: usize) -> Self {
Expect {
dialect: Some(dialect),
declared: Some(declared),
notes: None,
failure: None,
items,
golden: None,
}
}
const fn fails(
stage: &'static str,
reason: &'static str,
declared: Option<&'static str>,
) -> Self {
Expect {
dialect: None,
declared,
notes: None,
failure: Some((stage, reason)),
items: 0,
golden: None,
}
}
const fn notes(mut self, notes: &'static [&'static str]) -> Self {
self.notes = Some(notes);
self
}
const fn golden(mut self, index: usize, path: &'static str) -> Self {
self.golden = Some((index, path));
self
}
}
struct Case {
case: &'static str,
fixture: &'static str,
content_type: Option<&'static str>,
expect: Expect,
}
const fn case(
case: &'static str,
fixture: &'static str,
content_type: Option<&'static str>,
expect: Expect,
) -> Case {
Case {
case,
fixture,
content_type,
expect,
}
}
const CORPUS: &[Case] = &[
case(
"rss2_wellformed",
"rss2_wellformed.xml",
Some("application/rss+xml"),
Expect::parses("rss-2.0", "rss-2.0", 2)
.notes(&[])
.golden(0, "rss2_wellformed_item0.md"),
),
case(
"rss2_missing_channel_description",
"rss2_missing_channel_description.xml",
Some("application/rss+xml"),
Expect::parses("rss-2.0", "rss-2.0", 1)
.notes(&["missing-required-field: channel/description"]),
),
case(
"rss1_rdf",
"rss1_rdf.xml",
Some("application/rss+xml"),
Expect::parses("rss-1.0", "rss-1.0", 1).notes(&[]),
),
case(
"atom10",
"atom10.xml",
Some("application/atom+xml"),
Expect::parses("atom", "atom-1.0", 2)
.notes(&[])
.golden(0, "atom10_item0.md"),
),
case(
"lying_content_type",
"atom10.xml",
Some("application/rss+xml"),
Expect::parses("atom", "atom-1.0", 2)
.notes(&["content-type-mismatch: served application/rss+xml, parsed atom"]),
),
case(
"atom03",
"atom03.xml",
Some("application/atom+xml"),
Expect::parses("atom", "atom-0.3", 0).notes(&[
"missing-required-field: feed/title",
"missing-required-field: feed/updated",
]),
),
case(
"jsonfeed_11",
"jsonfeed_11.json",
Some("application/feed+json"),
Expect::parses("json-feed-1.x", "json-feed-1.1", 2)
.notes(&[])
.golden(0, "jsonfeed_11_item0.md"),
),
case(
"encoding_latin1_mislabeled",
"encoding_latin1_mislabeled.xml",
Some("application/rss+xml"),
Expect::parses("rss-2.0", "rss-2.0", 1).notes(&["sanitation: reencoded-to-utf8"]),
),
case(
"control_chars",
"control_chars.xml",
Some("application/rss+xml"),
Expect::parses("rss-2.0", "rss-2.0", 1).notes(&["sanitation: stripped-control-chars"]),
),
case(
"naked_ampersand",
"naked_ampersand.xml",
Some("application/rss+xml"),
Expect::parses("rss-2.0", "rss-2.0", 1)
.notes(&["sanitation: escaped-naked-ampersands"])
.golden(0, "naked_ampersand_item0.md"),
),
case(
"billion_laughs",
"billion_laughs.xml",
Some("application/rss+xml"),
Expect::fails(
"refused-internal-dtd",
"internal DTD subset refused",
Some("rss-2.0"),
),
),
case(
"hostile_markup",
"hostile_markup.xml",
Some("application/rss+xml"),
Expect::parses("rss-2.0", "rss-2.0", 1)
.notes(&[])
.golden(0, "hostile_markup_item0.md"),
),
case(
"plaintext_typed_markup",
"plaintext_typed_markup.xml",
Some("application/atom+xml"),
Expect::parses("atom", "atom-1.0", 1).notes(&[]),
),
case(
"markdown_structures",
"markdown_structures.xml",
Some("application/rss+xml"),
Expect::parses("rss-2.0", "rss-2.0", 1)
.notes(&[])
.golden(0, "markdown_structures_item0.md"),
),
case(
"truncated",
"truncated.xml",
Some("application/rss+xml"),
Expect::fails("strict-parse", "unable to parse XML", Some("rss-2.0")),
),
case(
"empty_feed",
"empty_feed.xml",
Some("application/rss+xml"),
Expect::parses("rss-2.0", "rss-2.0", 0).notes(&[]),
),
case(
"guidless_items",
"guidless_items.xml",
Some("application/rss+xml"),
Expect::parses("rss-2.0", "rss-2.0", 2).notes(&["entries-without-identity: 1"]),
),
];
fn fixture_bytes(name: &str) -> &'static [u8] {
match name {
"rss2_wellformed.xml" => include_bytes!("fixtures/rss2_wellformed.xml"),
"rss2_missing_channel_description.xml" => {
include_bytes!("fixtures/rss2_missing_channel_description.xml")
}
"rss1_rdf.xml" => include_bytes!("fixtures/rss1_rdf.xml"),
"atom10.xml" => include_bytes!("fixtures/atom10.xml"),
"atom03.xml" => include_bytes!("fixtures/atom03.xml"),
"jsonfeed_11.json" => include_bytes!("fixtures/jsonfeed_11.json"),
"encoding_latin1_mislabeled.xml" => {
include_bytes!("fixtures/encoding_latin1_mislabeled.xml")
}
"control_chars.xml" => include_bytes!("fixtures/control_chars.xml"),
"naked_ampersand.xml" => include_bytes!("fixtures/naked_ampersand.xml"),
"billion_laughs.xml" => include_bytes!("fixtures/billion_laughs.xml"),
"hostile_markup.xml" => include_bytes!("fixtures/hostile_markup.xml"),
"plaintext_typed_markup.xml" => include_bytes!("fixtures/plaintext_typed_markup.xml"),
"markdown_structures.xml" => include_bytes!("fixtures/markdown_structures.xml"),
"truncated.xml" => include_bytes!("fixtures/truncated.xml"),
"empty_feed.xml" => include_bytes!("fixtures/empty_feed.xml"),
"guidless_items.xml" => include_bytes!("fixtures/guidless_items.xml"),
other => panic!("corpus fixture {other} is not in the include_bytes! table"),
}
}
fn golden_str(path: &str) -> &'static str {
let raw = match path {
"rss2_wellformed_item0.md" => include_str!("fixtures/golden/rss2_wellformed_item0.md"),
"atom10_item0.md" => include_str!("fixtures/golden/atom10_item0.md"),
"jsonfeed_11_item0.md" => include_str!("fixtures/golden/jsonfeed_11_item0.md"),
"naked_ampersand_item0.md" => include_str!("fixtures/golden/naked_ampersand_item0.md"),
"hostile_markup_item0.md" => include_str!("fixtures/golden/hostile_markup_item0.md"),
"markdown_structures_item0.md" => {
include_str!("fixtures/golden/markdown_structures_item0.md")
}
other => panic!("golden {other} is not in the include_str! table"),
};
raw.strip_suffix('\n').unwrap_or(raw)
}
fn parse_case(name: &str) -> ParsedDocument {
let case = CORPUS
.iter()
.find(|c| c.case == name)
.unwrap_or_else(|| panic!("no corpus case named {name}"));
parse_feed_document(fixture_bytes(case.fixture), case.content_type)
.unwrap_or_else(|e| panic!("corpus case {name} must parse: {e:?}"))
}
fn item(name: &str, index: usize) -> ItemRow {
let doc = parse_case(name);
doc.items
.get(index)
.unwrap_or_else(|| panic!("corpus case {name} has no item {index}"))
.clone()
}
#[test]
fn every_corpus_fixture_parses_or_degrades_visibly() {
for Case {
case,
fixture,
content_type,
expect,
} in CORPUS
{
let got = parse_feed_document(fixture_bytes(fixture), *content_type);
match (&got, expect.failure) {
(Err(failure), Some((stage, reason))) => {
assert_eq!(failure.stage, stage, "{case}: wrong failure stage");
assert!(
failure.reason.contains(reason),
"{case}: reason must name {reason:?}, got {:?}",
failure.reason
);
assert_eq!(
failure.dialect_declared.as_deref(),
expect.declared,
"{case}: the declared sniff answers even on a failed parse"
);
}
(Ok(doc), None) => {
assert_eq!(Some(doc.dialect), expect.dialect, "{case}: wrong dialect");
assert_eq!(
doc.dialect_declared.as_deref(),
expect.declared,
"{case}: wrong declared dialect"
);
assert_eq!(
Some(doc.conformance_notes.as_slice()),
expect
.notes
.map(|notes| notes.iter().map(|n| n.to_string()).collect::<Vec<_>>())
.as_deref(),
"{case}: the conformance notes are pinned exactly"
);
assert_eq!(doc.items.len(), expect.items, "{case}: wrong row count");
if let Some((index, golden)) = expect.golden {
assert_eq!(
doc.items[index].content.as_deref(),
Some(golden_str(golden)),
"{case}: golden drift against fixtures/golden/{golden}"
);
}
}
(Ok(doc), Some((stage, _))) => panic!(
"{case}: expected failure at {stage}, parsed as {} with {:?}",
doc.dialect, doc.conformance_notes
),
(Err(failure), None) => panic!("{case}: expected a parse, got {failure:?}"),
}
}
}
#[test]
fn no_fixture_or_golden_is_orphaned() {
const NOT_CORPUS_DOCUMENTS: &[&str] = &["bomb.xml.gz", "golden_probe.html"];
let dir = concat!(
env!("CARGO_MANIFEST_DIR"),
"/src/sources/providers/rss/fixtures"
);
let mut unreferenced = Vec::new();
for entry in std::fs::read_dir(dir).expect("the fixtures directory exists") {
let entry = entry.expect("readable directory entry");
let name = entry.file_name().to_string_lossy().into_owned();
if entry.file_type().expect("file type").is_dir() || NOT_CORPUS_DOCUMENTS.contains(&&*name)
{
continue;
}
if !CORPUS.iter().any(|c| c.fixture == name) {
unreferenced.push(name);
}
}
assert!(
unreferenced.is_empty(),
"fixtures committed but not in CORPUS: {unreferenced:?}"
);
let golden_dir = format!("{dir}/golden");
let mut unreferenced_goldens = Vec::new();
for entry in std::fs::read_dir(&golden_dir).expect("the golden directory exists") {
let name = entry.expect("readable directory entry").file_name();
let name = name.to_string_lossy().into_owned();
if !CORPUS
.iter()
.any(|c| c.expect.golden.is_some_and(|(_, g)| g == name))
{
unreferenced_goldens.push(name);
}
}
assert!(
unreferenced_goldens.is_empty(),
"goldens committed but not pinned by any corpus row: {unreferenced_goldens:?}"
);
}
#[test]
fn every_corpus_fixture_extracts_deterministically() {
for Case {
case,
fixture,
content_type,
..
} in CORPUS
{
let bytes = fixture_bytes(fixture);
let a = parse_feed_document(bytes, *content_type);
let b = parse_feed_document(bytes, *content_type);
match (a, b) {
(Ok(a), Ok(b)) => {
assert_eq!(a.items, b.items, "{case}: rows differ between parses");
assert_eq!(a.meta, b.meta, "{case}: feed metadata differs");
assert_eq!(
a.conformance_notes, b.conformance_notes,
"{case}: notes differ"
);
}
(Err(a), Err(b)) => assert_eq!(a, b, "{case}: failures differ between parses"),
(a, b) => panic!("{case}: one parse succeeded and the other did not: {a:?} / {b:?}"),
}
}
}
#[test]
fn rss2_wellformed_full_row_assertions() {
let doc = parse_case("rss2_wellformed");
assert_eq!(
doc.meta,
FeedMeta {
title: Some("Corpus Weekly".to_string()),
site_url: Some("https://corpus.example/".to_string()),
description: Some("A well-formed RSS 2.0 channel.".to_string()),
}
);
assert_eq!(
doc.items[0],
ItemRow {
guid: "tag:corpus.example,2026:post-1".to_string(),
title: Some("First post".to_string()),
link: Some("https://corpus.example/posts/1".to_string()),
author: Some("Ada Lovelace".to_string()),
published_ms: Some(1_784_541_600_000), updated_ms: Some(1_784_541_600_000),
content: Some(golden_str("rss2_wellformed_item0.md").to_string()),
summary: Some("Plain summary for the first post.".to_string()),
categories: vec!["rust".to_string(), "news".to_string()],
enclosure_url: Some("https://corpus.example/audio/1.mp3".to_string()),
enclosure_type: Some("audio/mpeg".to_string()),
enclosure_length: Some(12345),
extensions_json: None,
}
);
assert_eq!(
doc.items[1].published_ms,
Some(1_784_637_000_000), "an RFC-822 numeric offset must be applied, not ignored"
);
}
#[test]
fn atom10_full_row_assertions() {
let doc = parse_case("atom10");
assert_eq!(
doc.meta,
FeedMeta {
title: Some("Corpus Atom".to_string()),
site_url: Some("https://corpus.example/".to_string()),
description: Some("An Atom 1.0 feed.".to_string()),
}
);
assert_eq!(
doc.items[0],
ItemRow {
guid: "urn:uuid:8b3f0c3e-0001-4000-8000-000000000000".to_string(),
title: Some("Atom post".to_string()),
link: Some("https://corpus.example/atom/1".to_string()),
author: Some("Radia Perlman".to_string()),
published_ms: Some(1_784_541_600_000), updated_ms: Some(1_784_631_600_000), content: Some(golden_str("atom10_item0.md").to_string()),
summary: Some("Plain 3 < 4 summary, stored verbatim.".to_string()),
categories: vec!["rust".to_string(), "atom".to_string()],
enclosure_url: Some("https://corpus.example/audio/atom-1.mp3".to_string()),
enclosure_type: Some("audio/mpeg".to_string()),
enclosure_length: Some(4242),
extensions_json: None,
}
);
}
#[test]
fn jsonfeed_11_full_row_assertions() {
let doc = parse_case("jsonfeed_11");
assert_eq!(
doc.meta,
FeedMeta {
title: Some("Corpus JSON Feed".to_string()),
site_url: Some("https://corpus.example/".to_string()),
description: Some("A JSON Feed 1.1 document.".to_string()),
}
);
assert_eq!(
doc.items[0],
ItemRow {
guid: "corpus-json-1".to_string(),
title: Some("JSON post".to_string()),
link: Some("https://corpus.example/json/1".to_string()),
author: Some("Katherine Johnson".to_string()),
published_ms: Some(1_784_541_600_000), updated_ms: Some(1_784_631_600_000), content: Some(golden_str("jsonfeed_11_item0.md").to_string()),
summary: Some("Summary supplied explicitly.".to_string()),
categories: vec!["rust".to_string(), "json".to_string()],
enclosure_url: Some("https://corpus.example/audio/json-1.mp3".to_string()),
enclosure_type: Some("audio/mpeg".to_string()),
enclosure_length: Some(4096),
extensions_json: None,
}
);
assert_eq!(
doc.items[1].content.as_deref(),
Some("Stored verbatim: 3 < 4 & *not italic* <b>not bold</b>"),
"content_text must pass through byte-exact"
);
assert_eq!(doc.items[1].summary, None);
}
#[test]
fn rss1_rdf_full_row_assertions() {
let doc = parse_case("rss1_rdf");
assert_eq!(
doc.meta,
FeedMeta {
title: Some("Corpus RDF".to_string()),
site_url: Some("https://corpus.example/".to_string()),
description: Some("An RSS 1.0 (RDF) channel.".to_string()),
}
);
assert_eq!(
doc.items[0],
ItemRow {
guid: "https://corpus.example/rss1/1".to_string(),
title: Some("RDF post".to_string()),
link: Some("https://corpus.example/rss1/1".to_string()),
author: Some("Grace Hopper".to_string()),
published_ms: Some(1_784_541_600_000), updated_ms: None,
content: None,
summary: Some("Summary of the RDF post.".to_string()),
categories: vec![],
enclosure_url: None,
enclosure_type: None,
enclosure_length: None,
extensions_json: None,
}
);
}
#[test]
fn control_chars_fixture_keeps_the_text_the_control_byte_split() {
let doc = parse_case("control_chars");
assert_eq!(doc.items[0].title.as_deref(), Some("Interrupted title"));
assert_eq!(
doc.meta.description.as_deref(),
Some("A channel description."),
"the forbidden character is dropped, not replaced by a substitute"
);
}
#[test]
fn encoding_latin1_fixture_recovers_the_accented_characters() {
let doc = parse_case("encoding_latin1_mislabeled");
assert_eq!(
doc.items[0].title.as_deref(),
Some("Café au lait"),
"the 0xE9 byte must arrive as U+00E9, not as mojibake or U+FFFD"
);
assert_eq!(doc.meta.title.as_deref(), Some("Café Corpus"));
assert_eq!(
doc.items[0].summary.as_deref(),
Some("Résumé en latin-1."),
"two 0xE9 bytes in one field, both recovered"
);
}
#[test]
fn naked_ampersand_fixture_keeps_cdata_intact_and_resolves_nbsp() {
let content = item("naked_ampersand", 0).content.expect("content present");
assert!(
content.contains("3 && 4"),
"a CDATA `&&` must survive untouched: {content:?}"
);
assert!(
content.contains('©'),
"`©` inside CDATA is a valid reference the HTML pass decodes: {content:?}"
);
assert!(
content.contains("hard\u{A0}space"),
"` ` must arrive as U+00A0, not as literal text: {content:?}"
);
let summary = item("naked_ampersand", 0).summary.expect("summary present");
assert!(summary.contains('\u{A0}'), "{summary:?}");
assert_eq!(
parse_case("naked_ampersand").meta.title.as_deref(),
Some("Fish & Chips")
);
}
#[test]
fn hostile_markup_content_carries_no_tag_and_no_script_or_style_text() {
let content = item("hostile_markup", 0).content.expect("content present");
assert!(
!content.contains('<'),
"no `<` at all in this fixture's output: {content:?}"
);
assert!(
!content.contains("leak"),
"a script/style/handler/iframe body reached the stored content: {content:?}"
);
assert!(
content.contains("custom element text"),
"an unknown element's text is content and must be kept: {content:?}"
);
assert!(content.contains("javascript:void"), "{content:?}");
}
#[test]
fn plaintext_typed_content_is_stored_byte_exact_tags_included() {
let row = item("plaintext_typed_markup", 0);
let content = row.content.clone().expect("content present");
assert_eq!(
content, "<script>alert(1)</script>",
"a text-typed value is stored verbatim, tag-shaped text included"
);
let summary = row.summary.clone().expect("summary present");
assert_eq!(
summary, "<b>not bold</b> & not escaped further",
"the same holds for `summary`, and the `&` decodes to a bare `&` \
rather than being re-escaped"
);
}
#[test]
fn guidless_items_fall_back_to_their_link_for_identity() {
let doc = parse_case("guidless_items");
assert_eq!(
doc.items
.iter()
.map(|i| i.guid.as_str())
.collect::<Vec<_>>(),
vec![
"https://corpus.example/guidless/1",
"https://corpus.example/guidless/2"
],
);
assert!(
doc.conformance_notes
.iter()
.any(|n| n == "entries-without-identity: 1"),
"{:?}",
doc.conformance_notes
);
}
#[test]
fn billion_laughs_is_refused_by_the_guard_not_by_the_parser() {
let bytes = fixture_bytes("billion_laughs.xml");
let failure = parse_feed_document(bytes, Some("application/rss+xml"))
.expect_err("an internal DTD subset must be refused");
assert_eq!(failure.stage, "refused-internal-dtd");
assert!(
failure.reason.contains("entity-expansion guard"),
"the reason names the class: {}",
failure.reason
);
let raw = feed_rs::parser::parse(bytes).expect("feed-rs itself tolerates the document");
let title = raw
.title
.expect("the channel title is present, unexpanded")
.content;
assert_eq!(
title, "&lol3;",
"feed-rs no longer leaves DTD entities unexpanded; the guard is now load-bearing \
against real expansion and this test's premise needs revisiting"
);
}
#[test]
fn the_truncated_fixture_reason_echoes_no_document_prose() {
let failure = parse_feed_document(fixture_bytes("truncated.xml"), Some("application/rss+xml"))
.expect_err("a document cut off mid-tag must fail");
for prose in ["Truncated Corpus", "Cut off here", "corpus.example"] {
assert!(
!failure.reason.contains(prose),
"the failure reason echoed document content ({prose:?}): {}",
failure.reason
);
}
}
#[test]
fn the_gzip_bomb_fixture_is_still_a_bomb() {
const CAP: u64 = 5 * 1024 * 1024; let path = concat!(
env!("CARGO_MANIFEST_DIR"),
"/src/sources/providers/rss/fixtures/bomb.xml.gz"
);
let gz = std::fs::read(path).expect("bomb.xml.gz is committed");
assert_eq!(&gz[..2], &[0x1F, 0x8B], "gzip magic");
let isize_bytes: [u8; 4] = gz[gz.len() - 4..].try_into().expect("four footer bytes");
let inflated = u64::from(u32::from_le_bytes(isize_bytes));
assert!(
inflated > CAP,
"inflated to {inflated} bytes, which no longer exceeds the {CAP}-byte cap"
);
assert!(
(gz.len() as u64) < 64 * 1024,
"the wire form has grown to {} bytes; keep it small enough to commit",
gz.len()
);
}