use crate::{
ArticleScraper,
article::Article,
constants,
error::FullTextParserError,
test_server,
test_util::{assert_html_eq, check_fixture, check_text_fixture},
};
use super::{
FullTextParser,
config::{ConfigEntry, PageLink},
};
use crate::selector::Selector;
use dom_query::Document;
use reqwest::{Client, Url};
use test_log::test;
/// FTR fixtures in `resources/tests/ftr/`, with the URL each page is parsed as.
pub(crate) const FTR_FIXTURES: &[(&str, &str)] = &[
("golem", "https://www.golem.de/"),
("phoronix", "https://www.phoronix.com/"),
("youtube", "https://www.youtube.com/"),
("hardwareluxx", "https://www.hardwareluxx.de/"),
("heise-1", "https://www.heise.de/"),
("spiegel-1", "https://www.spiegel.de/"),
(
"wienerzeitung",
"https://www.wienerzeitung.at/a/die-moeglichkeit-einer-wiese",
),
(
"lasemainedelallier",
"https://www.lasemainedelallier.fr/lapalisse-nature-cimetiere/",
),
(
"pluralistic",
"https://pluralistic.net/2023/07/31/seize-the-means-of-computation/",
),
(
"arstechnica",
"https://arstechnica.com/tech-policy/2012/02/gigabit-internet-for-80-the-unlikely-success-of-californias-sonicnet/",
),
(
"esquire",
"https://www.esquire.com/news-politics/a7922/price-is-right-perfect-bid-0810/",
),
("lwn", "https://lwn.net/Articles/668318/"),
(
"mercatornet",
"https://www.mercatornet.com/hosing_down_the_biggest_moral_panic_in_canadian_history",
),
(
"bbc",
"https://www.bbc.com/worklife/article/20200121-why-procrastination-is-about-managing-emotions-not-time",
),
(
"fr",
"https://www.fr.de/frankfurt/die-nfl-kommt-nach-frankfurt-91329620.html",
),
(
"moto-net-multipage",
"https://www.moto-net.com/article/comparatif-superbike-2018-aprilia-rsv4-rf-vs-bmw-s1000rr-vs-ducati-panigale-v4-s-1-comparo-sbk-2018-page-1-les-nouvelles-du-vieux-continent.html",
),
(
"jbpress-multipage",
"https://jbpress.ismedia.jp/articles/-/92430",
),
(
"marcobehler",
"https://www.marcobehler.com/guides/java-microservices-a-practical-guide",
),
(
"commondreams",
"https://www.commondreams.org/news/2019/08/19/17-million-turn-out-hong-kong-demonstration-vow-keep-fighting-protests-enter-11th",
),
(
"legrandcontinent",
"https://legrandcontinent.eu/fr/2023/08/03/jai-eu-des-doutes-deux-conversations-avec-robert-oppenheimer/",
),
(
"scripting",
"http://scripting.com/stories/2011/07/08/yeahImStillYawning.html",
),
];
pub(crate) fn fixture_path(name: &str, file: &str) -> String {
format!("./resources/tests/ftr/{name}/{file}")
}
/// File name of page `index` (0-based) of a fixture: `source.html`, `source-2.html`, …
fn source_file(index: usize) -> String {
if index == 0 {
"source.html".into()
} else {
format!("source-{}.html", index + 1)
}
}
fn fixture_url(name: &str) -> Result<Url, String> {
FTR_FIXTURES
.iter()
.find(|(fixture, _)| *fixture == name)
.map(|(_, url)| Url::parse(url).unwrap())
.ok_or_else(|| format!("{name} is missing in FTR_FIXTURES"))
}
/// Runs the FTR pipeline on `resources/tests/ftr/{name}/source.html` (plus `source-2.html`, …
/// for articles spanning several pages). Shared by the tests and the fixture review report.
pub(crate) async fn ftr_fixture_output(name: &str) -> Result<Article, String> {
let url = fixture_url(name)?;
let mut pages = Vec::new();
while let Ok(html) = std::fs::read_to_string(fixture_path(name, &source_file(pages.len()))) {
pages.push(html);
}
if pages.is_empty() {
return Err(format!(
"Failed to read {}",
fixture_path(name, "source.html")
));
}
let parser = FullTextParser::new(None).await;
parser
.parse_offline(pages, None, Some(url))
.map_err(|error| error.to_string())
}
/// Downloads the source of a new FTR fixture the way `ArticleScraper::parse` does: with the
/// site config's HTTP headers, its encoding handling, and following `single_page_link` /
/// `next_page_link`. Register the fixture in `FTR_FIXTURES` first, then run
///
/// ```sh
/// FIXTURE=<name> cargo test -p article_scraper --all-features --lib -- --ignored fetch_ftr_fixture
/// ```
///
/// and generate `expected.html` with `UPDATE_FIXTURES=1`.
#[test(tokio::test)]
#[ignore = "downloads a fixture source from the web; run on request"]
async fn fetch_ftr_fixture() {
let name = std::env::var("FIXTURE").expect("set FIXTURE=<name from FTR_FIXTURES>");
let url = fixture_url(&name).unwrap();
let client = Client::builder()
.user_agent("Mozilla/5.0 (X11; Linux x86_64; rv:140.0) Gecko/20100101 Firefox/140.0")
.build()
.unwrap();
let parser = FullTextParser::new(None).await;
let config = parser.get_grabber_config(&url);
let global_config = parser.config_files.get("global.txt").unwrap();
let html = FullTextParser::download(&url, &client, config, global_config)
.await
.unwrap();
let pages = parser
.download_all_pages(html, &client, config, global_config, &url)
.await
.unwrap();
std::fs::create_dir_all(fixture_path(&name, "")).unwrap();
for (index, page) in pages.iter().enumerate() {
std::fs::write(fixture_path(&name, &source_file(index)), page).unwrap();
}
println!("{name}: {} page(s) written", pages.len());
}
/// Title, author, date, thumbnail and the native ad flag of an article, one per line, as
/// stored in `expected-metadata.txt`. Strings are quoted so surrounding whitespace shows.
pub(crate) fn metadata_snapshot(article: &Article) -> String {
format!(
"title: {:?}\nauthor: {:?}\ndate: {:?}\nthumbnail: {:?}\nnative_ad: {}\n",
article.title,
article.author,
article.date.map(|date| date.to_rfc3339()),
article.thumbnail_url,
article.native_ad,
)
}
async fn run_test(name: &str) {
let article = ftr_fixture_output(name).await.unwrap();
check_text_fixture(
&fixture_path(name, "expected-metadata.txt"),
&metadata_snapshot(&article),
);
check_fixture(&fixture_path(name, "expected.html"), &article.html.unwrap());
}
#[test(tokio::test)]
async fn golem() {
run_test("golem").await
}
#[test(tokio::test)]
async fn phoronix() {
run_test("phoronix").await
}
#[test(tokio::test)]
async fn youtube() {
run_test("youtube").await
}
#[test(tokio::test)]
async fn hardwareluxx() {
run_test("hardwareluxx").await
}
#[test(tokio::test)]
async fn heise_1() {
run_test("heise-1").await
}
#[test(tokio::test)]
async fn spiegel_1() {
run_test("spiegel-1").await
}
#[test(tokio::test)]
async fn wienerzeitung() {
run_test("wienerzeitung").await
}
#[test(tokio::test)]
async fn lasemainedelallier() {
run_test("lasemainedelallier").await
}
#[test(tokio::test)]
async fn pluralistic() {
run_test("pluralistic").await
}
#[test(tokio::test)]
async fn arstechnica() {
run_test("arstechnica").await
}
#[test(tokio::test)]
async fn esquire() {
run_test("esquire").await
}
#[test(tokio::test)]
async fn lwn() {
run_test("lwn").await
}
#[test(tokio::test)]
async fn mercatornet() {
run_test("mercatornet").await
}
#[test(tokio::test)]
async fn bbc() {
run_test("bbc").await
}
#[test(tokio::test)]
async fn fr() {
run_test("fr").await
}
#[test(tokio::test)]
async fn moto_net_multipage() {
run_test("moto-net-multipage").await
}
#[test(tokio::test)]
async fn jbpress_multipage() {
run_test("jbpress-multipage").await
}
#[test(tokio::test)]
async fn marcobehler() {
run_test("marcobehler").await
}
#[test(tokio::test)]
async fn commondreams() {
run_test("commondreams").await
}
#[test(tokio::test)]
async fn legrandcontinent() {
run_test("legrandcontinent").await
}
#[test(tokio::test)]
async fn scripting() {
run_test("scripting").await
}
#[test(tokio::test)]
#[ignore = "downloads content from the web"]
async fn encoding_windows_1252() {
let url = url::Url::parse("https://www.aerzteblatt.de/nachrichten/139511/Scholz-zuversichtlich-mit-Blick-auf-Coronasituation-im-Winter").unwrap();
let html = FullTextParser::download(&url, &Client::new(), None, &ConfigEntry::default())
.await
.unwrap();
assert!(html.contains("Bund-Länder-Konferenz"));
}
#[test(tokio::test)]
async fn unwrap_noscript_images() {
let html = r#"
<p>Lorem ipsum dolor sit amet,
<span class="lazyload">
<img src="foto-m0101.jpg" alt="image description">
<noscript><img src="foto-m0102.jpg" alt="image description"></noscript>
</span>
consectetur adipiscing elit.
</p>
"#;
let expected = r#"<!DOCTYPE html PUBLIC "-//W3C//DTD HTML 4.0 Transitional//EN" "http://www.w3.org/TR/REC-html40/loose.dtd">
<html><body>
<p>Lorem ipsum dolor sit amet,
<span class="lazyload">
<img src="foto-m0102.jpg" alt="image description" data-old-src="foto-m0101.jpg">
</span>
consectetur adipiscing elit.
</p>
</body></html>
"#;
let empty_config = ConfigEntry::default();
let document = crate::FullTextParser::parse_html(html, None, &empty_config);
crate::FullTextParser::unwrap_noscript_images(&document);
let res = document.html().to_string();
assert_html_eq(expected, &res);
}
#[test(tokio::test)]
async fn crash_sample() {
let html = r#"
<html>
<body>
<article>
<div>
<article>
</article>
</div>
</article>
</body>
</html>
"#
.to_string();
let config = ConfigEntry {
title: vec![Selector::xpath("//h1").unwrap()],
body: vec![Selector::xpath("//article").unwrap()],
..Default::default()
};
let parser = FullTextParser::new(None).await;
let article = parser
.parse_offline(vec![html], Some(&config), None)
.unwrap();
let _content = article.html.unwrap();
}
#[test(tokio::test)]
async fn unwrap_noscript_images_2() {
let html = r#"
<picture class="c-lead-image__image">
<source srcset="https://cdn.citylab.com/media/img/citylab/2019/04/mr1/300.jpg?mod=1556645448" media="(max-width: 575px)" />
<img class="c-lead-image__img" srcset="https://cdn.citylab.com/media/img/citylab/2019/04/mr1/300.jpg?mod=1556645448" alt="" itemprop="contentUrl" onload="performance.mark("citylab_lead_image_loaded")" />
<noscript>
<img class="c-lead-image__img" src="https://cdn.citylab.com/media/img/citylab/2019/04/mr1/300.jpg?mod=1556645448" alt="" />
</noscript>
</picture>
"#;
let expected = r#"<!DOCTYPE html PUBLIC "-//W3C//DTD HTML 4.0 Transitional//EN" "http://www.w3.org/TR/REC-html40/loose.dtd">
<html><body>
<picture class="c-lead-image__image">
<source srcset="https://cdn.citylab.com/media/img/citylab/2019/04/mr1/300.jpg?mod=1556645448" media="(max-width: 575px)"></source>
<img class="c-lead-image__img" src="https://cdn.citylab.com/media/img/citylab/2019/04/mr1/300.jpg?mod=1556645448" alt="" srcset="https://cdn.citylab.com/media/img/citylab/2019/04/mr1/300.jpg?mod=1556645448">
</picture>
</body></html>
"#;
let empty_config = ConfigEntry::default();
let document = crate::FullTextParser::parse_html(html, None, &empty_config);
crate::FullTextParser::unwrap_noscript_images(&document);
let res = document.html().to_string();
assert_html_eq(expected, &res);
}
#[test]
fn extract_thumbnail_golem() {
let html = r#"
<img src="https://www.golem.de/2306/175204-387164-387163_rc.jpg" width="140" height="140" loading="lazy" />Im staubigen
Utah sind die Fossilien eines urzeitlichen Meeresreptils entdeckt worden. Nun haben Forscher eine Studie dazu
herausgebracht. (<a href="https://www.golem.de/specials/fortschritt/" rel="noopener noreferrer" target="_blank"
referrerpolicy="no-referrer">Fortschritt</a>, <a href="https://www.golem.de/specials/wissenschaft/"
rel="noopener noreferrer" target="_blank" referrerpolicy="no-referrer">Wissenschaft</a>)
"#;
let doc = Document::from(html);
let thumb = FullTextParser::check_for_thumbnail(&doc, None, None).unwrap();
assert_eq!(
thumb,
"https://www.golem.de/2306/175204-387164-387163_rc.jpg"
)
}
#[test]
fn extract_thumbnail_spiegel() {
let html = r#"
<article><section data-article-el="body">
<div data-area="top_element>image">
<figure>
<div data-sara-component="{"id":"a4573666-f15e-4290-8c73-a0c6cd4ad3b2","name":"image","title":"\u003cp\u003eGrünenpolitiker Hofreiter: »Unternehmen werden in großem Umfang erpresst, unter Wert ihre Betriebe zu verkaufen«\u003c/p\u003e","type":"media"}">
<picture>
<source srcset="https://cdn.prod.www.spiegel.de/images/a4573666-f15e-4290-8c73-a0c6cd4ad3b2_w948_r1.778_fpx29.99_fpy44.98.webp 948w, https://cdn.prod.www.spiegel.de/images/a4573666-f15e-4290-8c73-a0c6cd4ad3b2_w520_r1.778_fpx29.99_fpy44.98.webp 520w" sizes="(max-width: 519px) 100vw, (min-width: 520px) and (max-width: 719px) 520px, (min-width: 720px) and (max-width: 919px) 100vw, (min-width: 920px) and (max-width: 1011px) 920px, (min-width: 1012px) 948px" type="image/webp">
<img data-image-el="img" src="https://cdn.prod.www.spiegel.de/images/a4573666-f15e-4290-8c73-a0c6cd4ad3b2_w948_r1.778_fpx29.99_fpy44.98.jpg" width="948" height="533" title="Grünenpolitiker Hofreiter: »Unternehmen werden in großem Umfang erpresst, unter Wert ihre Betriebe zu verkaufen«" alt="Grünenpolitiker Hofreiter: »Unternehmen werden in großem Umfang erpresst, unter Wert ihre Betriebe zu verkaufen«" data-image-animation-origin="91086ec8-2db6-4a72-be06-66c9e5db9058"/>
</source></picture>
</div>
<figcaption>
<p>Grünenpolitiker Hofreiter: »Unternehmen werden in großem Umfang erpresst, unter Wert ihre Betriebe zu verkaufen«</p>
<span>
Foto: IMAGO / IMAGO/Political-Moments
</span>
</figcaption>
</figure>
</div>
<div data-area="body">
<div data-pos="1" data-sara-click-el="body_element" data-area="text">
<p>Der Töne aus Berlin in Richtung Budapest werden giftiger. Der Grünen-Europapolitiker <a href="https://www.spiegel.de/thema/anton_hofreiter/" data-link-flag="spon" target="_blank">Anton Hofreiter</a> wirft der ungarischen Regierung vor, deutsche Unternehmen mit »Mafiamethoden« zum Verkauf ihres <a href="https://www.spiegel.de/thema/ungarn/" data-link-flag="spon" target="_blank">Ungarn</a>-Geschäfts zu bringen. »Ungarn bewegt sich von einer autoritären Herrschaft in Richtung eines Mafiastaats«, sagte Hofreiter in Brüssel. »Unternehmen werden in großem Umfang erpresst, unter Wert ihre Betriebe zu verkaufen.«</p>
</div>
<div data-area="text" data-sara-click-el="body_element" data-pos="3">
<p>Aus der deutschen Wirtschaft gebe es Klagen über zahlreiche Fälle, in denen Firmen »mit illegalen Methoden« vom Markt gedrängt worden seien oder entsprechende Versuche stattgefunden hätten.</p><p>Während Ungarns Regierungschef <a href="https://www.spiegel.de/thema/viktor_orban/" data-link-flag="spon" target="_blank">Viktor Orbán</a> deutsche Autohersteller weiterhin mit niedrigen Steuern und wenig Bürokratie verwöhne, bekämen andere Firmen die Folgen von Orbáns Strategie der Nationalisierung von als strategisch wichtig geltenden Branchen zu spüren. Selbst Großunternehmen wie Lidl oder die Telekom würden inzwischen »massiv unter Druck gesetzt«, so Hofreiter.</p>
</div>
<div data-sara-click-el="body_element" data-area="image" data-pos="5">
<figure>
<div data-sara-component="{"id":"cce7cbb0-2a7e-449d-a24e-6a24a73108b2","name":"image","title":"Ungarns Regierungschef Viktor Orbán","type":"media"}">
<picture>
<source data-srcset="https://cdn.prod.www.spiegel.de/images/cce7cbb0-2a7e-449d-a24e-6a24a73108b2_w718_r1.5001583782071588_fpx53.97_fpy44.98.jpg 718w, https://cdn.prod.www.spiegel.de/images/cce7cbb0-2a7e-449d-a24e-6a24a73108b2_w488_r1.5001583782071588_fpx53.97_fpy44.98.jpg 488w, https://cdn.prod.www.spiegel.de/images/cce7cbb0-2a7e-449d-a24e-6a24a73108b2_w616_r1.5001583782071588_fpx53.97_fpy44.98.jpg 616w" data-sizes="(max-width: 487px) 100vw, (min-width: 488px) and (max-width: 719px) 488px, (min-width: 720px) and (max-width: 1011px) 718px, (min-width: 1012px) 616px">
<img data-image-el="img" data-src-disabled="data:image/svg+xml,%3Csvg xmlns='http://www.w3.org/2000/svg' viewBox='0 0 718 479' width='718' height='479' %3E%3C/svg%3E" src="https://cdn.prod.www.spiegel.de/images/cce7cbb0-2a7e-449d-a24e-6a24a73108b2_w718_r1.5001583782071588_fpx53.97_fpy44.98.jpg" width="718" height="479" title="Ungarns Regierungschef Viktor Orbán" alt="Ungarns Regierungschef Viktor Orbán" data-image-animation-origin="38ca0348-88da-4a3b-8a9d-f1c059e79c77"/>
</source></picture>
</div>
</figure>
<figcaption>
<p>Ungarns Regierungschef Viktor Orbán</p>
<span>
Foto: IMAGO/Vaclav Salek / IMAGO/CTK Photo
</span>
</figcaption></div>
<div data-pos="6" data-sara-click-el="body_element" data-area="text">
<p>Die Masche des Systems Orbán ist die immer gleiche, wie Unternehmen und Politiker <a href="https://www.spiegel.de/wirtschaft/ungarn-wie-viktor-orban-deutsche-unternehmen-aus-dem-land-mobbt-a-2b345c3e-5223-4ae6-97bc-ba1718f20907" target="_blank">schon seit Monaten beklagen</a>: Die Regierung macht die Unternehmen erst Schikanen mürbe und unterbreitet dann wieder und wieder Kaufangebote. Die Firmen würden so gedrängt, ihre ungarischen Aktivitäten an Günstlinge Orbáns zu verkaufen – zwar nicht zu ruinösen Schleuderpreisen, aber üblicherweise für nur etwa 70 bis 80 Prozent des Marktwerts, sagt Hofreiter.</p>
</div>
<div data-pos="8" data-sara-click-el="body_element" data-area="text">
<p>In Ungarn gehe es nicht mehr nur um die bereits weit fortgeschrittene Zerstörung des Rechtsstaats – »sondern inzwischen auch eindeutig um das Funktionieren des Binnenmarkts« der EU. »Der klassische ökonomische Teil des Binnenmarkts wird angegriffen.«</p><p>Die Kommission hat wegen Ungarns Rechtsstaatsverstößen bereits <a rel="noopener noreferrer" href="https://www.spiegel.de/ausland/eu-friert-saemtliche-strukturfoerdermittel-fuer-ungarn-ein-a-739da28a-243a-4fe1-af8c-a515fb8bf967" target="_blank">Milliardenzahlungen an das Land eingefroren</a>. Das aber genüge nicht mehr, sagt Hofreiter – und fordert von der Kommission deshalb, neue Sanktionsinstrumente zu entwickeln: »Man muss sich Mechanismen zum Schutz des Binnenmarkts überlegen.«</p><h3>Ungarns Außenminister beklagt »politisch motivierte Kampagne«</h3><p>Der grüne Europaabgeordnete Daniel Freund verlangt außerdem eine Beschleunigung laufender und künftiger Verfahren gegen Ungarn wegen der Verletzung der EU-Verträge. »Wenn eine Firma wegen eines Regierungsdekrets Monat für Monat Millionen an Steuern bezahlen muss, kann sie nicht Jahre warten, ehe ein solches Verfahren abgeschlossen ist.«</p>
</div>
<div data-pos="10" data-area="text" data-sara-click-el="body_element">
<p>Ungarns Außen- und Handelsminister Péter Szijjártó <a href="https://abouthungary.hu/news-in-brief/fm-a-politically-motivated-campaign-is-underway-against-hungary-over-german-investments" target="_blank">bezeichnete </a> die Vorwürfe kürzlich als »politisch motivierte Kampagne« und »emotionale Erpressung«. Seit 2014 habe Budapest 183 deutsche Unternehmen gefördert. Insgesamt würden rund 6000 deutsche Firmen in Ungarn etwa 300.000 Menschen beschäftigen.</p>
</div>
<div data-sara-click-el="body_element" data-pos="12" data-area="text">
<p>Orbáns Politik stößt nicht nur bei den Grünen auf Kritik, sondern auch bei den deutschen Unionsparteien. Bis März 2021 waren sie gemeinsam mit Orbáns Fidesz-Partei in der Europäischen Volkspartei; jahrelang hofierten sie den Autokraten aus Budapest.</p><p><a href="https://www.spiegel.de/thema/monika_hohlmeier/" data-link-flag="spon" target="_blank">Monika Hohlmeier</a> (CSU) etwa, Vorsitzende des Haushaltskontrollausschusses im EU-Parlament, sieht in Orbán mittlerweile »einen Mann mit kleptokratischen Zügen«, in dessen System »rechtsstaatliche Prinzipien mit Füßen getreten werden«. Erfolgreiche ausländische Unternehmer müssten in Ungarn damit rechnen, »dass ein Oligarch auftaucht, der sich deine Firma unter den Nagel reißen will«.</p>
</div>
</div>
</section></article>
"#;
let doc = Document::from(html);
let thumb = FullTextParser::check_for_thumbnail(&doc, None, None).unwrap();
assert_eq!(
thumb,
"https://cdn.prod.www.spiegel.de/images/a4573666-f15e-4290-8c73-a0c6cd4ad3b2_w948_r1.778_fpx29.99_fpy44.98.jpg"
)
}
#[test]
fn extract_thumbnail_a_chacon() {
let html = std::fs::read_to_string("./resources/tests/thumbnails/a-chacon.html")
.expect("Failed to read source HTML");
let doc = Document::from(html.as_str());
let thumb = FullTextParser::check_for_thumbnail(&doc, None, None).unwrap();
assert_eq!(
thumb,
"https://a-chacon.com/assets/images/rails8-poc-api-auth.webp"
)
}
#[test(tokio::test)]
#[ignore = "downloads content from the web"]
async fn hardwareluxx_failed_http() {
let url = Url::parse("https://www.hardwareluxx.de/index.php/news/software/spiele/68138-aus-fuer-the-outer-worlds-obsidian-stampft-die-spielreihe-ein.html").unwrap();
let client = Client::builder().build().unwrap();
let scraper = ArticleScraper::new(None).await;
#[cfg(not(feature = "image-downloader"))]
let _ = scraper.parse(&url, &client).await.unwrap();
#[cfg(feature = "image-downloader")]
let _ = scraper.parse(&url, &client, false, None).await.unwrap();
}
#[test(tokio::test)]
async fn strip_instapaper_ignore() {
let html = r#"<html><body><article>
<p>keep</p>
<div class="instapaper_ignore">ignored 1</div>
<div class="share instapaper_ignore">ignored 2</div>
<p class="entry-unrelated">ignored 3</p>
<p class="not-entry-unrelated">keep too</p>
</article></body></html>"#
.to_string();
let config = ConfigEntry {
body: vec![Selector::xpath("//article").unwrap()],
..Default::default()
};
let parser = FullTextParser::new(None).await;
let article = parser
.parse_offline(vec![html], Some(&config), None)
.unwrap();
let content = article.html.unwrap();
assert!(content.contains("keep too"), "{content}");
assert!(!content.contains("ignored"), "{content}");
}
#[test]
fn fix_iframe_size_without_empty_wrapper() {
let html = r#"<div id="video"><iframe src="https://www.youtube.com/embed/abc"></iframe></div>"#;
let document = Document::from(html);
FullTextParser::fix_iframe_size(&document, "youtube.com");
let content = document.html().to_string();
let iframe = document.select("iframe");
assert_eq!(iframe.attr("width").as_deref(), Some("480"), "{content}");
assert_eq!(iframe.attr("height").as_deref(), Some("360"), "{content}");
// No (empty) wrapper element may be appended to the iframe's container. Readability
// scores and converts the container differently when it has a block child.
assert_eq!(document.select("#video > *").length(), 1, "{content}");
}
#[test]
fn simplify_nested_elements_keeps_text_next_to_single_div() {
let document = Document::from(
r#"<html><body><article><div>Intro text<div><p>para</p></div></div></article></body></html>"#,
);
let article = document.select("article").nodes()[0];
FullTextParser::post_process_document(&article).unwrap();
let content = document.select("article").html().to_string();
assert!(content.contains("Intro text"), "{content}");
assert!(content.contains("<p>para</p>"), "{content}");
}
#[test]
fn remove_single_cell_tables_nested() {
let document = Document::from(
r#"<html><body><article><table><tr><td><table><tr><td><b>inner</b></td></tr></table></td></tr></table></article></body></html>"#,
);
let article = document.select("article").nodes()[0];
FullTextParser::remove_single_cell_tables(&article);
let content = document.select("article").html().to_string();
assert_eq!(content, "<article><div><p><b>inner</b></p></div></article>");
// the table is the root itself
let document = Document::from(
r#"<html><body><table><tr><td><table><tr><td>inner</td></tr></table></td></tr></table></body></html>"#,
);
let table = document.select("body > table").nodes()[0];
FullTextParser::remove_single_cell_tables(&table);
let content = document.select("body").html().to_string();
assert_eq!(content, "<body><div><p>inner</p></div></body>");
}
#[test]
fn fix_urls_srcset_without_src() {
let document = Document::from(
r#"<html><body>
<img id="no-src" srcset="a.jpg 1x, https://cdn.example.com/b.jpg 2x">
<picture><source id="source" srcset="c.webp 480w, d.webp 800w"><img id="img" src="e.jpg"></picture>
</body></html>"#,
);
let url = Url::parse("https://example.com/article/").unwrap();
FullTextParser::fix_urls(&document, &url);
let attr = |id: &str, name: &str| document.select(&format!("#{id}")).attr(name);
assert_eq!(
attr("no-src", "srcset").as_deref(),
Some("https://example.com/article/a.jpg 1x, https://cdn.example.com/b.jpg 2x")
);
assert_eq!(
attr("source", "srcset").as_deref(),
Some("https://example.com/article/c.webp 480w, https://example.com/article/d.webp 800w")
);
assert_eq!(
attr("img", "src").as_deref(),
Some("https://example.com/article/e.jpg")
);
}
#[test]
fn fix_lazy_images_base64_placeholders() {
let document = Document::from(
r#"<html><body>
<img id="no-payload" src="data:image/png;base64" data-src="a.jpg">
<img id="placeholder" src="data:image/gif;base64,R0lGODlhAQABAAAAACw=" data-src="b.jpg">
<img id="url" src="https://example.com/base64/c.jpg" data-src="d.jpg">
</body></html>"#,
);
FullTextParser::fix_lazy_images(&document);
let src = |id: &str| document.select(&format!("#{id}")).attr("src");
// not a data URL (no comma), so nothing to measure; used to underflow
assert_eq!(src("no-payload").as_deref(), Some("data:image/png;base64"));
assert_eq!(src("placeholder").as_deref(), Some("b.jpg"));
// "base64" in a normal URL isn't a data URL
assert_eq!(
src("url").as_deref(),
Some("https://example.com/base64/c.jpg")
);
}
#[test]
fn lazy_load_attr_that_is_no_css_identifier() {
let document = Document::from(r#"<html><body><img data-x="a.jpg" 1x="b.jpg"></body></html>"#);
FullTextParser::apply_lazy_load_attr(&document, "1x");
assert_eq!(document.select("img").attr("src").as_deref(), Some("b.jpg"));
}
#[test]
fn follows_page_links_only_with_rules() {
let none = ConfigEntry::default();
let link = || vec![PageLink::from(Selector::xpath("//a[@rel='next']").unwrap())];
let next_page = ConfigEntry {
next_page_link: link(),
..Default::default()
};
let single_page = ConfigEntry {
single_page_link: link(),
..Default::default()
};
let autodetect = |value| ConfigEntry {
autodetect_next_page: Some(value),
..Default::default()
};
assert!(!FullTextParser::follows_page_links(None, &none));
assert!(!FullTextParser::follows_page_links(Some(&none), &none));
assert!(FullTextParser::follows_page_links(Some(&next_page), &none));
assert!(FullTextParser::follows_page_links(
Some(&single_page),
&none
));
assert!(FullTextParser::follows_page_links(None, &next_page));
assert!(FullTextParser::follows_page_links(None, &autodetect(true)));
// the site config's `autodetect_next_page: no` wins over the global one
assert!(!FullTextParser::follows_page_links(
Some(&autodetect(false)),
&autodetect(true)
));
}
#[test(tokio::test)]
async fn next_page_link_falls_back_to_global_rule() {
let html = r#"<html><body><a rel="next" href="/page/2">next</a></body></html>"#;
let document = Document::from(html);
let article_url = Url::parse("https://example.com/page/1").unwrap();
let site_config = ConfigEntry {
next_page_link: vec![
Selector::xpath("//a[@class='does-not-exist']")
.unwrap()
.into(),
],
..Default::default()
};
let global_config = ConfigEntry {
next_page_link: vec![Selector::xpath("//a[@rel='next']").unwrap().into()],
..Default::default()
};
let parser = FullTextParser::new(None).await;
let expected = Url::parse("https://example.com/page/2").unwrap();
for config in [Some(&site_config), Some(&ConfigEntry::default()), None] {
let next = parser.check_for_next_page(&document, config, &global_config, &article_url);
assert_eq!(next.as_ref(), Some(&expected));
}
}
#[test(tokio::test)]
async fn autodetect_next_page() {
let html = r#"<html><head><link rel="next" href="/page/2"></head><body></body></html>"#;
let document = Document::from(html);
let article_url = Url::parse("https://example.com/page/1").unwrap();
let parser = FullTextParser::new(None).await;
let global_config = ConfigEntry::default();
let next = |autodetect: Option<bool>| {
let config = ConfigEntry {
autodetect_next_page: autodetect,
..Default::default()
};
parser
.check_for_next_page(&document, Some(&config), &global_config, &article_url)
.map(String::from)
};
assert_eq!(next(None), None);
assert_eq!(next(Some(false)), None);
assert_eq!(
next(Some(true)).as_deref(),
Some("https://example.com/page/2")
);
}
#[test(tokio::test)]
async fn page_link_rules_in_config_order() {
let html = r#"<html><body>
<a class="single" href="/all">all on one page</a>
<a rel="next" href="/page/2">next</a>
</body></html>"#;
let document = Document::from(html);
let article_url = Url::parse("https://example.com/page/1").unwrap();
// Several lines are tried in order; the first one that matches wins.
let rules = |exprs: &[&str]| -> Vec<PageLink> {
exprs
.iter()
.map(|e| Selector::xpath(e).unwrap().into())
.collect()
};
let config = ConfigEntry {
single_page_link: rules(&["//a[@class='missing']", "//a[@class='single']"]),
next_page_link: rules(&["//a[@class='missing']", "//a[@rel='next']", "//a"]),
..Default::default()
};
let single = FullTextParser::find_page_link(
&document,
Some(&config.single_page_link),
&[],
&article_url,
);
assert_eq!(single.unwrap().as_str(), "https://example.com/all");
let next =
FullTextParser::find_page_link(&document, Some(&config.next_page_link), &[], &article_url);
assert_eq!(next.unwrap().as_str(), "https://example.com/page/2");
}
/// Serves two pages whose `next_page_link`s point at each other. After `max_requests` it
/// closes, so code without a cycle guard fails instead of hanging.
fn serve_page_cycle(max_requests: usize) -> Url {
test_server::serve(max_requests, |request_line, stream| {
let next = if request_line.contains("/a ") {
"/b#top"
} else {
"/a"
};
let body = format!(
r#"<html><body><article><p>page</p></article><a rel="next" href="{next}">next</a></body></html>"#
);
test_server::write_response(stream, "text/html; charset=utf-8", &body);
})
}
#[test(tokio::test)]
async fn next_page_link_cycle_stops() {
let base = serve_page_cycle(10);
let first_page = base.join("/a").unwrap();
let client = Client::new();
let config = ConfigEntry {
next_page_link: vec![Selector::xpath("//a[@rel='next']").unwrap().into()],
..Default::default()
};
let parser = FullTextParser::new(None).await;
let global_config = parser.config_files.get("global.txt").unwrap();
let html = FullTextParser::download(&first_page, &client, Some(&config), global_config)
.await
.unwrap();
let pages = parser
.download_all_pages(html, &client, Some(&config), global_config, &first_page)
.await
.unwrap();
// /a -> /b#top -> /a: the second link to /a is a page that was already fetched.
assert_eq!(pages.len(), 2);
}
#[test(tokio::test)]
async fn next_page_link_stops_at_page_limit() {
// Every page links to the one after it, like a `?page=N` the server answers for any `N`.
let base = test_server::serve(constants::MAX_PAGES + 5, |request_line, stream| {
let page = request_line
.split_whitespace()
.nth(1)
.and_then(|path| path.strip_prefix("/page/"))
.and_then(|page| page.parse::<usize>().ok())
.unwrap_or(0);
let body = format!(
r#"<html><body><article><p>page {page}</p></article><a rel="next" href="/page/{}">next</a></body></html>"#,
page + 1
);
test_server::write_response(stream, "text/html; charset=utf-8", &body);
});
let first_page = base.join("/page/1").unwrap();
let client = Client::new();
let config = ConfigEntry {
next_page_link: vec![Selector::xpath("//a[@rel='next']").unwrap().into()],
..Default::default()
};
let parser = FullTextParser::new(None).await;
let global_config = parser.config_files.get("global.txt").unwrap();
let html = FullTextParser::download(&first_page, &client, Some(&config), global_config)
.await
.unwrap();
let pages = parser
.download_all_pages(html, &client, Some(&config), global_config, &first_page)
.await
.unwrap();
assert_eq!(pages.len(), constants::MAX_PAGES);
}
#[test(tokio::test)]
async fn failed_single_page_link_keeps_pages() {
// /a links to a single page that answers with an empty body, and to /b as its next page.
let base = test_server::serve(10, |request_line, stream| {
if request_line.contains("/single ") {
test_server::write_response(stream, "text/html", "");
return;
}
let body = if request_line.contains("/a ") {
r#"<html><body><article><p>page a</p></article><a rel="single" href="/single">all</a><a rel="next" href="/b">next</a></body></html>"#
} else {
r#"<html><body><article><p>page b</p></article></body></html>"#
};
test_server::write_response(stream, "text/html; charset=utf-8", body);
});
let first_page = base.join("/a").unwrap();
let client = Client::new();
let config = ConfigEntry {
single_page_link: vec![Selector::xpath("//a[@rel='single']").unwrap().into()],
next_page_link: vec![Selector::xpath("//a[@rel='next']").unwrap().into()],
..Default::default()
};
let parser = FullTextParser::new(None).await;
let global_config = parser.config_files.get("global.txt").unwrap();
let html = FullTextParser::download(&first_page, &client, Some(&config), global_config)
.await
.unwrap();
let pages = parser
.download_all_pages(html, &client, Some(&config), global_config, &first_page)
.await
.unwrap();
assert_eq!(pages.len(), 2);
assert!(pages[0].contains("page a"));
assert!(pages[1].contains("page b"));
}
#[test(tokio::test)]
async fn error_page_or_pdf_single_page_link_keeps_pages() {
// /a (passed in directly) links to a single page that is an HTML error page (/404) or a
// PDF (/pdf), and to /b as its next page.
let base = test_server::serve(10, |request_line, stream| {
if request_line.contains("/404 ") {
test_server::write_status_response(
stream,
"404 Not Found",
"text/html",
"<html><body><p>not found</p></body></html>",
);
return;
}
if request_line.contains("/pdf ") {
test_server::write_response(stream, "application/pdf", "%PDF-1.4");
return;
}
test_server::write_response(
stream,
"text/html; charset=utf-8",
r#"<html><body><article><p>page b</p></article></body></html>"#,
);
});
let client = Client::new();
let parser = FullTextParser::new(None).await;
let global_config = parser.config_files.get("global.txt").unwrap();
for single_page in ["/404", "/pdf"] {
let config = ConfigEntry {
single_page_link: vec![Selector::xpath("//a[@rel='single']").unwrap().into()],
next_page_link: vec![Selector::xpath("//a[@rel='next']").unwrap().into()],
..Default::default()
};
let first_page = base.join("/a").unwrap();
let html = format!(
r#"<html><body><article><p>page a</p></article><a rel="single" href="{single_page}">all</a><a rel="next" href="/b">next</a></body></html>"#
);
let pages = parser
.download_all_pages(html, &client, Some(&config), global_config, &first_page)
.await
.unwrap();
assert_eq!(pages.len(), 2, "{single_page}");
assert!(pages[0].contains("page a"), "{single_page}");
assert!(pages[1].contains("page b"), "{single_page}");
}
}
#[test(tokio::test)]
async fn error_page_or_pdf_is_rejected() {
let base = test_server::serve(2, |request_line, stream| {
if request_line.contains("/500 ") {
test_server::write_status_response(
stream,
"500 Internal Server Error",
"text/html",
"<html><body><p>server error</p></body></html>",
);
} else {
test_server::write_response(stream, "application/pdf", "%PDF-1.4");
}
});
let client = Client::new();
let parser = FullTextParser::new(None).await;
let global_config = parser.config_files.get("global.txt").unwrap();
let url = base.join("/500").unwrap();
let result = FullTextParser::download(&url, &client, None, global_config).await;
assert!(
matches!(result, Err(FullTextParserError::Http)),
"{result:?}"
);
let url = base.join("/pdf").unwrap();
let result = FullTextParser::download(&url, &client, None, global_config).await;
assert!(
matches!(result, Err(FullTextParserError::ContentType)),
"{result:?}"
);
}
#[test(tokio::test)]
async fn xhtml_is_accepted() {
let base = test_server::serve(2, |request_line, stream| {
let content_type = if request_line.contains("/xhtml ") {
"application/xhtml+xml; charset=utf-8"
} else {
"Text/HTML"
};
test_server::write_response(
stream,
content_type,
r#"<html xmlns="http://www.w3.org/1999/xhtml"><body><p>page</p></body></html>"#,
);
});
let client = Client::new();
let parser = FullTextParser::new(None).await;
let global_config = parser.config_files.get("global.txt").unwrap();
for path in ["/xhtml", "/uppercase"] {
let url = base.join(path).unwrap();
let result = FullTextParser::download(&url, &client, None, global_config).await;
assert!(result.is_ok_and(|html| html.contains("page")), "{path}");
}
}
#[test(tokio::test)]
async fn oversized_html_is_rejected() {
let base = test_server::serve(2, |request_line, stream| {
if request_line.contains("/huge-header ") {
test_server::write_huge_content_length(stream, "text/html");
} else {
test_server::write_endless_body(stream, "text/html");
}
});
let client = Client::new();
let parser = FullTextParser::new(None).await;
let global_config = parser.config_files.get("global.txt").unwrap();
for path in ["/huge-header", "/endless"] {
let url = base.join(path).unwrap();
let result = FullTextParser::download(&url, &client, None, global_config).await;
assert!(
matches!(result, Err(FullTextParserError::TooLarge)),
"{path}: {result:?}"
);
}
}
#[test(tokio::test)]
async fn multi_step_body_rules() {
let html = r#"<html><body>
<main><div id="content"><p>article text</p><div><p>nested div</p></div></div><div>second</div></main>
<div class="post"><div class="post">A nested match, long enough to survive cleanup.</div></div>
</body></html>"#;
let parser = FullTextParser::new(None).await;
let extract = |body: &str| {
let config = ConfigEntry {
body: vec![Selector::xpath(body).unwrap()],
..Default::default()
};
parser
.parse_offline(vec![html.to_string()], Some(&config), None)
.unwrap()
.html
.unwrap()
};
// Multi-step and parent-step rules match (no readability fallback, which would also pick
// up the other divs).
let content = extract("//main/div[1]");
assert!(
content.contains("article text") && !content.contains("second"),
"{content}"
);
let content = extract("//p[contains(text(), 'article text')]/..");
assert!(
content.contains("article text") && !content.contains("second"),
"{content}"
);
// A match inside another match is moved along with it, not a second time.
let content = extract("//div[@class='post']");
assert_eq!(content.matches("A nested match").count(), 1, "{content}");
}
#[test(tokio::test)]
async fn post_processing_stays_inside_the_body_match() {
// Post-processing the first match used to walk on through the rest of the page and strip
// all class attributes there, so later body rules selecting by class found nothing.
let html = r#"<html><body>
<div class="intro"><p>The intro is a long enough sentence to survive the cleanup.</p></div>
<div class="text"><p>The main text is a long enough sentence to survive the cleanup.</p></div>
</body></html>"#;
let config = ConfigEntry {
body: vec![
Selector::xpath("//div[@class='intro']").unwrap(),
Selector::xpath("//div[@class='text']").unwrap(),
],
..Default::default()
};
let parser = FullTextParser::new(None).await;
let content = parser
.parse_offline(vec![html.to_string()], Some(&config), None)
.unwrap()
.html
.unwrap();
assert!(content.contains("The intro"), "{content}");
assert!(content.contains("The main text"), "{content}");
}
#[test(tokio::test)]
async fn body_union_in_document_order() {
let html = r#"<html><body>
<div id="lead"><p>The lead is a long enough sentence to survive the cleanup.</p></div>
<div id="text"><p>The main text is a long enough sentence to survive the cleanup.</p>
<div id="lead-2"><p>A nested match is a long enough sentence to survive the cleanup.</p></div>
</div>
</body></html>"#;
// '|' is a union: one result in document order, not one alternative after the other.
let config = ConfigEntry {
body: vec![Selector::xpath("//div[@id='text'] | //div[starts-with(@id, 'lead')]").unwrap()],
..Default::default()
};
let parser = FullTextParser::new(None).await;
let content = parser
.parse_offline(vec![html.to_string()], Some(&config), None)
.unwrap()
.html
.unwrap();
let lead = content.find("The lead").unwrap();
let text = content.find("The main text").unwrap();
let nested = content.find("A nested match").unwrap();
assert!(lead < text && text < nested, "{content}");
assert_eq!(content.matches("A nested match").count(), 1, "{content}");
}
#[test(tokio::test)]
async fn metadata_from_json_ld() {
let html = r#"<html><head>
<title>Something else | Example</title>
<script type="application/ld+json">
{"@context": "https://schema.org", "@type": "NewsArticle", "headline": "The headline",
"author": {"@type": "Person", "name": " Jane Doe "}, "datePublished": "2024-01-02T03:04:05Z",
"image": "/img/lead.jpg"}
</script>
</head><body><article><p>Some text.</p></article></body></html>"#;
let config = ConfigEntry {
body: vec![Selector::xpath("//article").unwrap()],
..Default::default()
};
let parser = FullTextParser::new(None).await;
let url = Url::parse("https://example.com/2024/article").unwrap();
let article = parser
.parse_offline(vec![html.into()], Some(&config), Some(url))
.unwrap();
assert_eq!(article.title.as_deref(), Some("The headline"));
assert_eq!(article.author.as_deref(), Some("Jane Doe"));
assert_eq!(
article.date.map(|date| date.to_rfc3339()).as_deref(),
Some("2024-01-02T03:04:05+00:00")
);
assert_eq!(
article.thumbnail_url.as_deref(),
Some("https://example.com/img/lead.jpg")
);
// `skip_json_ld: yes`: only the <title> is left (readability keeps the site name when
// fewer than three words would remain)
let config = ConfigEntry {
skip_json_ld: Some(true),
..config
};
let url = Url::parse("https://example.com/2024/article").unwrap();
let article = parser
.parse_offline(vec![html.into()], Some(&config), Some(url))
.unwrap();
assert_eq!(article.title.as_deref(), Some("Something else | Example"));
assert_eq!(article.author, None);
assert_eq!(article.date, None);
}
#[test(tokio::test)]
async fn author_from_readability_byline() {
let paragraph = "This is a paragraph of the article, long enough and with enough commas, \
sentences, and words in it, so that readability considers it content. ";
let html = format!(
r#"<html><head><title>An article</title></head><body>
<div class="content">
<p class="byline">Jane Doe</p>
<p>{0}{0}</p><p>{0}{0}</p><p>{0}{0}</p>
</div>
</body></html>"#,
paragraph
);
let parser = FullTextParser::new(None).await;
let url = Url::parse("https://example.com/article").unwrap();
let article = parser.parse_offline(vec![html], None, Some(url)).unwrap();
assert_eq!(article.author.as_deref(), Some("Jane Doe"));
// readability removes the byline from the content
assert!(!article.html.unwrap().contains("Jane Doe"));
}
#[test(tokio::test)]
async fn string_valued_rules() {
let html = r#"<html><head><title>The title | Example</title></head><body>
<p class="by">Written by Jane Doe</p>
<article><p>Some text.</p><a class="all" href="?page=all">all</a></article>
</body></html>"#;
let config = ConfigEntry {
title: vec![Selector::xpath("substring-before(//title, ' | ')").unwrap()],
author: vec![
// an empty string counts as no match
Selector::xpath("substring-after(//p[@class='by'], 'nothing')").unwrap(),
Selector::xpath("substring-after(//p[@class='by'], 'by ')").unwrap(),
],
body: vec![Selector::xpath("//article").unwrap()],
single_page_link: vec![
Selector::xpath("concat(//a[@class='all']/@href, '&x=1')")
.unwrap()
.into(),
],
..Default::default()
};
let parser = FullTextParser::new(None).await;
let url = Url::parse("https://example.com/article").unwrap();
let article = parser
.parse_offline(vec![html.into()], Some(&config), Some(url.clone()))
.unwrap();
assert_eq!(article.title.as_deref(), Some("The title"));
assert_eq!(article.author.as_deref(), Some("Jane Doe"));
let document = Document::from(html);
assert_eq!(
FullTextParser::find_page_link(&document, Some(&config.single_page_link), &[], &url)
.map(String::from)
.as_deref(),
Some("https://example.com/article?page=all&x=1")
);
}
#[test(tokio::test)]
async fn insert_detected_image() {
let page = |body: &str| {
format!(
r#"<html><head><meta property="og:image" content="https://example.com/lead.jpg"></head>
<body><article>{body}</article></body></html>"#
)
};
let parser = FullTextParser::new(None).await;
let url = Url::parse("https://example.com/article").unwrap();
let parse = |html: String, insert: Option<bool>| {
let config = ConfigEntry {
body: vec![Selector::xpath("//article").unwrap()],
insert_detected_image: insert,
..Default::default()
};
parser
.parse_offline(vec![html], Some(&config), Some(url.clone()))
.unwrap()
.html
.unwrap()
};
let html = parse(page("<p>No image here.</p>"), None);
assert!(
html.contains(r#"<img src="https://example.com/lead.jpg">"#),
"{html}"
);
let html = parse(page("<p>No image here.</p>"), Some(false));
assert!(!html.contains("<img"), "{html}");
let html = parse(
page(r#"<p>Text</p><img src="https://example.com/other.jpg">"#),
None,
);
assert!(!html.contains("lead.jpg"), "{html}");
}
#[test(tokio::test)]
async fn autodetect_on_failure() {
let paragraph = "This is a paragraph of the article, long enough and with enough commas, \
sentences, and words in it, so that readability considers it content. ";
let html = format!(
"<html><body><div><p>{0}{0}</p><p>{0}{0}</p><p>{0}{0}</p></div></body></html>",
paragraph
);
let parser = FullTextParser::new(None).await;
let url = Url::parse("https://example.com/article").unwrap();
let parse = |autodetect: Option<bool>| {
let config = ConfigEntry {
body: vec![Selector::xpath("//article").unwrap()],
autodetect_on_failure: autodetect,
..Default::default()
};
parser.parse_offline(vec![html.clone()], Some(&config), Some(url.clone()))
};
// the body rule matches nothing: readability takes over, unless the config says no
assert!(
parse(None)
.unwrap()
.html
.unwrap()
.contains("readability considers")
);
assert!(matches!(
parse(Some(false)),
Err(crate::FullTextParserError::Scrape)
));
}
#[test(tokio::test)]
async fn page_link_if_page_contains() {
let article_url = Url::parse("https://example.com/page/1").unwrap();
let rules = vec![PageLink {
selector: Selector::xpath("//a[@class='single']").unwrap(),
if_page_contains: Some(Selector::xpath("//div[@id='paged']").unwrap()),
}];
let link = |html: &str| {
FullTextParser::find_page_link(&Document::from(html), Some(&rules), &[], &article_url)
.map(String::from)
};
let link_html = r#"<a class="single" href="/all">all on one page</a>"#;
assert_eq!(link(link_html), None);
assert_eq!(
link(&format!(r#"<div id="paged"></div>{link_html}"#)).as_deref(),
Some("https://example.com/all")
);
}
#[test(tokio::test)]
async fn native_ad_clue() {
let html = r#"<html><head><meta property="og:url" content="https://example.com/sponsored/x"></head>
<body><article><p>Buy things.</p></article></body></html>"#;
let parser = FullTextParser::new(None).await;
let url = Url::parse("https://example.com/sponsored/x").unwrap();
let parse = |clue: &str| {
let config = ConfigEntry {
body: vec![Selector::xpath("//article").unwrap()],
native_ad_clue: vec![Selector::xpath(clue).unwrap()],
..Default::default()
};
parser
.parse_offline(vec![html.into()], Some(&config), Some(url.clone()))
.unwrap()
.native_ad
};
assert!(parse(
r#"//meta[@property="og:url" and contains(@content, '/sponsored/')]"#
));
assert!(!parse(
r#"//meta[@property="article:section" and @content="Advertiser"]"#
));
}
#[test(tokio::test)]
async fn post_strip_attr() {
let html = r#"<html><body><article>
<p>Text</p><img src="https://example.com/a.jpg" title="a title" alt="alt text">
</article></body></html>"#;
let config = ConfigEntry {
body: vec![Selector::xpath("//article").unwrap()],
post_strip_attr: vec![Selector::xpath("//img/@title").unwrap()],
..Default::default()
};
let parser = FullTextParser::new(None).await;
let url = Url::parse("https://example.com/article").unwrap();
let html = parser
.parse_offline(vec![html.into()], Some(&config), Some(url))
.unwrap()
.html
.unwrap();
assert!(!html.contains("a title"), "{html}");
assert!(html.contains("alt text"), "{html}");
}
#[test(tokio::test)]
async fn src_lazy_load_attr() {
let html = r#"<html><body><article><p>Text</p>
<img src="data:image/gif;base64,R0lGODlhAQABAAAAACw=" data-full-src="https://example.com/full.jpg">
<img src="https://example.com/b.jpg" data-full-src="">
</article></body></html>"#;
let config = ConfigEntry {
body: vec![Selector::xpath("//article").unwrap()],
src_lazy_load_attr: Some("data-full-src".into()),
..Default::default()
};
let parser = FullTextParser::new(None).await;
let url = Url::parse("https://example.com/article").unwrap();
let html = parser
.parse_offline(vec![html.into()], Some(&config), Some(url))
.unwrap()
.html
.unwrap();
assert!(
html.contains(r#"src="https://example.com/full.jpg""#),
"{html}"
);
assert!(
html.contains(r#"src="https://example.com/b.jpg""#),
"{html}"
);
}
#[test(tokio::test)]
async fn move_into_body() {
let html = r#"<html><body>
<div id="nav">Navigation</div>
<div id="thephoto"><img src="https://example.com/photo.jpg"></div>
<div id="description"><p>About the photo.</p></div>
</body></html>"#;
let config = ConfigEntry {
move_into_body: vec![
Selector::xpath("//div[@id='thephoto']").unwrap(),
Selector::xpath("//div[@id='description']").unwrap(),
],
..Default::default()
};
let parser = FullTextParser::new(None).await;
let url = Url::parse("https://example.com/photo/1").unwrap();
let html = parser
.parse_offline(vec![html.into()], Some(&config), Some(url))
.unwrap()
.html
.unwrap();
assert!(
html.contains("photo.jpg") && html.contains("About the photo."),
"{html}"
);
assert!(!html.contains("Navigation"), "{html}");
}
#[test]
fn extract_thumbnail_relative() {
let html = r#"<article><img src="/images/hero.jpg" width="800" height="450" /></article>"#;
let doc = Document::from(html);
let base_url = Url::parse("https://example.com/news/article.html").unwrap();
let thumb = FullTextParser::check_for_thumbnail(&doc, None, Some(&base_url));
assert_eq!(
thumb.as_deref(),
Some("https://example.com/images/hero.jpg")
);
// Without a base URL only absolute URLs count.
assert_eq!(FullTextParser::check_for_thumbnail(&doc, None, None), None);
let html = r#"<head><link rel="image_src" src="lead.jpg"></head><body><p>text</p></body>"#;
let doc = Document::from(html);
let thumb = FullTextParser::thumbnail_from_images(&doc, Some(&base_url));
assert_eq!(thumb.as_deref(), Some("https://example.com/news/lead.jpg"));
}
#[test(tokio::test)]
async fn config_title_and_author_are_decoded_once() {
// html5ever already decodes `&amp;` to the text `&`; it must not become `&`.
let html = r#"<html><head><title>Ignored</title></head><body>
<h1>Tom &amp; Jerry &lt;3</h1>
<span class="author">A &amp; B</span>
<article><p>Some article text that is long enough to be kept as content.</p></article>
</body></html>"#;
let config = ConfigEntry {
title: vec![Selector::xpath("//h1").unwrap()],
author: vec![Selector::xpath("//span[@class='author']").unwrap()],
body: vec![Selector::xpath("//article").unwrap()],
..Default::default()
};
let parser = FullTextParser::new(None).await;
let article = parser
.parse_offline(vec![html.to_string()], Some(&config), None)
.unwrap();
assert_eq!(article.title.as_deref(), Some("Tom & Jerry <3"));
assert_eq!(article.author.as_deref(), Some("A & B"));
}
#[test]
fn parse_date_formats() {
use super::metadata::{ParsedDate, parse_date};
let date = |s: &str, naive: bool| {
Some(ParsedDate {
date: chrono::DateTime::parse_from_rfc3339(s).unwrap().to_utc(),
naive,
})
};
assert_eq!(
parse_date("2023-08-03T10:00:00+02:00"),
date("2023-08-03T08:00:00Z", false)
);
assert_eq!(
parse_date("Thu, 03 Aug 2023 10:00:00 +0200"),
date("2023-08-03T08:00:00Z", false)
);
assert_eq!(
parse_date("2022-02-09T22:01:14+0100"),
date("2022-02-09T21:01:14Z", false)
);
assert_eq!(
parse_date("2022-02-09 22:01:14.5+01:00"),
date("2022-02-09T21:01:14.5Z", false)
);
assert_eq!(
parse_date("2023-08-03 10:00:15"),
date("2023-08-03T10:00:15Z", true)
);
assert_eq!(
parse_date("2023-08-03T10:00:15.250"),
date("2023-08-03T10:00:15.250Z", true)
);
assert_eq!(
parse_date("2023-08-03T10:00"),
date("2023-08-03T10:00:00Z", true)
);
assert_eq!(
parse_date(" 2023-08-03 "),
date("2023-08-03T00:00:00Z", true)
);
assert_eq!(parse_date("3. August 2023"), None);
}
#[test(tokio::test)]
async fn config_date_rfc_2822() {
let html = r#"<html><head><title>Title</title></head><body>
<span class="date">Thu, 03 Aug 2023 10:00:00 +0200</span>
<article><p>Some article text that is long enough to be kept as content.</p></article>
</body></html>"#;
let config = ConfigEntry {
date: vec![Selector::xpath("//span[@class='date']").unwrap()],
body: vec![Selector::xpath("//article").unwrap()],
..Default::default()
};
let parser = FullTextParser::new(None).await;
let article = parser
.parse_offline(vec![html.to_string()], Some(&config), None)
.unwrap();
assert_eq!(
article.date.map(|date| date.to_rfc3339()).as_deref(),
Some("2023-08-03T08:00:00+00:00")
);
}
#[test(tokio::test)]
async fn config_date_without_offset_loses_to_json_ld() {
let html = r#"<html><head><title>Title</title>
<script type="application/ld+json">{"@context": "https://schema.org", "@type": "NewsArticle",
"headline": "Title", "datePublished": "2022-02-09T22:01:14+0100"}</script>
</head><body>
<time datetime="2022-02-09 22:01">09.02.2022, 22:01 Uhr</time>
<article><p>Some article text that is long enough to be kept as content.</p></article>
</body></html>"#;
let config = ConfigEntry {
date: vec![Selector::xpath("//time/@datetime").unwrap()],
body: vec![Selector::xpath("//article").unwrap()],
..Default::default()
};
let parser = FullTextParser::new(None).await;
let article = parser
.parse_offline(vec![html.to_string()], Some(&config), None)
.unwrap();
assert_eq!(
article.date.map(|date| date.to_rfc3339()).as_deref(),
Some("2022-02-09T21:01:14+00:00")
);
}