pub mod config;
mod fingerprints;
mod metadata;
mod page;
mod readability;
#[cfg(test)]
mod fixture_report;
#[cfg(test)]
mod tests;
use self::config::{ConfigCollection, ConfigEntry, PageLink};
use self::page::Page;
pub use self::readability::Readability;
use crate::article::Article;
use crate::constants::{self, CHARSET, HTML_CHARSET};
use crate::dom;
use crate::error::FullTextParserError;
use crate::selector::Selector;
use crate::util::Util;
use dom_query::{Document, NodeRef};
use dom_query_xpath::XNode;
use encoding_rs::Encoding;
use fingerprints::Fingerprints;
use futures::StreamExt;
use reqwest::header::HeaderMap;
use reqwest::{Client, Response, Url};
use std::collections::HashSet;
use std::path::Path;
use std::str::from_utf8;
pub struct FullTextParser {
config_files: ConfigCollection,
}
impl FullTextParser {
pub async fn new(config_path: Option<&Path>) -> Self {
let config_files = ConfigCollection::parse(config_path).await;
Self { config_files }
}
pub(crate) async fn parse(
&self,
url: &url::Url,
client: &Client,
) -> Result<Article, FullTextParserError> {
tracing::debug!(%url, "Scraping article");
let config = self.get_grabber_config(url);
let global_config = self
.config_files
.get("global.txt")
.ok_or(FullTextParserError::Config)?;
let headers = Util::generate_headers(config, global_config)?;
let (response, new_url) = Self::get_response(url, client, headers).await?;
let url = if let Some(new_url) = new_url {
tracing::debug!(%url, %new_url, "Url redirects");
new_url
} else {
url.clone()
};
if !Util::check_content_type(&response)? {
return Err(FullTextParserError::ContentType);
}
let html = Self::get_body(response).await?;
if html.is_empty() {
tracing::error!("Empty response body");
return Err(FullTextParserError::Http);
}
let config = if config.is_none() {
if let Some(host) = Fingerprints::detect(&html) {
self.config_for_host(host)
} else {
config
}
} else {
config
};
let pages = self
.download_all_pages(html, client, config, global_config, &url)
.await?;
self.parse_offline(pages, config, Some(url))
}
pub fn parse_offline(
&self,
pages: Vec<String>,
config: Option<&ConfigEntry>,
url: Option<Url>,
) -> Result<Article, FullTextParserError> {
let url = url.unwrap_or_else(|| url::Url::parse("http://fakehost/test/base/").unwrap());
let config = if config.is_none() {
self.get_grabber_config(&url)
} else {
config
};
let global_config = self
.config_files
.get("global.txt")
.ok_or(FullTextParserError::Config)?;
let mut article = Article {
title: None,
author: None,
url: url.clone(),
date: None,
thumbnail_url: None,
html: None,
native_ad: false,
};
let document = dom::new_article_document();
let root = dom::article_root(&document).ok_or(FullTextParserError::Dom)?;
let mut detected_image = None;
for page_html in pages {
let page_image =
self.parse_page(&mut article, &page_html, &root, config, global_config)?;
detected_image = detected_image.or(page_image);
}
for selector in config
.map(|config| config.post_strip_attr.as_slice())
.unwrap_or_default()
.iter()
.chain(&global_config.post_strip_attr)
{
_ = Util::strip_attributes_at(&root, selector);
}
Self::post_process_document(&root)?;
let insert_detected_image = config
.and_then(|config| config.insert_detected_image)
.or(global_config.insert_detected_image)
.unwrap_or(true);
if insert_detected_image && let Some(image) = detected_image {
Self::insert_detected_image(&root, &image);
}
article.html = Some(root.html().to_string());
Ok(article)
}
async fn download_all_pages(
&self,
html: String,
client: &Client,
config: Option<&ConfigEntry>,
global_config: &ConfigEntry,
article_url: &Url,
) -> Result<Vec<String>, FullTextParserError> {
if !Self::follows_page_links(config, global_config) {
return Ok(vec![html]);
}
let mut html = html;
let mut pages = vec![html.clone()];
let mut visited = HashSet::from([Self::page_key(article_url)]);
let mut follow_single_page = true;
while let Ok(page_result) = self.evaluate_page(
&html,
follow_single_page,
config,
global_config,
article_url,
) {
if let Page::Single(single_page_url) = page_result {
match Self::download(&single_page_url, client, config, global_config).await {
Ok(single_page_html) => return Ok(vec![single_page_html]),
Err(error) => {
tracing::warn!(%single_page_url, %error, "Single page download failed, keeping the pages");
follow_single_page = false;
continue;
}
}
} else if let Page::Multi(next_page_url) = page_result {
if let Some(next_page_url) = next_page_url {
if pages.len() >= constants::MAX_PAGES {
tracing::warn!(%next_page_url, "Reached the page limit, stopping");
break;
}
if !visited.insert(Self::page_key(&next_page_url)) {
tracing::warn!(%next_page_url, "Next page was already fetched, stopping");
break;
}
let next_page_html =
Self::download(&next_page_url, client, config, global_config).await?;
pages.push(next_page_html.clone());
html = next_page_html;
continue;
} else {
break;
}
}
}
Ok(pages)
}
fn follows_page_links(config: Option<&ConfigEntry>, global_config: &ConfigEntry) -> bool {
let has_rules = |config: &ConfigEntry| {
!config.single_page_link.is_empty() || !config.next_page_link.is_empty()
};
config.is_some_and(has_rules)
|| has_rules(global_config)
|| Self::autodetect_next_page(config, global_config)
}
fn autodetect_next_page(config: Option<&ConfigEntry>, global_config: &ConfigEntry) -> bool {
config
.and_then(|config| config.autodetect_next_page)
.or(global_config.autodetect_next_page)
.unwrap_or(false)
}
fn evaluate_page(
&self,
html: &str,
follow_single_page: bool,
config: Option<&ConfigEntry>,
global_config: &ConfigEntry,
article_url: &Url,
) -> Result<Page, FullTextParserError> {
let document = Self::parse_html(html, config, global_config);
let single_page_url = follow_single_page
.then(|| {
Self::find_page_link(
&document,
config.map(|c| c.single_page_link.as_slice()),
&global_config.single_page_link,
article_url,
)
})
.flatten();
if let Some(single_page_url) = single_page_url {
tracing::trace!(%single_page_url, "Single page link found");
return Ok(Page::Single(single_page_url));
}
let next_page_url = self.check_for_next_page(&document, config, global_config, article_url);
Ok(Page::Multi(next_page_url))
}
fn parse_page(
&self,
article: &mut Article,
html: &str,
root: &NodeRef,
config: Option<&ConfigEntry>,
global_config: &ConfigEntry,
) -> Result<Option<String>, FullTextParserError> {
let document = Self::parse_html(html, config, global_config);
if !article.native_ad {
article.native_ad = Self::is_native_ad(&document, config, global_config);
}
let skip_json_ld = config
.and_then(|config| config.skip_json_ld)
.or(global_config.skip_json_ld)
.unwrap_or(false);
let (document, page_metadata) = metadata::page_metadata(document, skip_json_ld);
metadata::extract(
&document,
config,
Some(global_config),
&page_metadata,
article,
);
let json_ld_image = page_metadata
.image
.and_then(|image| article.url.join(&image).ok())
.map(String::from);
let detected_image = Self::detected_image(&document, json_ld_image);
if article.thumbnail_url.is_none() {
article.thumbnail_url = detected_image
.clone()
.or_else(|| Self::thumbnail_from_images(&document, Some(&article.url)));
}
Self::prep_content(
&document,
config,
global_config,
&article.url,
article.title.as_deref(),
);
let found_body =
Self::extract_body(&document, root, config, global_config).unwrap_or(false);
let autodetect_on_failure = config
.and_then(|config| config.autodetect_on_failure)
.or(global_config.autodetect_on_failure)
.unwrap_or(true);
let has_body_rules = config.is_some_and(|config| !config.body.is_empty());
if !found_body && !autodetect_on_failure && has_body_rules {
tracing::warn!("ftr failed to find content, and the config disables readability");
return Err(FullTextParserError::Scrape);
}
if !found_body {
tracing::warn!("ftr failed to find content. trying readabilty");
match Readability::extract_body(document, root, Some(&article.url)) {
Ok(byline) => {
if article.author.is_none() {
article.author = byline;
}
}
Err(error) => {
tracing::error!("Both ftr and readability failed to find content: {error}");
return Err(error);
}
}
}
Ok(detected_image)
}
pub(crate) fn parse_html(
html: &str,
config: Option<&ConfigEntry>,
global_config: &ConfigEntry,
) -> Document {
let mut html = html.to_owned();
if let Some(config) = config {
for replace in &config.replace {
html = html.replace(&replace.to_replace, &replace.replace_with);
}
}
for replace in &global_config.replace {
html = html.replace(&replace.to_replace, &replace.replace_with);
}
Document::from(html.as_str())
}
async fn get_response(
url: &url::Url,
client: &Client,
headers: HeaderMap,
) -> Result<(Response, Option<Url>), FullTextParserError> {
Self::get_response_impl(url, client, headers, 0).await
}
async fn get_response_impl(
url: &url::Url,
client: &Client,
headers: HeaderMap,
depth: u32,
) -> Result<(Response, Option<Url>), FullTextParserError> {
let response = client
.get(url.as_str())
.headers(headers.clone())
.send()
.await
.map_err(|err| {
tracing::error!(%url, ?err, "Downloading HTML failed: GET");
FullTextParserError::Http
})?;
let status = response.status();
tracing::debug!(%status);
if response.status().is_redirection()
&& let Some(new_location) = response
.headers()
.get("location")
.and_then(|header| std::str::from_utf8(header.as_bytes()).ok())
{
tracing::debug!(new_location);
if depth >= constants::MAX_REDIRECTS {
tracing::error!(depth, "max redirects reached");
return Err(FullTextParserError::Http);
}
if new_location == url.as_str() {
tracing::error!(new_location, "redirect url is same as original");
return Err(FullTextParserError::Http);
}
let Ok(new_url) = Url::parse(new_location) else {
tracing::error!(new_location, "not a valid url");
return Err(FullTextParserError::Http);
};
tracing::debug!(new_location, "redirect");
let (response, redirected_url) = Box::pin(Self::get_response_impl(
&new_url,
client,
headers,
depth + 1,
))
.await?;
if redirected_url.is_some() {
return Ok((response, redirected_url));
}
return Ok((response, Some(new_url)));
}
Ok((response, None))
}
async fn get_body(response: Response) -> Result<String, FullTextParserError> {
let headers = response.headers().clone();
let content_length = headers
.get(reqwest::header::CONTENT_LENGTH)
.and_then(|hv| hv.to_str().ok())
.and_then(|str| str.parse::<u64>().ok());
if content_length == Some(0) {
tracing::error!("Empty response body");
return Err(FullTextParserError::Http);
}
if let Some(content_length) = content_length
&& content_length > constants::MAX_HTML_SIZE as u64
{
tracing::error!(content_length, "Response body is too large");
return Err(FullTextParserError::TooLarge);
}
let mut bytes = Vec::with_capacity(content_length.unwrap_or(0) as usize);
let mut stream = response.bytes_stream();
while let Some(chunk) = stream.next().await {
let chunk = chunk.map_err(|_| FullTextParserError::Http)?;
if bytes.len() + chunk.len() > constants::MAX_HTML_SIZE {
tracing::error!("Response body is too large");
return Err(FullTextParserError::TooLarge);
}
bytes.extend_from_slice(&chunk);
}
match from_utf8(&bytes) {
Ok(utf8_str) => {
tracing::trace!(utf8_str, "Valid utf-8 string");
Ok(utf8_str.into())
}
Err(error) => {
let lossy_string = std::string::String::from_utf8_lossy(&bytes);
tracing::trace!(%lossy_string, "Invalid utf-8 string");
if let Some(encoding) = Self::get_encoding_from_html(&lossy_string) {
tracing::debug!(encoding, "Encoding extracted from HTML");
if let Some(decoded_html) = Self::decode_html(&bytes, encoding) {
return Ok(decoded_html);
}
}
if let Some(encoding) = Self::get_encoding_from_http_header(&headers) {
tracing::debug!(encoding, "Encoding extracted from headers");
if let Some(decoded_html) = Self::decode_html(&bytes, encoding) {
return Ok(decoded_html);
}
}
Err(FullTextParserError::Utf8(error))
}
}
}
pub async fn download(
url: &url::Url,
client: &Client,
config: Option<&ConfigEntry>,
global_config: &ConfigEntry,
) -> Result<String, FullTextParserError> {
let headers = Util::generate_headers(config, global_config)?;
let (response, _) = Self::get_response(url, client, headers).await?;
if !Util::check_content_type(&response)? {
return Err(FullTextParserError::ContentType);
}
let body = Self::get_body(response).await?;
if body.is_empty() {
tracing::error!("Empty response body");
Err(FullTextParserError::Http)
} else {
Ok(body)
}
}
fn get_encoding_from_http_header(headers: &reqwest::header::HeaderMap) -> Option<&str> {
headers
.get(reqwest::header::CONTENT_TYPE)
.and_then(|header| header.to_str().ok())
.and_then(|content_type| CHARSET.captures(content_type))
.and_then(|captures| captures.get(1))
.map(|regex_match| regex_match.as_str())
}
fn get_encoding_from_html(html: &str) -> Option<&str> {
if let Some(captures) = HTML_CHARSET.captures(html)
&& let Some(regex_match) = captures.get(1)
{
return Some(regex_match.as_str());
}
None
}
fn decode_html(bytes: &[u8], encoding: &str) -> Option<String> {
if let Some(encoding) = Encoding::for_label(encoding.as_bytes()) {
let (decoded_html, _, invalid_chars) = encoding.decode(bytes);
if !invalid_chars {
return Some(decoded_html.into_owned());
}
}
tracing::warn!(encoding, "Could not decode HTML. Encoding:");
None
}
fn get_host_name(url: &url::Url) -> Result<String, FullTextParserError> {
match url.host_str() {
Some(name) => {
let mut name = name;
if name.starts_with("www.") && name.len() > 4 {
name = &name[4..]
}
Ok(name.into())
}
None => {
tracing::error!("Getting config failed due to bad Url");
Err(FullTextParserError::Config)
}
}
}
fn get_grabber_config(&self, url: &url::Url) -> Option<&ConfigEntry> {
let conf = Self::get_host_name(url)
.ok()
.and_then(|host| self.config_for_host(&host));
if conf.is_none() {
tracing::warn!(%url, "No config found");
}
conf
}
fn config_for_host(&self, host: &str) -> Option<&ConfigEntry> {
self.config_files.get(&format!("{host}.txt"))
}
pub fn thumbnail_from_html(html: &str) -> Option<String> {
Self::check_for_thumbnail(&Document::from(html), None, None)
}
pub(crate) fn check_for_thumbnail(
document: &Document,
json_ld_image: Option<String>,
base_url: Option<&Url>,
) -> Option<String> {
Self::detected_image(document, json_ld_image)
.or_else(|| Self::thumbnail_from_images(document, base_url))
}
fn detected_image(document: &Document, json_ld_image: Option<String>) -> Option<String> {
if let Some(thumb) =
Util::get_attribute(document, "meta", ("name", "twitter:image"), "content")
{
return Some(thumb);
}
if let Some(thumb) = Util::get_attribute(document, "meta", ("name", "og:image"), "content")
{
return Some(thumb);
}
if let Some(thumb) =
Util::get_attribute(document, "meta", ("property", "twitter:image"), "content")
{
return Some(thumb);
}
if let Some(thumb) =
Util::get_attribute(document, "meta", ("property", "og:image"), "content")
{
return Some(thumb);
}
if let Some(thumb) = Util::get_attribute(document, "link", ("rel", "image_src"), "href") {
return Some(thumb);
}
json_ld_image
}
fn thumbnail_from_images(document: &Document, base_url: Option<&Url>) -> Option<String> {
let absolute = |url: &str| {
let url = url.trim();
match base_url {
Some(base_url) => base_url.join(url),
None => Url::parse(url),
}
.ok()
.map(String::from)
};
let img_nodes = dom::select(document, "img");
if !img_nodes.is_empty() {
let mut best: Option<(String, i32)> = None;
let len = img_nodes.len();
for (index, img_node) in img_nodes.into_iter().enumerate() {
let src = if let Some(src) = dom::attr(&img_node, "src") {
src
} else {
continue;
};
let score = Util::score_image_url(&src);
let score = score + Util::score_img_attr(&img_node);
let score = score + Util::score_by_parents(&img_node);
let score = score + Util::score_by_sibling(&img_node);
let score = score + Util::score_by_dimensions(&img_node);
let score = score + Util::score_by_position(len, index);
let score = score + Util::score_by_alt(&img_node);
if best
.as_ref()
.is_none_or(|(_, best_score)| score > *best_score)
{
best = Some((src, score));
}
}
if let Some((top_src, top_score)) = best
&& top_score > 0
&& let Some(top_url) = absolute(&top_src)
{
return Some(top_url);
}
}
if let Some(first_link_node) = dom::select(document, "link")
.into_iter()
.find(|link| dom::attr(link, "rel").as_deref() == Some("image_src"))
{
for attribute in ["src", "href", "value"] {
if let Some(url) =
dom::attr(&first_link_node, attribute).and_then(|url| absolute(&url))
{
return Some(url);
}
}
}
None
}
fn apply_lazy_load_attr(document: &Document, attribute: &str) {
for img in dom::elements_by_tag_name(&document.root(), "img") {
if let Some(src) = dom::attr(&img, attribute).filter(|src| !src.trim().is_empty()) {
img.set_attr("src", src.trim());
}
}
}
fn fix_lazy_images(document: &Document) {
let nodes = document.root().descendants_it().filter(|node| {
dom::local_name(node)
.is_some_and(|name| matches!(name.as_ref(), "img" | "picture" | "figure"))
});
for node in nodes.collect::<Vec<_>>() {
if let Some(src) = dom::attr(&node, "src")
&& let Some(data_url) = constants::BASE64_DATA_URL.captures(&src)
{
if &data_url[1] == "image/svg+xml" {
continue;
}
let mut src_could_be_removed = false;
for (name, val) in dom::attrs(&node) {
if name == "src" {
continue;
}
if constants::IS_IMAGE.is_match(&val) {
src_could_be_removed = true;
break;
}
}
let b64length = src.len() - data_url[0].len();
if src_could_be_removed && b64length < 133 {
node.remove_attr("src");
}
}
let class_contains_lazy = dom::attr(&node, "class")
.map(|c| c.to_lowercase().contains("lazy"))
.unwrap_or(false);
let has_scr = node.has_attr("src");
let has_srcset = node.has_attr("srcset");
if (has_scr || has_srcset) && !class_contains_lazy {
continue;
}
for (name, val) in dom::attrs(&node) {
if name == "src" || name == "srcset" || name == "alt" {
continue;
}
let mut copy_to: Option<&str> = None;
if constants::COPY_TO_SRCSET.is_match(&val) {
copy_to = Some("srcset");
} else if constants::COPY_TO_SRC.is_match(&val) {
copy_to = Some("src");
}
if let Some(copy_to) = copy_to {
if dom::tag_name_in(&node, &["img", "picture"]) {
node.set_attr(copy_to, &val);
} else if dom::tag_name_is(&node, "figure")
&& !dom::has_any_descendant_tag(&node, &["img", "picture"])
{
let img = node.tree.new_element("img");
img.set_attr(copy_to, &val);
dom::append_child(&node, &img);
}
}
}
}
}
fn fix_iframe_size(document: &Document, site_name: &str) {
for node in dom::elements_by_tag_name(&document.root(), "iframe") {
if !dom::attr(&node, "src").is_some_and(|src| src.contains(site_name)) {
continue;
}
node.set_attr("width", "480");
node.set_attr("height", "360");
node.set_attr("aspect-ratio", "auto");
}
}
fn strip_junk(document: &Document) {
let mut strip = Vec::new();
let mut links = Vec::new();
let mut late_strip = Vec::new();
let nodes = document
.root()
.descendants_it()
.filter(|node| node.is_element() || node.is_comment())
.collect::<Vec<_>>();
for node in nodes {
let Some(name) = dom::local_name(&node) else {
strip.push(node);
continue;
};
match name.as_ref() {
"html" => node.remove_attr("class"),
"a" => {
node.remove_attr("onclick");
links.push(node);
}
"img" => {
node.remove_attr("decoding");
node.remove_attr("loading");
}
_ => {}
}
let ignored = node.attr("class").is_some_and(|class| {
class
.split_whitespace()
.any(|name| name == "entry-unrelated" || name == "instapaper_ignore")
});
let hidden = node.attr("style").is_some_and(|style| {
style.contains("display:none") || style.contains("display: none")
});
node.remove_attr("style");
let junk = matches!(
name.as_ref(),
"form" | "input" | "textarea" | "select" | "button" | "script" | "style"
);
if ignored || hidden || junk {
strip.push(node);
continue;
}
if node.attr("type").as_deref() == Some("text/css")
|| matches!(
name.as_ref(),
"iframe" | "object" | "embed" | "footer" | "link" | "aside"
)
{
late_strip.push(node);
}
}
Util::strip_nodes(strip);
Util::strip_nodes(
links
.into_iter()
.filter(|node| node.first_child().is_none()),
);
Util::strip_nodes(late_strip);
}
fn repair_url(node: &NodeRef, attribute: &str, article_url: &url::Url) {
if let Some(url) = dom::attr(node, attribute) {
let trimmed_url = url.trim();
if url.starts_with('#') || url.starts_with("\\#") {
return;
}
let is_relative_url = url::Url::parse(&url)
.err()
.map(|err| err == url::ParseError::RelativeUrlWithoutBase)
.unwrap_or(false);
let is_javascript = trimmed_url.contains("javascript:");
if is_relative_url {
let completed_url = match article_url.join(trimmed_url) {
Ok(joined_url) => joined_url,
Err(_) => return,
};
node.set_attr(attribute, completed_url.as_str());
} else if is_javascript {
let child_nodes = node.children();
let child_count = child_nodes.len();
let first_child_is_text = child_nodes.first().is_some_and(NodeRef::is_text);
if node.parent().is_some() {
let new_node = if child_count == 1 && first_child_is_text {
node.tree.new_text(dom::text(node))
} else {
let container = node.tree.new_element("span");
for child in child_nodes {
dom::append_child(&container, &child);
}
container
};
dom::replace_with(node, &new_node);
}
} else if let Ok(parsed_url) = Url::parse(trimmed_url) {
node.set_attr(attribute, parsed_url.as_str());
} else {
node.set_attr(attribute, trimmed_url);
};
}
}
fn repair_srcset(node: &NodeRef, article_url: &url::Url) {
let Some(srcset) = dom::attr(node, "srcset") else {
return;
};
let res = constants::SRC_SET_URL
.captures_iter(&srcset)
.map(|cap| {
let cap0 = cap.get(0).map_or("", |m| m.as_str());
let cap1 = cap.get(1).map_or("", |m| m.as_str());
let cap2 = cap.get(2).map_or("", |m| m.as_str());
let cap3 = cap.get(3).map_or("", |m| m.as_str());
let is_relative_url = url::Url::parse(cap1)
.err()
.map(|err| err == url::ParseError::RelativeUrlWithoutBase)
.unwrap_or(false);
if is_relative_url {
let completed_url = article_url
.join(cap1)
.map(|u| u.as_str().to_owned())
.unwrap_or_default();
format!("{completed_url}{cap2}{cap3}")
} else {
cap0.to_string()
}
})
.collect::<Vec<String>>()
.join(" ");
node.set_attr("srcset", res.as_str());
}
fn fix_urls(document: &Document, url: &Url) {
const URL_ATTRIBUTES: [(&str, &str); 4] = [
("img", "src"),
("a", "href"),
("object", "data"),
("iframe", "src"),
];
const SRCSET_TAGS: [&str; 2] = ["img", "source"];
let nodes = document
.root()
.descendants_it()
.filter_map(|node| dom::local_name(&node).map(|name| (name, node)))
.filter(|(name, _)| {
URL_ATTRIBUTES.iter().any(|(tag, _)| name.as_ref() == *tag)
|| SRCSET_TAGS.contains(&name.as_ref())
})
.collect::<Vec<_>>();
for (name, node) in &nodes {
if SRCSET_TAGS.contains(&name.as_ref()) {
Self::repair_srcset(node, url);
}
}
for (tag, attribute) in URL_ATTRIBUTES {
for (_, node) in nodes.iter().filter(|(name, _)| name.as_ref() == tag) {
Self::repair_url(node, attribute, url);
}
}
}
pub(crate) fn prep_content(
document: &Document,
config: Option<&ConfigEntry>,
global_config: &ConfigEntry,
url: &Url,
title: Option<&str>,
) {
for font_node in dom::elements_by_tag_name(&document.root(), "font") {
font_node.rename("span");
}
Util::mark_data_tables(document);
if let Some(config) = config {
for wrap in &config.wrap_in {
_ = Util::wrap_in(document, &wrap.tag, &wrap.selector);
}
}
for wrap in &global_config.wrap_in {
_ = Util::wrap_in(document, &wrap.tag, &wrap.selector);
}
if let Some(config) = config {
for move_into in &config.move_into {
_ = Util::move_into(document, &move_into.target, &move_into.selector);
}
}
for move_into in &global_config.move_into {
_ = Util::move_into(document, &move_into.target, &move_into.selector);
}
if let Some(config) = config {
for selector in &config.strip {
_ = Util::strip_selector(document, selector);
}
}
for selector in &global_config.strip {
_ = Util::strip_selector(document, selector);
}
if let Some(config) = config {
for selector in &config.strip_attr {
_ = Util::strip_attributes(document, selector);
}
}
for selector in &global_config.strip_attr {
_ = Util::strip_attributes(document, selector);
}
if let Some(config) = config {
for selector in &config.dissolve {
_ = Util::dissolve(document, selector);
}
}
for selector in &global_config.dissolve {
_ = Util::dissolve(document, selector);
}
let site_config = config.into_iter();
Util::strip_id_or_class(
document,
site_config
.clone()
.chain([global_config])
.flat_map(|config| &config.strip_id_or_class)
.map(String::as_str),
);
Util::strip_image_src(
document,
site_config
.chain([global_config])
.flat_map(|config| &config.strip_image_src)
.map(String::as_str),
);
let headings = document.root().descendants_it().filter(|node| {
dom::local_name(node).is_some_and(|name| matches!(name.as_ref(), "h1" | "h2"))
});
for heading in headings.collect::<Vec<_>>() {
if dom::tag_name_is(&heading, "h1") {
heading.rename("h2");
}
if Util::header_duplicates_title(&heading, title) {
heading.remove_from_parent();
}
}
if let Some(attribute) = config
.and_then(|config| config.src_lazy_load_attr.as_deref())
.or(global_config.src_lazy_load_attr.as_deref())
{
Self::apply_lazy_load_attr(document, attribute);
}
Self::unwrap_noscript_images(document);
Util::strip_nodes(dom::elements_by_tag_name(&document.root(), "noscript"));
Self::fix_lazy_images(document);
Self::fix_iframe_size(document, "youtube.com");
Self::strip_junk(document);
let root = document.root();
Util::replace_brs(&root);
Util::replace_emoji_images(&root);
Self::fix_urls(document, url);
}
fn unwrap_noscript_images(document: &Document) {
for img_node in dom::elements_by_tag_name(&document.root(), "img") {
let attrs = dom::attrs(&img_node);
let keep = attrs.iter().any(|(name, value)| {
name == "src"
|| name == "srcset"
|| name == "data-src"
|| name == "data-srcset"
|| constants::IS_IMAGE.is_match(value)
});
if !keep {
img_node.remove_from_parent();
}
}
for noscript_node in dom::elements_by_tag_name(&document.root(), "noscript") {
if !Util::is_single_image(&noscript_node) {
continue;
}
if let Some(prev) = noscript_node.prev_element_sibling()
&& Util::is_single_image(&prev)
{
{
let mut prev_img = prev;
if !dom::tag_name_is(&prev_img, "img")
&& let Some(img_node) = dom::first_element_by_tag_name(&prev_img, "img")
{
prev_img = img_node;
}
let new_img = dom::first_element_by_tag_name(&noscript_node, "img");
if let Some(new_img) = new_img {
for (key, value) in dom::attrs(&prev_img) {
if value.is_empty() {
continue;
}
if key == "src"
|| key == "srcset"
|| constants::IS_IMAGE.is_match(&value)
{
if dom::attr(&new_img, &key).as_deref() == Some(&value) {
continue;
}
let mut attr_name = key;
if new_img.has_attr(&attr_name) {
attr_name = format!("data-old-{attr_name}");
}
new_img.set_attr(&attr_name, &value);
}
}
}
}
if noscript_node.parent().is_some()
&& let Some(first_child) = noscript_node.first_element_child()
{
dom::replace_with(&prev, &first_child);
noscript_node.remove_from_parent();
}
}
}
}
fn extract_body(
document: &Document,
root: &NodeRef,
config: Option<&ConfigEntry>,
global_config: &ConfigEntry,
) -> Result<bool, FullTextParserError> {
let mut found_something = false;
let prune = config
.and_then(|config| config.prune)
.or(global_config.prune)
.unwrap_or(true);
if let Some(config) = config {
for selector in &config.body {
if Self::extract_body_single(document, root, selector, prune)? {
found_something = true;
}
}
for selector in &config.move_into_body {
if Self::extract_body_single(document, root, selector, prune)? {
found_something = true;
}
}
}
if !found_something {
for selector in &global_config.body {
if Self::extract_body_single(document, root, selector, prune)? {
found_something = true;
}
}
}
Ok(found_something)
}
fn extract_body_single(
document: &Document,
root: &NodeRef,
selector: &Selector,
prune: bool,
) -> Result<bool, FullTextParserError> {
let nodes = Util::select(document, selector)?;
let nodes = nodes
.iter()
.filter_map(XNode::as_node)
.filter(|node| node.is_element() || node.is_text())
.copied()
.collect::<Vec<_>>();
let nodes = nodes
.iter()
.filter(|node| !dom::has_ancestor_in(node, &nodes))
.copied()
.collect::<Vec<_>>();
if nodes.is_empty() {
return Ok(false);
}
for node in &nodes {
node.remove_attr("style");
Self::post_process_page(node, prune)?;
}
dom::move_into(root, nodes);
Ok(true)
}
fn check_for_next_page(
&self,
document: &Document,
config: Option<&ConfigEntry>,
global_config: &ConfigEntry,
article_url: &url::Url,
) -> Option<url::Url> {
let autodetect_next_page = Self::autodetect_next_page(config, global_config);
let next_page_url = Self::find_page_link(
document,
config.map(|c| c.next_page_link.as_slice()),
&global_config.next_page_link,
article_url,
)
.or_else(|| {
autodetect_next_page
.then(|| Self::rel_next_link(document, article_url))
.flatten()
});
if let Some(next_page_url) = &next_page_url {
tracing::debug!(%next_page_url);
}
next_page_url
}
fn rel_next_link(document: &Document, article_url: &url::Url) -> Option<url::Url> {
dom::select(document, r#"link[rel~="next"][href], a[rel~="next"][href]"#)
.iter()
.filter_map(|node| dom::attr(node, "href"))
.find_map(|href| {
let href = href.trim();
(!href.is_empty())
.then(|| article_url.join(href).ok())
.flatten()
})
}
fn find_page_link(
document: &Document,
site_rules: Option<&[PageLink]>,
global_rules: &[PageLink],
article_url: &url::Url,
) -> Option<url::Url> {
site_rules
.unwrap_or_default()
.iter()
.chain(global_rules)
.filter(|rule| {
rule.if_page_contains.as_ref().is_none_or(|condition| {
Util::select(document, condition).is_ok_and(|nodes| !nodes.is_empty())
})
})
.find_map(|rule| {
tracing::trace!(selector = %rule.selector, "Page link rule");
Util::find_page_url(document, &rule.selector, article_url)
})
}
fn page_key(url: &Url) -> Url {
let mut url = url.clone();
url.set_fragment(None);
url
}
fn is_native_ad(
document: &Document,
config: Option<&ConfigEntry>,
global_config: &ConfigEntry,
) -> bool {
config
.map(|config| config.native_ad_clue.as_slice())
.unwrap_or_default()
.iter()
.chain(&global_config.native_ad_clue)
.any(|selector| Util::select(document, selector).is_ok_and(|nodes| !nodes.is_empty()))
}
fn insert_detected_image(root: &NodeRef, image: &str) {
if dom::first_element_by_tag_name(root, "img").is_some() {
return;
}
tracing::debug!(image, "Content has no image, inserting the detected one");
let img = root.tree.new_element("img");
img.set_attr("src", image);
root.prepend_child(&img);
}
pub(crate) fn post_process_document(root: &NodeRef) -> Result<(), FullTextParserError> {
Self::simplify_nested_elements(root)?;
Self::clean_attributes(root)?;
Self::remove_single_cell_tables(root);
Self::remove_extra_p_and_div(root);
Ok(())
}
pub(crate) fn post_process_page(
node: &NodeRef,
prune: bool,
) -> Result<(), FullTextParserError> {
if prune {
Util::clean_headers(node);
}
Util::replace_schema_org_orbjects(node);
if prune {
Util::clean_conditionally(node, "fieldset");
Util::clean_conditionally(node, "table");
Util::clean_conditionally(node, "ul");
Util::clean_conditionally(node, "div");
}
Self::clean_page(node)
}
pub(crate) fn post_process_readability_page(node: &NodeRef) -> Result<(), FullTextParserError> {
Self::split_paragraphs_at_blocks(node);
Util::replace_schema_org_orbjects(node);
Self::clean_page(node)
}
pub(crate) fn split_paragraphs_at_blocks(root: &NodeRef) {
for paragraph in dom::elements_by_tag_name(root, "p") {
let mut anchor = paragraph;
while paragraph.parent().is_some()
&& let Some(block) = Self::first_nested_block(¶graph)
{
let inline_chain = block
.ancestors_it(None)
.take_while(|ancestor| ancestor.id != paragraph.id)
.collect::<Vec<_>>();
if inline_chain.is_empty() {
let mut next = Some(block);
while let Some(node) = next {
next = node.next_sibling();
anchor.insert_after(&node);
anchor = node;
}
} else {
anchor.insert_after(&block);
anchor = block;
for inline in inline_chain {
if !dom::tag_name_in(&inline, constants::FORMATTING_TAGS) {
continue;
}
let Some(name) = dom::local_name(&inline) else {
continue;
};
let copy = block.tree.new_element(&name);
for (name, value) in dom::attrs(&inline) {
copy.set_attr(&name, &value);
}
for child in block.children() {
dom::append_child(©, &child);
}
dom::append_child(&block, ©);
}
}
}
}
}
fn first_nested_block<'a>(paragraph: &NodeRef<'a>) -> Option<NodeRef<'a>> {
fn find<'a>(node: &NodeRef<'a>) -> Option<NodeRef<'a>> {
for child in node.element_children() {
if dom::tag_name_in(&child, constants::P_CLOSING_TAGS) {
return Some(child);
}
if let Some(block) = find(&child) {
return Some(block);
}
}
None
}
find(paragraph)
}
fn clean_page(node: &NodeRef) -> Result<(), FullTextParserError> {
Self::remove_share_elements(node);
Self::clean_attributes(node)?;
Self::remove_single_cell_tables(node);
Self::remove_extra_p_and_div(node);
Self::remove_empty_nodes(node);
Ok(())
}
pub(crate) fn remove_single_cell_tables(root: &NodeRef) {
let mut node_iter = Some(*root);
while let Some(node) = node_iter {
if dom::tag_name_is(&node, "table") {
let t_body = if Util::has_single_tag_inside_element(&node, "tbody") {
node.element_children().into_iter().next().unwrap()
} else {
node
};
if Util::has_single_tag_inside_element(&t_body, "tr") {
let row = t_body.element_children().first().copied();
if let Some(row) = row
&& Util::has_single_tag_inside_element(&row, "td")
{
let cell = row.element_children().first().copied();
if let Some(cell) = cell {
let all_phrasing_content = cell
.element_children()
.iter()
.all(Util::is_phrasing_content);
cell.rename(if all_phrasing_content { "p" } else { "div" });
if node.parent().is_some() {
dom::replace_with(&node, &cell);
if node.id == root.id {
Self::remove_single_cell_tables(&cell);
return;
}
node_iter = Some(cell);
continue;
}
}
}
}
}
node_iter = dom::next_node_within(&node, root, false);
}
}
pub(crate) fn remove_extra_p_and_div(root: &NodeRef) {
let mut node_iter = Some(*root);
while let Some(node) = node_iter {
if dom::tag_name_in(&node, &["p", "div"])
&& !dom::has_text_or_descendant_tag(
&node,
&["img", "video", "embed", "object", "iframe"],
)
{
node_iter = dom::remove_and_next_within(&node, root);
continue;
}
node_iter = dom::next_node_within(&node, root, false);
}
}
pub(crate) fn remove_share_elements(root: &NodeRef) {
let mut node_iter = Some(*root);
while let Some(node) = node_iter {
let match_string = format!(
"{} {}",
dom::attr(&node, "class").unwrap_or_default(),
dom::attr(&node, "id").unwrap_or_default()
);
if constants::SHARE_ELEMENTS.is_match(&match_string)
&& node.text().len() < constants::DEFAULT_CHAR_THRESHOLD
{
node_iter = dom::remove_and_next_within(&node, root);
} else {
node_iter = dom::next_node_within(&node, root, false);
}
}
}
pub(crate) fn clean_attributes(root: &NodeRef) -> Result<(), FullTextParserError> {
let mut node_iter = Some(*root);
while let Some(node) = node_iter {
if !node.is_element() {
node_iter = dom::next_node_within(&node, root, false);
continue;
}
node.remove_attrs(constants::PRESENTATIONAL_ATTRIBUTES);
if dom::tag_name_in(&node, constants::DEPRECATED_SIZE_ATTRIBUTE_ELEMS) {
node.remove_attrs(&["width", "height"]);
}
node.remove_attrs(&[
"class",
"align",
constants::SCORE_ATTR,
constants::DATA_TABLE_ATTR,
]);
if dom::tag_name_is(&node, "a")
&& dom::attr(&node, "href")
.map(|href| !href.starts_with('#'))
.unwrap_or(false)
{
node.set_attr("target", "_blank");
}
node_iter = dom::next_node_within(&node, root, false);
}
Ok(())
}
fn simplify_nested_elements(root: &NodeRef) -> Result<(), FullTextParserError> {
let mut node_iter = Some(*root);
while let Some(node) = node_iter {
if dom::tag_name_is(&node, "article") || node.parent().is_none() {
node_iter = dom::next_node_within(&node, root, false);
continue;
}
if !dom::tag_name_in(&node, &["div", "section"]) {
node_iter = dom::next_node_within(&node, root, false);
continue;
}
if Util::is_element_without_content(&node) {
node_iter = dom::remove_and_next_within(&node, root);
continue;
} else if (Util::has_single_tag_inside_element(&node, "div")
|| Util::has_single_tag_inside_element(&node, "section"))
&& let Some(parent) = node.parent()
&& let Some(child) = node.element_children().into_iter().next()
{
for (k, v) in dom::attrs(&node) {
child.set_attr(&k, &v);
}
dom::replace_with(&node, &child);
node_iter = if node.id == root.id {
None
} else {
dom::next_node_within(&parent, root, false)
};
continue;
}
node_iter = dom::next_node_within(&node, root, false);
}
Ok(())
}
pub(crate) fn remove_empty_nodes(root: &NodeRef) {
let mut node_iter = Some(*root);
while let Some(node) = node_iter {
if dom::tag_name_in(&node, constants::VALID_EMPTY_TAGS) {
node_iter = dom::next_node_within(&node, root, false);
continue;
}
if Util::is_element_without_children(&node) {
node_iter = dom::remove_and_next_within(&node, root);
continue;
}
node_iter = dom::next_node_within(&node, root, false);
}
}
}