use dom_query::{Document, NodeRef};
use dom_query_xpath::{Value, XNode};
use reqwest::Response;
#[cfg(feature = "image-downloader")]
use reqwest::header::{CONTENT_LENGTH, CONTENT_TYPE};
use reqwest::header::{HeaderMap, HeaderName, HeaderValue};
use std::ffi::OsStr;
use tokio::fs::DirEntry;
#[cfg(feature = "image-downloader")]
use crate::error::ImageDownloadError;
use crate::{
constants::{self, NEGATIVE_LEAD_IMAGE_URL_HINTS_REGEX},
dom,
error::FullTextParserError,
full_text_parser::config::ConfigEntry,
image_object::ImageObject,
selector::Selector,
video_object::VideoObject,
};
pub struct Util;
impl Util {
pub fn check_extension(path: &DirEntry, extension: &str) -> bool {
path.path()
.extension()
.and_then(OsStr::to_str)
.map(|ext| ext == extension)
.unwrap_or(false)
}
pub fn str_extract_value<'a>(identifier: &str, line: &'a str) -> &'a str {
line.strip_prefix(identifier).unwrap_or_default().trim()
}
pub fn generate_headers(
site_specific_rule: Option<&ConfigEntry>,
global_rule: &ConfigEntry,
) -> Result<HeaderMap, FullTextParserError> {
let mut headers = HeaderMap::new();
if let Some(config) = site_specific_rule {
for header in &config.header {
tracing::trace!(header.name, header.value, "site specific header");
let name = HeaderName::from_bytes(header.name.as_bytes())
.map_err(|_| FullTextParserError::Config)?;
let value = header
.value
.parse::<HeaderValue>()
.map_err(|_| FullTextParserError::Config)?;
headers.insert(name, value);
}
}
for header in &global_rule.header {
tracing::trace!(header.name, header.value, "global header");
let name = HeaderName::from_bytes(header.name.as_bytes())
.map_err(|_| FullTextParserError::Config)?;
let value = header
.value
.parse::<HeaderValue>()
.map_err(|_| FullTextParserError::Config)?;
headers.insert(name, value);
}
Ok(headers)
}
pub fn find_page_url(
document: &Document,
selector: &Selector,
base_url: &url::Url,
) -> Option<url::Url> {
let resolve = |url_str: &str| {
let url_str = url_str.trim();
if url_str.is_empty() {
return None;
}
base_url.join(url_str).ok()
};
match Self::evaluate(document, selector).ok()? {
Value::NodeSet(nodes) => nodes.iter().find_map(|node| {
let url_str = match node {
XNode::Node(element) if element.is_element() => {
dom::attr(element, "href").unwrap_or_else(|| dom::text(element))
}
other => other.string_value(),
};
resolve(&url_str)
}),
Value::String(url_str) => resolve(&url_str),
other => {
tracing::debug!(%selector, result = other.type_name(), "Page link rule returns neither nodes nor a string");
None
}
}
}
pub fn evaluate<'a>(
document: &'a Document,
selector: &Selector,
) -> Result<Value<'a>, FullTextParserError> {
Ok(selector.evaluate(document).inspect_err(|error| {
tracing::debug!(%selector, %error, "Evaluating selector failed");
})?)
}
pub fn select<'a>(
document: &'a Document,
selector: &Selector,
) -> Result<Vec<XNode<'a>>, FullTextParserError> {
let nodes = selector.select(document).inspect_err(|error| {
tracing::debug!(%selector, %error, "Evaluating selector failed");
})?;
if nodes.is_empty() {
tracing::debug!(%selector, "Selector yielded no results");
}
Ok(nodes)
}
pub fn check_content_type(response: &Response) -> Result<bool, FullTextParserError> {
if !response.status().is_success() {
tracing::error!(status = %response.status(), "Response is not a success");
return Err(FullTextParserError::Http);
}
if let Some(content_type) = response.headers().get(reqwest::header::CONTENT_TYPE)
&& let Ok(content_type) = content_type.to_str()
{
let media_type = content_type.split(';').next().unwrap_or_default().trim();
if media_type.eq_ignore_ascii_case("text/html")
|| media_type.eq_ignore_ascii_case("application/xhtml+xml")
{
tracing::debug!(content_type);
return Ok(true);
}
}
tracing::error!("Content type is not HTML or XHTML");
Ok(false)
}
pub fn get_attribute(
document: &Document,
tag: &str,
(filter_attribute, needle): (&str, &str),
attribute: &str,
) -> Option<String> {
dom::select(document, tag)
.iter()
.filter(|node| dom::attr(node, filter_attribute).is_some_and(|v| v.contains(needle)))
.find_map(|node| dom::attr(node, attribute))
}
pub fn extract_value(
document: &Document,
selector: &Selector,
) -> Result<String, FullTextParserError> {
let value = match Self::evaluate(document, selector)? {
Value::NodeSet(nodes) => nodes.first().map(XNode::string_value),
Value::String(value) => Some(value),
other => {
tracing::debug!(%selector, result = other.type_name(), "Selector returns neither nodes nor a string");
None
}
};
value
.map(|value| value.trim().to_string())
.filter(|value| !value.is_empty())
.ok_or(FullTextParserError::NotFound)
}
pub fn extract_title(
document: &Document,
selector: &Selector,
) -> Result<String, FullTextParserError> {
let title = Util::extract_value(document, selector)?;
Ok(title.split_whitespace().collect::<Vec<_>>().join(" "))
}
pub fn strip_selector(
document: &Document,
selector: &Selector,
) -> Result<(), FullTextParserError> {
for node in Self::select(document, selector)? {
match node {
XNode::Node(node) => Self::strip_nodes([node]),
XNode::Attribute { .. } => node.remove(),
}
}
Ok(())
}
pub fn strip_attributes(
document: &Document,
selector: &Selector,
) -> Result<(), FullTextParserError> {
Self::strip_attributes_at(&document.root(), selector)
}
pub fn strip_attributes_at(
context: &NodeRef,
selector: &Selector,
) -> Result<(), FullTextParserError> {
let nodes = match selector.evaluate_at(context).inspect_err(|error| {
tracing::debug!(%selector, %error, "Evaluating selector failed");
})? {
Value::NodeSet(nodes) => nodes,
_ => Vec::new(),
};
for node in nodes {
match node {
XNode::Attribute { .. } => node.remove(),
XNode::Node(_) => {
tracing::debug!(%selector, "strip_attr rule matches a node, not an attribute");
}
}
}
Ok(())
}
pub fn wrap_in(
document: &Document,
tag: &str,
selector: &Selector,
) -> Result<(), FullTextParserError> {
for node in Self::select(document, selector)? {
let XNode::Node(node) = node else {
continue;
};
if !(node.is_element() || node.is_text()) || node.parent().is_none() {
continue;
}
let wrapper = node.tree.new_element(tag);
dom::replace_with(&node, &wrapper);
dom::append_child(&wrapper, &node);
}
Ok(())
}
pub fn move_into(
document: &Document,
target: &Selector,
selector: &Selector,
) -> Result<(), FullTextParserError> {
let Some(target) = Self::select(document, target)?
.into_iter()
.find_map(|node| node.as_node().copied().filter(NodeRef::is_element))
else {
return Ok(());
};
for node in Self::select(document, selector)? {
let XNode::Node(node) = node else {
continue;
};
if !(node.is_element() || node.is_text())
|| node.id == target.id
|| dom::has_ancestor_in(&target, &[node])
{
continue;
}
dom::append_child(&target, &node);
}
Ok(())
}
pub fn dissolve(document: &Document, selector: &Selector) -> Result<(), FullTextParserError> {
for node in Self::select(document, selector)? {
let XNode::Node(element) = node else {
continue;
};
if !element.is_element() || element.parent().is_none() {
continue;
}
for child in element.children() {
element.insert_before(&child);
}
element.remove_from_parent();
}
Ok(())
}
pub fn strip_nodes<'a>(nodes: impl IntoIterator<Item = NodeRef<'a>>) {
for node in nodes {
if dom::tag_name_in(&node, constants::EMBED_TAG_NAMES)
&& dom::attrs(&node)
.iter()
.any(|(_name, value)| constants::VIDEOS.is_match(value))
{
continue;
}
node.remove_from_parent();
}
}
fn strip_value(value: &str) -> Option<String> {
let value = value.replace(['\'', '"'], "");
let value = value.trim();
(!value.is_empty()).then(|| value.to_string())
}
pub fn strip_id_or_class<'a>(document: &Document, values: impl IntoIterator<Item = &'a str>) {
let values: Vec<String> = values.into_iter().filter_map(Self::strip_value).collect();
if values.is_empty() {
return;
}
for node in dom::elements_by_tag_name(&document.root(), "*") {
let matches = ["class", "id"].iter().any(|attr| {
node.attr(attr)
.is_some_and(|v| values.iter().any(|value| v.contains(value.as_str())))
});
if matches {
node.remove_from_parent();
}
}
}
pub fn strip_image_src<'a>(document: &Document, values: impl IntoIterator<Item = &'a str>) {
let values: Vec<String> = values.into_iter().filter_map(Self::strip_value).collect();
if values.is_empty() {
return;
}
for node in dom::elements_by_tag_name(&document.root(), "img") {
if node
.attr("src")
.is_some_and(|src| values.iter().any(|value| src.contains(value.as_str())))
{
node.remove_from_parent();
}
}
}
pub fn get_signature(node: &NodeRef) -> String {
let match_string = dom::attr(node, "class")
.unwrap_or_default()
.split_whitespace()
.fold(String::new(), |a, b| format!("{a} {b}"));
match dom::attr(node, "id") {
Some(id) => format!("{match_string} {id}"),
None => match_string,
}
}
pub fn get_inner_text(node: &NodeRef, normalize_spaces: bool) -> String {
let content = node.text().trim().to_owned();
if normalize_spaces {
constants::NORMALIZE.replace_all(&content, " ").into()
} else {
content
}
}
pub fn text_similarity(a: &str, b: &str) -> f64 {
let a = a.to_lowercase();
let b = b.to_lowercase();
let tokens_a = constants::TOKENIZE
.split(&a)
.filter(|token| !token.is_empty())
.collect::<Vec<_>>();
let tokens_b = constants::TOKENIZE
.split(&b)
.filter(|token| !token.is_empty())
.collect::<Vec<_>>();
if tokens_a.is_empty() || tokens_b.is_empty() {
return 0.0;
}
let tokens_b_total = tokens_b.join(" ").len() as f64;
let uniq_tokens_b = tokens_b
.into_iter()
.filter(|token| !tokens_a.iter().any(|t| t == token))
.collect::<Vec<_>>();
let uniq_tokens_b_total = uniq_tokens_b.join(" ").len() as f64;
let distance_b = uniq_tokens_b_total / tokens_b_total;
1.0 - distance_b
}
pub fn header_duplicates_title(node: &NodeRef, title: Option<&str>) -> bool {
if !dom::tag_name_is(node, "h1") && !dom::tag_name_is(node, "h2") {
return false;
}
let heading = Util::get_inner_text(node, false);
if let Some(title) = title {
Util::text_similarity(title, &heading) > 0.75
} else {
false
}
}
pub fn has_ancestor_tag<F>(
node: &NodeRef,
tag_name: &str,
max_depth: Option<u64>,
filter: Option<F>,
) -> bool
where
F: Fn(&NodeRef) -> bool,
{
let max_depth = max_depth.unwrap_or(3);
let mut depth = 0;
let mut node = node.parent();
loop {
if depth > max_depth {
return false;
}
let tmp_node = match node {
Some(node) => node,
None => return false,
};
if dom::tag_name_is(&tmp_node, tag_name)
&& filter
.as_ref()
.map(|filter| filter(&tmp_node))
.unwrap_or(true)
{
return true;
}
node = tmp_node.parent();
depth += 1;
}
}
pub fn has_single_tag_inside_element(node: &NodeRef, tag: &str) -> bool {
let children = node.element_children();
if children.len() != 1
|| children
.first()
.map(|n| !dom::tag_name_is(n, tag))
.unwrap_or(false)
{
return false;
}
!node
.children_it(false)
.any(|n| n.is_text() && constants::HAS_CONTENT.is_match(&n.text()))
}
pub fn is_element_without_content(node: &NodeRef) -> bool {
if !node.is_element() {
return false;
}
let len = node.children_it(false).count();
(len == 0
|| len
== dom::elements_by_tag_name(node, "br").len()
+ dom::elements_by_tag_name(node, "hr").len())
&& node.text().trim().is_empty()
}
pub fn is_element_without_children(node: &NodeRef) -> bool {
if !node.is_element() {
return false;
}
!dom::has_text_or_descendant_tag(node, constants::VALID_EMPTY_TAGS)
}
pub fn get_link_density(node: &NodeRef) -> f64 {
let text_length = Util::get_inner_text(node, true).len();
if text_length == 0 {
return 0.0;
}
let mut link_length = 0.0;
let link_nodes = dom::elements_by_tag_name(node, "a");
for link_node in link_nodes {
if let Some(href) = dom::attr(&link_node, "href") {
let coefficient = if constants::HASH_URL.is_match(&href) {
0.3
} else {
1.0
};
link_length += Util::get_inner_text(&link_node, true).len() as f64 * coefficient;
}
}
link_length / text_length as f64
}
pub fn has_tag_name(node: Option<&NodeRef>, tag_name: &str) -> bool {
node.map(|n| dom::tag_name_is(n, tag_name)).unwrap_or(false)
}
pub fn is_single_image(node: &NodeRef) -> bool {
if dom::tag_name_is(node, "img") {
true
} else if node.element_children().len() != 1 || node.text().trim() != "" {
false
} else if let Some(first_child) = node.element_children().first() {
Self::is_single_image(first_child)
} else {
false
}
}
pub fn clean_headers(root: &NodeRef) {
let mut nodes = dom::elements_by_tag_name(root, "h1");
nodes.append(&mut dom::elements_by_tag_name(root, "h2"));
for node in nodes.into_iter().rev() {
if Util::get_class_weight(&node) < 0 {
let name = dom::tag_name(&node);
let class = dom::attr(&node, "class").unwrap_or_default();
tracing::trace!(name, class, "Removing header with low class weight",);
node.remove_from_parent();
}
}
}
pub fn has_schema_org_type(node: &NodeRef, type_name: &str) -> bool {
dom::attr(node, "itemtype").is_some_and(|item_type| {
item_type.split_whitespace().any(|item_type| {
item_type
.strip_prefix("https://")
.or_else(|| item_type.strip_prefix("http://"))
.and_then(|item_type| item_type.strip_prefix("schema.org/"))
== Some(type_name)
})
})
}
pub fn replace_schema_org_orbjects(root: &NodeRef) {
let nodes = dom::elements_by_tag_name(root, "div");
for node in nodes.into_iter().rev() {
if let Some(video_object) = VideoObject::parse_node(&node) {
_ = video_object.replace(&node);
} else if let Some(image_object) = ImageObject::parse_node(&node) {
_ = image_object.replace(&node);
}
}
}
pub fn replace_emoji_images(root: &NodeRef) {
let img_nodes = dom::elements_by_tag_name(root, "img");
for img_node in img_nodes {
if let Some(img_alt) = dom::attr(&img_node, "alt")
&& Self::is_emoji(&img_alt)
&& img_node.parent().is_some()
{
let emoji_text_node = img_node.tree.new_text(img_alt);
dom::replace_with(&img_node, &emoji_text_node);
}
}
}
pub fn is_emoji(text: &str) -> bool {
let mut alt_chars = text.chars();
let first_char = alt_chars.next();
let second_char = alt_chars.next();
if let (Some(char), None) = (first_char, second_char) {
unic_emoji_char::is_emoji(char)
} else {
false
}
}
pub fn clean_conditionally(root: &NodeRef, tag: &str) {
let nodes = dom::elements_by_tag_name(root, tag);
for node in nodes.into_iter().rev() {
if Self::should_remove(&node, tag) {
node.remove_from_parent();
}
}
}
fn should_remove(node: &NodeRef, tag: &str) -> bool {
let mut is_list = tag == "ul" || tag == "ol";
if !is_list {
let mut list_length = 0.0;
let ul_nodes = dom::elements_by_tag_name(node, "ul");
let ol_nodes = dom::elements_by_tag_name(node, "ol");
for list_node in ul_nodes {
list_length += Util::get_inner_text(&list_node, false).len() as f64;
}
for list_node in ol_nodes {
list_length += Util::get_inner_text(&list_node, false).len() as f64;
}
is_list = (list_length / Util::get_inner_text(node, false).len() as f64) > 0.9;
}
if tag == "table" && Self::is_data_table(node) {
return false;
}
if Self::has_ancestor_tag(node, "table", Some(u64::MAX), Some(Self::is_data_table)) {
return false;
}
if Self::has_ancestor_tag(node, "code", None, None::<fn(&NodeRef) -> bool>) {
return false;
}
let weight = Self::get_class_weight(node);
if weight < 0 {
return true;
}
if Self::get_char_count(node, ',') < 10 {
let p = dom::elements_by_tag_name(node, "p").len();
let img = dom::elements_by_tag_name(node, "img").len();
let li = dom::elements_by_tag_name(node, "li").len() as i64 - 100;
let input = dom::elements_by_tag_name(node, "input").len();
let heading_density =
Self::get_text_density(node, &["h1", "h2", "h3", "h4", "h5", "h6"]);
let mut embed_count = 0;
let embed_tags = ["object", "embed", "iframe"];
for embed_tag in embed_tags {
for embed_node in dom::elements_by_tag_name(node, embed_tag) {
for (_name, value) in dom::attrs(&embed_node) {
if constants::VIDEOS.is_match(&value) {
return false;
}
}
embed_count += 1;
}
}
let link_density = Self::get_link_density(node);
let content = Self::get_inner_text(node, true);
let content_length = content.len();
let has_figure_ancestor =
Self::has_ancestor_tag(node, "figure", None, None::<fn(&NodeRef) -> bool>);
let image_obj_count = dom::elements_by_tag_name(node, "imageobject").len();
let video_obj_count = dom::elements_by_tag_name(node, "videoobject").len();
let video_tag_count = dom::elements_by_tag_name(node, "video").len();
if image_obj_count > 0 || video_obj_count > 0 || video_tag_count > 0 {
return false;
}
let have_to_remove = (img > 1 && (p as f64 / img as f64) < 0.5 && !has_figure_ancestor)
|| (!is_list && li > p as i64)
|| (input as f64 > f64::floor(p as f64 / 3.0))
|| (!is_list
&& heading_density < 0.9
&& content_length < 25
&& (img == 0 || img > 2)
&& !has_figure_ancestor)
|| (!is_list && weight < 25 && link_density > 0.2)
|| (weight >= 25 && link_density > 0.5)
|| ((embed_count == 1 && content_length < 75) || embed_count > 1);
if is_list && have_to_remove {
for child in node.element_children() {
if child.element_children().len() > 1 {
return have_to_remove;
}
}
let li_count = dom::elements_by_tag_name(node, "li").len();
if img == li_count {
return false;
}
}
have_to_remove
} else {
false
}
}
pub fn get_class_weight(node: &NodeRef) -> i64 {
let mut weight = 0;
if let Some(class_names) = dom::attr(node, "class") {
if constants::NEGATIVE.is_match(&class_names) {
weight -= 25;
}
if constants::POSITIVE.is_match(&class_names) {
weight += 25;
}
}
if let Some(class_names) = dom::attr(node, "id") {
if constants::NEGATIVE.is_match(&class_names) {
weight -= 25;
}
if constants::POSITIVE.is_match(&class_names) {
weight += 25;
}
}
weight
}
fn get_char_count(node: &NodeRef, char: char) -> usize {
Util::get_inner_text(node, false).split(char).count() - 1
}
fn get_text_density(node: &NodeRef, tags: &[&str]) -> f64 {
let text_length = Util::get_inner_text(node, false).len();
if text_length == 0 {
return 0.0;
}
let mut children_length = 0;
for tag in tags {
for child in dom::elements_by_tag_name(node, tag) {
children_length += Util::get_inner_text(&child, false).len()
}
}
children_length as f64 / text_length as f64
}
fn is_data_table(node: &NodeRef) -> bool {
dom::attr(node, constants::DATA_TABLE_ATTR)
.and_then(|is_data_table| is_data_table.parse::<bool>().ok())
.unwrap_or(false)
}
pub fn mark_data_tables(document: &Document) {
for node in dom::elements_by_tag_name(&document.root(), "table") {
if dom::attr(&node, "role")
.map(|role| role == "presentation")
.unwrap_or(false)
{
node.set_attr(constants::DATA_TABLE_ATTR, "false");
continue;
}
if dom::attr(&node, "datatable")
.map(|role| role == "0")
.unwrap_or(false)
{
node.set_attr(constants::DATA_TABLE_ATTR, "false");
continue;
}
if node.has_attr("summary") {
node.set_attr(constants::DATA_TABLE_ATTR, "true");
continue;
}
if let Some(first_caption) = dom::elements_by_tag_name(&node, "caption").first()
&& first_caption.first_child().is_some()
{
node.set_attr(constants::DATA_TABLE_ATTR, "true");
continue;
}
let data_table_descendants = ["col", "colgroup", "tfoot", "thead", "th"];
if data_table_descendants
.iter()
.any(|descendant| !dom::elements_by_tag_name(&node, descendant).is_empty())
{
node.set_attr(constants::DATA_TABLE_ATTR, "true");
continue;
}
if !dom::elements_by_tag_name(&node, "table").is_empty() {
node.set_attr(constants::DATA_TABLE_ATTR, "false");
continue;
}
let (rows, columns) = Self::get_row_and_column_count(&node);
if rows >= 10 || columns > 4 {
node.set_attr(constants::DATA_TABLE_ATTR, "true");
continue;
}
node.set_attr(
constants::DATA_TABLE_ATTR,
if rows * columns > 10 { "true" } else { "false" },
);
}
}
pub fn get_row_and_column_count(node: &NodeRef) -> (usize, usize) {
if !dom::tag_name_is(node, "table") {
return (0, 0);
}
let mut rows = 0;
let mut columns = 0;
let trs = dom::elements_by_tag_name(node, "tr");
for tr in trs {
let row_span = dom::attr(&tr, "rowspan")
.and_then(|span| span.parse::<usize>().ok())
.unwrap_or(1);
rows += row_span;
let mut columns_in_this_row = 0;
let cells = dom::elements_by_tag_name(&tr, "td");
for cell in cells {
let colspan = dom::attr(&cell, "colspan")
.and_then(|span| span.parse::<usize>().ok())
.unwrap_or(1);
columns_in_this_row += colspan;
}
columns = usize::max(columns, columns_in_this_row);
}
(rows, columns)
}
pub fn is_phrasing_content(node: &NodeRef) -> bool {
node.is_text()
|| dom::tag_name_in(node, constants::PHRASING_ELEMS)
|| (dom::tag_name_in(node, &["a", "del", "ins"])
&& node
.children_it(false)
.all(|n| Self::is_phrasing_content(&n)))
}
pub fn replace_brs(node: &NodeRef) {
let br_nodes = dom::elements_by_tag_name(node, "br");
for br_node in br_nodes {
let mut next = br_node.next_sibling();
let mut replaced = false;
while let Some(n) = next {
let is_text_whitespace = dom::is_whitespace_text(&n);
let is_br_node = dom::tag_name_is(&n, "br");
let next_is_br_node = n
.next_sibling()
.map(|n| dom::tag_name_is(&n, "br"))
.unwrap_or(false);
if !is_text_whitespace && !is_br_node {
break;
}
next = n.next_sibling();
if is_br_node || (is_text_whitespace && next_is_br_node) {
replaced = true;
n.remove_from_parent();
}
}
if !replaced {
continue;
}
if br_node.parent().is_none() {
continue;
}
let p = br_node.tree.new_element("p");
dom::replace_with(&br_node, &p);
next = p.next_sibling();
while let Some(next_node) = next {
if dom::tag_name_is(&next_node, "br")
&& let Some(next_elem) = next_node.next_element_sibling()
&& dom::tag_name_is(&next_elem, "br")
{
break;
}
if !Self::is_phrasing_content(&next_node) {
break;
}
let sibling = next_node.next_sibling();
dom::append_child(&p, &next_node);
next = sibling;
}
if p.element_children().is_empty() && p.text().trim().is_empty() {
p.remove_from_parent();
continue;
}
while let Some(last_child) = p.last_child() {
if dom::is_whitespace_text(&last_child) {
last_child.remove_from_parent();
} else {
break;
}
}
if let Some(parent) = p.parent()
&& dom::tag_name_is(&parent, "p")
{
parent.rename("div");
}
}
}
pub fn score_image_url(url: &str) -> i32 {
let url = url.trim();
let mut score = 0;
if constants::POSITIVE_LEAD_IMAGE_URL_HINTS_REGEX.is_match(url) {
score += 20;
}
if NEGATIVE_LEAD_IMAGE_URL_HINTS_REGEX.is_match(url) {
score -= 20;
}
if constants::GIF_REGEX.is_match(url) {
score -= 10;
}
if constants::JPG_REGEX.is_match(url) {
score += 10;
}
score
}
pub fn score_img_attr(img: &NodeRef) -> i32 {
if img.has_attr("alt") { 5 } else { 0 }
}
pub fn score_by_parents(img: &NodeRef) -> i32 {
let mut score = 0;
let parent = img.parent();
let grand_parent = parent.as_ref().and_then(|n| n.parent());
if Self::has_tag_name(parent.as_ref(), "figure")
|| Self::has_tag_name(grand_parent.as_ref(), "figure")
{
score += 25;
}
if let Some(parent) = parent.as_ref() {
let signature = Util::get_signature(parent);
if constants::PHOTO_HINTS_REGEX.is_match(&signature) {
score += 15;
}
}
if let Some(grand_parent) = grand_parent.as_ref() {
let signature = Util::get_signature(grand_parent);
if constants::PHOTO_HINTS_REGEX.is_match(&signature) {
score += 15;
}
}
score
}
pub fn score_by_sibling(img: &NodeRef) -> i32 {
let mut score = 0;
let sibling = img.next_element_sibling();
if let Some(sibling) = sibling.as_ref() {
if dom::tag_name_is(sibling, "figcaption") {
score += 25;
}
let signature = Util::get_signature(sibling);
if constants::PHOTO_HINTS_REGEX.is_match(&signature) {
score += 15;
}
}
score
}
pub fn score_by_dimensions(img: &NodeRef) -> i32 {
let mut score = 0;
let width = dom::attr(img, "width").and_then(|w| w.parse::<f32>().ok());
let height = dom::attr(img, "height").and_then(|w| w.parse::<f32>().ok());
let src = dom::attr(img, "src").unwrap_or_default();
if let Some(width) = width
&& width <= 50.0
{
score -= 50;
}
if let Some(height) = height
&& height <= 50.0
{
score -= 50;
}
if let Some(width) = width
&& let Some(height) = height
&& !src.contains("sprite")
{
let area = width * height;
if area < 5000.0 {
score -= 100;
} else {
score += f32::round(area / 1000.0) as i32;
}
}
score
}
pub fn score_by_position(len: usize, index: usize) -> i32 {
((len as f32 / 2.0) - index as f32) as i32
}
pub fn score_by_alt(node: &NodeRef) -> i32 {
if let Some(alt) = dom::attr(node, "alt") {
if Self::is_emoji(&alt) { -100 } else { 0 }
} else {
0
}
}
#[cfg(feature = "image-downloader")]
pub fn get_content_length(response: &Response) -> Result<usize, ImageDownloadError> {
let status_code = response.status();
if !status_code.is_success() {
let url = response.url();
tracing::warn!(%url, ?status_code, "response");
return Err(ImageDownloadError::Http);
}
response
.headers()
.get(CONTENT_LENGTH)
.and_then(|content_length| content_length.to_str().ok())
.and_then(|content_length| content_length.parse::<usize>().ok())
.ok_or(ImageDownloadError::ContentLength)
}
#[cfg(feature = "image-downloader")]
pub fn get_content_type(response: &Response) -> Result<String, ImageDownloadError> {
let status_code = response.status();
if !status_code.is_success() {
let url = response.url();
tracing::warn!(%url, ?status_code, "response");
return Err(ImageDownloadError::Http);
}
response
.headers()
.get(CONTENT_TYPE)
.and_then(|val| val.to_str().ok())
.map(|val| val.to_string())
.ok_or(ImageDownloadError::ContentType)
}
}
#[cfg(test)]
mod tests {
use super::Util;
use crate::{dom, selector::Selector, test_util::assert_html_eq};
use dom_query::Document;
fn replace_brs(source: &str, expected: &str) {
let document = Document::from(source);
let div = dom::select(&document, "body > div")[0];
Util::replace_brs(&document.root());
let result = div.html();
assert_html_eq(expected, &result);
}
#[test]
fn replace_brs_1() {
replace_brs(
"<div>foo<br>bar<br> <br><br>abc</div>",
"<div>foo<br/>bar<p>abc</p></div>",
)
}
#[test]
fn replace_brs_2() {
let source = r#"
<div>
<p>
It might have been curiosity or it might have been the nagging sensation that chewed at his brain for the three weeks that he researched the subject of the conversation. All For One was a cryptid. Mystical in more ways than one, he was only a rumour on a network that was two-hundred years old. There were whispers of a shadowy figure who once ruled Japan, intermingled with a string of conspiracies and fragmented events.
</p>
<p>
Izuku had even braved the dark web, poking and prodding at some of the seedier elements of the world wide web. The internet had rumours, but the dark web had stories.<br/>
</p>
<p>
An implied yakuza wrote about his grandfather who lost a fire manipulation Quirk and his sanity without any reason. His grandfather had been institutionalised, crying and repeating βhe took it, he took itβ until his dying days. No one could console him.
</p>
</div>
"#;
replace_brs(source, source.trim())
}
fn page_url(source: &str, xpath: &str) -> Option<String> {
let document = Document::from(source);
let base = url::Url::parse("https://example.com/articles/1").unwrap();
let selector = Selector::xpath(xpath).unwrap();
Util::find_page_url(&document, &selector, &base).map(|url| url.to_string())
}
fn strip(source: &str, strip: impl Fn(&Document)) -> String {
let document = Document::from(source);
strip(&document);
document.body().unwrap().html().to_string()
}
#[test]
fn config_values_may_contain_a_hash() {
assert_eq!(
Util::str_extract_value("strip:", "strip: //a[@href='#related'] "),
"//a[@href='#related']"
);
assert_eq!(
Util::str_extract_value("replace_string(", "replace_string(>permalink</a>): >#</a>"),
">permalink</a>): >#</a>"
);
}
#[test]
fn strip_id_or_class_quoted() {
let source = r#"<html><body>
<div class="mod-paywall"><p class="paywall-text">gone</p></div>
<div class="article"><p>kept</p></div>
</body></html>"#;
let html = strip(source, |document| {
Util::strip_id_or_class(document, ["'paywall'", "''"]);
});
assert!(!html.contains("gone"), "{html}");
assert!(html.contains("kept"), "{html}");
}
#[test]
fn strip_image_src_quoted() {
let source = r#"<html><body>
<img src="/_images/ad.png"><img src="/media/photo.jpg"><img alt="no src">
</body></html>"#;
let html = strip(source, |document| {
Util::strip_image_src(document, ["'/_images/'"]);
});
assert!(!html.contains("ad.png"), "{html}");
assert!(html.contains("photo.jpg"), "{html}");
assert!(html.contains("no src"), "{html}");
}
#[test]
fn strip_selector_keeps_video_embeds_and_removes_attributes() {
let source = r#"<html><body>
<iframe src="https://www.youtube.com/embed/x"></iframe>
<iframe src="https://ads.example.com/"></iframe>
<a href="/x" onclick="track()">link</a>
</body></html>"#;
let html = strip(source, |document| {
Util::strip_selector(document, &Selector::xpath("//iframe").unwrap()).unwrap();
Util::strip_selector(document, &Selector::xpath("//a/@onclick").unwrap()).unwrap();
});
assert!(html.contains("youtube.com"), "{html}");
assert!(!html.contains("ads.example.com"), "{html}");
assert!(html.contains("href") && !html.contains("onclick"), "{html}");
}
#[test]
fn strip_attributes_ignores_element_matches() {
let source = r#"<html><body>
<img src="a.jpg" srcset="a-2x.jpg 2x" loading="lazy">
</body></html>"#;
let html = strip(source, |document| {
Util::strip_attributes(document, &Selector::xpath("//img/@srcset").unwrap()).unwrap();
Util::strip_attributes(document, &Selector::xpath("//img").unwrap()).unwrap();
});
assert!(html.contains(r#"src="a.jpg""#), "{html}");
assert!(
html.contains("loading") && !html.contains("srcset"),
"{html}"
);
}
#[test]
fn wrap_in_wraps_elements_and_text() {
let source = r#"<html><body>
<div class="quote"><p>a</p></div><div class="quote"><div class="quote">b</div></div>
<figure><img src="x.jpg"><div class="text">caption</div></figure>
</body></html>"#;
let html = strip(source, |document| {
let quote = Selector::xpath("//div[@class='quote']").unwrap();
Util::wrap_in(document, "blockquote", "e).unwrap();
let caption = Selector::xpath("//div[@class='text']/text()").unwrap();
Util::wrap_in(document, "figcaption", &caption).unwrap();
});
crate::test_util::assert_html_eq(
r#"<html><body>
<blockquote><div class="quote"><p>a</p></div></blockquote>
<blockquote><div class="quote"><blockquote><div class="quote">b</div></blockquote></div></blockquote>
<figure><img src="x.jpg"><div class="text"><figcaption>caption</figcaption></div></figure>
</body></html>"#,
&html,
);
}
#[test]
fn move_into_appends_to_the_first_target() {
let source = r#"<html><body>
<div id="intro"><p>intro</p></div><div id="body"><p>body</p></div>
<div id="outer"><div id="inner"></div></div>
</body></html>"#;
let html = strip(source, |document| {
let xpath = |e| Selector::xpath(e).unwrap();
Util::move_into(
document,
&xpath("//div[@id='intro']"),
&xpath("//div[@id='body']"),
)
.unwrap();
Util::move_into(
document,
&xpath("//div[@id='inner']"),
&xpath("//div[@id='outer']"),
)
.unwrap();
});
crate::test_util::assert_html_eq(
r#"<html><body>
<div id="intro"><p>intro</p><div id="body"><p>body</p></div></div>
<div id="outer"><div id="inner"></div></div>
</body></html>"#,
&html,
);
}
#[test]
fn dissolve_keeps_children_in_place() {
let source = r#"<html><body>
<p>a <a href="/x">link <b>text</b></a> b</p><div class="x"><p>c</p></div>
</body></html>"#;
let html = strip(source, |document| {
Util::dissolve(
document,
&Selector::xpath("//a | //div[@class='x']").unwrap(),
)
.unwrap();
});
crate::test_util::assert_html_eq(
"<html><body><p>a link <b>text</b> b</p><p>c</p></body></html>",
&html,
);
}
#[test]
fn find_page_url_from_element_attribute_and_text() {
let html = r#"<p><a class="next" href="/articles/1?page=2">Next page</a>
<a class="empty" href="">nothing</a>
<span class="url">https://example.org/print/1</span></p>"#;
let page_2 = Some("https://example.com/articles/1?page=2".to_string());
assert_eq!(page_url(html, "//a[@class='next']"), page_2);
assert_eq!(page_url(html, "//a[@class='next']/@href"), page_2);
assert_eq!(
page_url(html, "//span[@class='url']/text()").as_deref(),
Some("https://example.org/print/1")
);
assert_eq!(page_url(html, "//a[@class='empty']"), None);
assert_eq!(
page_url(html, "//a[@class='empty'] | //a[@class='next']"),
page_2
);
}
fn data_table_flags(source: &str) -> Vec<Option<String>> {
let document = Document::from(source);
Util::mark_data_tables(&document);
dom::select(&document, "table")
.iter()
.map(|table| dom::attr(table, crate::constants::DATA_TABLE_ATTR))
.collect()
}
#[test]
fn mark_data_tables_header_cells_win() {
let flags = data_table_flags(
"<table><tr><th>head</th></tr><tr><td><table><tr><td>x</td></tr></table></td></tr></table>",
);
assert_eq!(flags[0].as_deref(), Some("true"));
}
#[test]
fn mark_data_tables_small_layout_table() {
let flags = data_table_flags("<table><tr><td>a</td><td>b</td></tr></table>");
assert_eq!(flags, vec![Some("false".to_string())]);
}
fn replace_emojis(source: &str, expected: &str) {
let document = Document::from(source);
let p = dom::select(&document, "body > p")[0];
Util::replace_emoji_images(&document.root());
let result = p.html();
assert_html_eq(expected, &result);
}
#[test]
fn replace_emojis_1() {
replace_emojis(
"<p>Letβs see if I did a better job of it this time by telling him he was using Arch wrong. <img src=\"https://s0.wp.com/wp-content/mu-plugins/wpcom-smileys/twemoji/2/72x72/1f600.png\" alt=\"π\"/></p>",
"<p>Letβs see if I did a better job of it this time by telling him he was using Arch wrong. π</p>",
)
}
#[test]
fn replace_emojis_2() {
replace_emojis(
"<p><img src=\"https://abc.com/img.jpeg\"/><img src=\"https://s0.wp.com/wp-content/mu-plugins/wpcom-smileys/twemoji/2/72x72/1f600.png\" alt=\"π\"/> Abc</p>",
"<p><img src=\"https://abc.com/img.jpeg\"/>π Abc</p>",
)
}
fn replace_schema_org_objects(source: &str) -> String {
let document = Document::from(source);
Util::replace_schema_org_orbjects(&document.root());
document.body().unwrap().html().to_string()
}
#[test]
fn schema_org_objects_with_either_scheme() {
for scheme in ["http", "https"] {
let html = replace_schema_org_objects(&format!(
r#"<html><body>
<div itemtype="{scheme}://schema.org/ImageObject">
<meta itemprop="url" content="https://example.com/a.jpg">
</div>
<div itemtype=" {scheme}://schema.org/VideoObject ">
<meta itemprop="name" content="A video">
</div>
</body></html>"#
));
assert!(html.contains("<imageobject>"), "{scheme}: {html}");
assert!(html.contains("<videoobject>"), "{scheme}: {html}");
}
}
#[test]
fn schema_org_objects_keep_their_position() {
let html = replace_schema_org_objects(
r#"<html><body><div id="content">
<p>before image</p>
<a href="https://example.com/page"><div itemprop="image">
<meta itemprop="url" content="https://example.com/a.jpg">
</div></a>
<p>before video</p>
<div itemprop="video"><meta itemprop="name" content="A video"></div>
<p>after</p>
</div></body></html>"#,
);
let positions = [
"before image",
"<imageobject>",
"before video",
"<videoobject>",
"after",
]
.map(|needle| {
html.find(needle)
.unwrap_or_else(|| panic!("{needle}: {html}"))
});
assert!(positions.is_sorted(), "{html}");
}
}