article_scraper 3.0.0-alpha1

Scrap article contents from the web. Powered by fivefilters full text feed configurations & mozilla readability.
Documentation
use super::config::ConfigEntry;
use crate::{article::Article, constants, util::Util};
use chrono::{DateTime, NaiveDate, NaiveDateTime, Utc};
use dom_query::Document;
use std::str::FromStr;

/// Date-time formats with an offset that [`parse_date`] accepts besides RFC 3339 and RFC 2822.
/// `%z` also takes `+0100`, which RFC 3339 doesn't allow.
const DATE_TIME_FORMATS: [&str; 2] = ["%Y-%m-%dT%H:%M:%S%.f%z", "%Y-%m-%d %H:%M:%S%.f%z"];

/// Date-time formats without an offset that [`parse_date`] accepts, read as UTC.
const NAIVE_DATE_TIME_FORMATS: [&str; 4] = [
    "%Y-%m-%d %H:%M:%S%.f",
    "%Y-%m-%dT%H:%M:%S%.f",
    "%Y-%m-%d %H:%M",
    "%Y-%m-%dT%H:%M",
];

/// A date read from the page. `naive` if the text had no offset, so `date` is its local time
/// read as UTC and may be off by the site's offset.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(crate) struct ParsedDate {
    pub date: DateTime<Utc>,
    pub naive: bool,
}

/// What dom_smoothie finds on a page without extracting the content: JSON-LD first, then
/// `<meta>` tags, then (for the title) the cleaned-up `<title>` or `<h1>`.
#[derive(Debug, Default)]
pub struct PageMetadata {
    pub title: Option<String>,
    pub author: Option<String>,
    pub date: Option<ParsedDate>,
    pub image: Option<String>,
}

/// Reads [`PageMetadata`] from `document`, without JSON-LD for `skip_json_ld`. It has to run
/// before `prep_content`, which removes the `<script>` elements holding the JSON-LD.
///
/// dom_smoothie takes the document by value; it is handed back unchanged.
pub fn page_metadata(document: Document, skip_json_ld: bool) -> (Document, PageMetadata) {
    // Without a URL `with_document` can't fail (it only rejects relative URLs). The URL is only
    // used for the favicon, which isn't needed.
    let readability = dom_smoothie::Readability::with_document(document, None, None)
        .expect("dom_smoothie::Readability::with_document without a URL");
    let json_ld = if skip_json_ld {
        None
    } else {
        readability.parse_json_ld()
    };
    let metadata = readability.get_article_metadata(json_ld);

    let non_empty = |value: Option<String>| {
        value
            .map(|value| value.trim().to_string())
            .filter(|value| !value.is_empty())
    };
    let date = non_empty(metadata.published_time).and_then(|date_string| {
        let date = parse_date(&date_string);
        if date.is_none() {
            tracing::warn!(date_string, "Parsing date failed");
        }
        date
    });
    let page_metadata = PageMetadata {
        title: non_empty(Some(metadata.title)),
        author: non_empty(metadata.byline),
        date,
        image: non_empty(metadata.image),
    };
    tracing::debug!(?page_metadata, "dom_smoothie metadata");

    (readability.doc, page_metadata)
}

/// Fills the article's empty `title`, `author` and `date`: from the site config's rules, then
/// the global config's, then `page`.
pub fn extract(
    document: &Document,
    config: Option<&ConfigEntry>,
    global_config: Option<&ConfigEntry>,
    page: &PageMetadata,
    article: &mut Article,
) {
    if article.title.is_none() {
        // The rule values are text content, which html5ever has already decoded, so they
        // must not be decoded again (`&amp;amp;` in the source is the text `&amp;`).
        article.title = extract_title(document, config, global_config)
            .map(|title| {
                // clean titles that contain separators
                if constants::TITLE_SEPARATOR.is_match(&title) {
                    let new_title = constants::TITLE_CUT_END.replace(&title, "$1");
                    let word_count = constants::WORD_COUNT.split(&title).count();
                    if word_count < 3 {
                        constants::TITLE_CUT_FRONT
                            .replace(&title, "$1")
                            .trim()
                            .to_string()
                    } else {
                        new_title.trim().to_string()
                    }
                } else {
                    title
                }
            })
            .or_else(|| page.title.clone());
    }

    if article.author.is_none() {
        article.author =
            extract_author(document, config, global_config).or_else(|| page.author.clone());
    }

    if article.date.is_none() {
        // A rule's value without an offset loses to page metadata that has one.
        article.date = match (extract_date(document, config, global_config), page.date) {
            (Some(rule), Some(page)) if rule.naive && !page.naive => Some(page.date),
            (rule, page) => rule.or(page).map(|parsed| parsed.date),
        };
    }
}

fn extract_title(
    document: &Document,
    config: Option<&ConfigEntry>,
    global_config: Option<&ConfigEntry>,
) -> Option<String> {
    // check site specific config
    if let Some(config) = config {
        for selector in &config.title {
            if let Ok(title) = Util::extract_title(document, selector) {
                tracing::debug!(title, "Article title (site specific config)");
                return Some(title);
            }
        }
    }

    // check global config
    if let Some(global_config) = global_config {
        for selector in &global_config.title {
            if let Ok(title) = Util::extract_title(document, selector) {
                tracing::debug!(title, "Article title (global config)");
                return Some(title);
            }
        }
    }

    None
}

fn extract_author(
    document: &Document,
    config: Option<&ConfigEntry>,
    global_config: Option<&ConfigEntry>,
) -> Option<String> {
    // check site specific config
    if let Some(config) = config {
        for selector in &config.author {
            if let Ok(author) = Util::extract_value(document, selector) {
                tracing::debug!(author, "Site config");
                return Some(author);
            }
        }
    }

    // check global config
    if let Some(global_config) = global_config {
        for selector in &global_config.author {
            if let Ok(author) = Util::extract_value(document, selector) {
                tracing::debug!(author, "global config");
                return Some(author);
            }
        }
    }

    None
}

fn extract_date(
    document: &Document,
    config: Option<&ConfigEntry>,
    global_config: Option<&ConfigEntry>,
) -> Option<ParsedDate> {
    // check site specific config
    if let Some(config) = config {
        for selector in &config.date {
            if let Ok(date_string) = Util::extract_value(document, selector) {
                tracing::debug!(date_string, "site config");
                if let Some(date) = parse_date(&date_string) {
                    return Some(date);
                } else {
                    tracing::warn!(date_string, "Parsing date failed",);
                }
            }
        }
    }

    // check global config
    if let Some(global_config) = global_config {
        for selector in &global_config.date {
            if let Ok(date_string) = Util::extract_value(document, selector) {
                tracing::debug!(date_string, "global config");
                if let Some(date) = parse_date(&date_string) {
                    return Some(date);
                } else {
                    tracing::warn!(date_string, "Parsing date failed",);
                }
            }
        }
    }

    None
}

/// Parses RFC 3339 (`2023-08-03T10:00:00+02:00`), RFC 2822 (`Thu, 03 Aug 2023 10:00:00 +0200`),
/// [`DATE_TIME_FORMATS`], and without an offset [`NAIVE_DATE_TIME_FORMATS`] and `%Y-%m-%d`.
pub(crate) fn parse_date(date_string: &str) -> Option<ParsedDate> {
    let date_string = date_string.trim();
    let with_offset = |date: DateTime<Utc>| ParsedDate { date, naive: false };
    let naive = |date: NaiveDateTime| ParsedDate {
        date: date.and_utc(),
        naive: true,
    };

    if let Ok(date) = DateTime::from_str(date_string) {
        return Some(with_offset(date));
    }
    if let Ok(date) = DateTime::parse_from_rfc2822(date_string) {
        return Some(with_offset(date.to_utc()));
    }
    for format in DATE_TIME_FORMATS {
        if let Ok(date) = DateTime::parse_from_str(date_string, format) {
            return Some(with_offset(date.to_utc()));
        }
    }
    for format in NAIVE_DATE_TIME_FORMATS {
        if let Ok(date) = NaiveDateTime::parse_from_str(date_string, format) {
            return Some(naive(date));
        }
    }
    NaiveDate::parse_from_str(date_string, "%Y-%m-%d")
        .ok()
        .and_then(|date| date.and_hms_opt(0, 0, 0))
        .map(naive)
}