article_scraper 3.0.0-alpha1

Scrap article contents from the web. Powered by fivefilters full text feed configurations & mozilla readability.
Documentation
use std::sync::LazyLock;

use regex::{Regex, RegexBuilder};

#[cfg(feature = "image-downloader")]
pub const UNKNOWN_CONTENT_SIZE_LIMIT: usize = 5 * 1024 * 1024;
/// Largest image that is downloaded; bigger ones are skipped.
#[cfg(feature = "image-downloader")]
pub const MAX_IMAGE_SIZE: usize = 32 * 1024 * 1024;
/// Most images `download_images_from_string` downloads at the same time.
#[cfg(feature = "image-downloader")]
pub const MAX_PARALLEL_IMAGE_DOWNLOADS: usize = 8;
/// JPEG quality of images `download_images_from_string` scales down.
#[cfg(feature = "image-downloader")]
pub const RESIZED_JPEG_QUALITY: u8 = 85;
/// Largest HTML page that is downloaded.
pub const MAX_HTML_SIZE: usize = 16 * 1024 * 1024;
/// Most pages (the first included) that are followed through `next_page_link`.
pub const MAX_PAGES: usize = 10;
pub const MAX_REDIRECTS: u32 = 12;
pub const DEFAULT_CHAR_THRESHOLD: usize = 500;
pub static IS_IMAGE: LazyLock<Regex> = LazyLock::new(|| {
    RegexBuilder::new(r#"\.(jpg|jpeg|png|webp)"#)
        .case_insensitive(true)
        .build()
        .expect("IS_IMAGE regex")
});
pub static COPY_TO_SRCSET: LazyLock<Regex> = LazyLock::new(|| {
    RegexBuilder::new(r#"\.(jpg|jpeg|png|webp)\s+\d"#)
        .case_insensitive(true)
        .build()
        .expect("COPY_TO_SRC regex")
});
pub static COPY_TO_SRC: LazyLock<Regex> = LazyLock::new(|| {
    RegexBuilder::new(r#"^\s*\S+\.(jpg|jpeg|png|webp)\S*\s*$"#)
        .case_insensitive(true)
        .build()
        .expect("COPY_TO_SRC regex")
});
pub static CHARSET: LazyLock<Regex> =
    LazyLock::new(|| regex::Regex::new(r#"charset=([^"']+)"#).expect("CHARSET regex"));
pub static HTML_CHARSET: LazyLock<Regex> =
    LazyLock::new(|| Regex::new(r#"<meta.*?charset="*(.*?)""#).expect("HTML_CHARSET regex"));
pub static NORMALIZE: LazyLock<Regex> =
    LazyLock::new(|| Regex::new(r#"\s{2,}"#).expect("NORMALIZE regex"));
pub static TOKENIZE: LazyLock<Regex> =
    LazyLock::new(|| Regex::new(r#"\W+"#).expect("TOKENIZE regex"));
pub static HAS_CONTENT: LazyLock<Regex> =
    LazyLock::new(|| Regex::new(r#"\S"#).expect("HAS_CONTENT regex"));
pub static HASH_URL: LazyLock<Regex> =
    LazyLock::new(|| Regex::new(r#"^#.+"#).expect("HASH_URL regex"));
pub static POSITIVE: LazyLock<Regex> =
    LazyLock::new(|| {
        RegexBuilder::new(
        r#"article|body|content|entry|hentry|h-entry|main|page|pagination|post|text|blog|story"#,
    ).case_insensitive(true).build()
    .expect("POSITIVE regex")
    });
pub static NEGATIVE: LazyLock<Regex> = LazyLock::new(|| {
    RegexBuilder::new(r#"-ad-|hidden|^hid$| hid$| hid |^hid |banner|combx|comment|com-|contact|foot|footer|footnote|gdpr|masthead|media|meta|outbrain|promo|related|scroll|share|shoutbox|sidebar|skyscraper|sponsor|shopping|tags|tool|widget"#).case_insensitive(true).build().expect("NEGATIVE regex")
});
pub static SHARE_ELEMENTS: LazyLock<Regex> = LazyLock::new(|| {
    RegexBuilder::new(r#"(\b|_)(share|sharedaddy)(\b|_)"#)
        .case_insensitive(true)
        .build()
        .expect("SHARE_ELEMENTS regex")
});
pub static SRC_SET_URL: LazyLock<Regex> = LazyLock::new(|| {
    Regex::new(r#"(\S+)(\s+[\d.]+[xw])?(\s*(?:,|$))"#).expect("SRC_SET_URL regex")
});
pub static TITLE_SEPARATOR: LazyLock<Regex> =
    LazyLock::new(|| Regex::new(r#" [-|—\\/>»] "#).expect("TITLE_SEPARATOR regex"));
pub static TITLE_CUT_END: LazyLock<Regex> = LazyLock::new(|| {
    RegexBuilder::new(r#"(.*)[-|—\\/>»] .*"#)
        .case_insensitive(true)
        .build()
        .expect("TITLE_CUT_END regex")
});
pub static WORD_COUNT: LazyLock<Regex> =
    LazyLock::new(|| Regex::new(r#"\s+"#).expect("WORD_COUNT regex"));
pub static TITLE_CUT_FRONT: LazyLock<Regex> = LazyLock::new(|| {
    RegexBuilder::new(r#"[^-|\\/>»]*[-|\\/>»](.*)"#)
        .case_insensitive(true)
        .build()
        .expect("TITLE_CUT_FRONT regex")
});
pub static VIDEOS: LazyLock<Regex> = LazyLock::new(|| {
    RegexBuilder::new(r#"(www\.)?((dailymotion|youtube|youtube-nocookie|player\.vimeo|v\.qq)\.com|(archive|upload\.wikimedia)\.org|player\.twitch\.tv)"#).case_insensitive(true).build().expect("VIDEOS regex")
});
pub static BASE64_DATA_URL: LazyLock<Regex> = LazyLock::new(|| {
    RegexBuilder::new(r#"^data:\s*([^\s;,]+)\s*;\s*base64\s*,"#)
        .case_insensitive(true)
        .build()
        .expect("BASE64_DATA_URL regex")
});
pub const SCORE_ATTR: &str = "content_score";
pub const DATA_TABLE_ATTR: &str = "is_data_table";

pub const PRESENTATIONAL_ATTRIBUTES: &[&str] = &[
    "align",
    "background",
    "bgcolor",
    "border",
    "cellpadding",
    "cellspacing",
    "frame",
    "hspace",
    "rules",
    "style",
    "valign",
    "vspace",
];
pub const DEPRECATED_SIZE_ATTRIBUTE_ELEMS: &[&str] = &["table", "th", "td", "hr", "pre"];

pub const VALID_EMPTY_TAGS: &[&str] = &[
    "area", "base", "br", "col", "embed", "hr", "img", "link", "meta", "source", "track", "iframe",
    "th", "td", "tr", "audio", "video", "wbr",
];

/// Formatting elements (HTML, "list of active formatting elements"): the parser re-creates
/// them around content that follows an implicitly closed element.
pub const FORMATTING_TAGS: &[&str] = &[
    "a", "b", "big", "code", "em", "font", "i", "nobr", "s", "small", "strike", "strong", "tt", "u",
];

/// Start tags that close an open `<p>` (HTML, "in body" insertion mode).
pub const P_CLOSING_TAGS: &[&str] = &[
    "address",
    "article",
    "aside",
    "blockquote",
    "details",
    "dialog",
    "div",
    "dl",
    "fieldset",
    "figcaption",
    "figure",
    "footer",
    "form",
    "h1",
    "h2",
    "h3",
    "h4",
    "h5",
    "h6",
    "header",
    "hgroup",
    "hr",
    "main",
    "menu",
    "nav",
    "ol",
    "p",
    "pre",
    "section",
    "table",
    "ul",
];

pub const EMBED_TAG_NAMES: &[&str] = &["object", "embed", "iframe"];

pub const PHRASING_ELEMS: &[&str] = &[
    // "canvas", "iframe", "svg", "video",
    "abbr", "audio", "b", "bdo", "br", "button", "cite", "code", "data", "datalist", "dfn", "em",
    "embed", "i", "img", "input", "kbd", "label", "mark", "math", "meter", "noscript", "object",
    "output", "progress", "q", "ruby", "samp", "script", "select", "small", "span", "strong",
    "sub", "sup", "textarea", "time", "var", "wbr",
];

pub const POSITIVE_LEAD_IMAGE_URL_HINTS: &[&str] =
    &["upload", "wp-content", "large", "photo", "wp-image"];

pub static POSITIVE_LEAD_IMAGE_URL_HINTS_REGEX: LazyLock<Regex> = LazyLock::new(|| {
    RegexBuilder::new(&POSITIVE_LEAD_IMAGE_URL_HINTS.join("|"))
        .case_insensitive(true)
        .build()
        .expect("POSITIVE_LEAD_IMAGE_URL_HINTS regex")
});

pub const NEGATIVE_LEAD_IMAGE_URL_HINTS: &[&str] = &[
    "spacer",
    "sprite",
    "blank",
    "throbber",
    "gradient",
    "tile",
    "bg",
    "background",
    "icon",
    "social",
    "header",
    "hdr",
    "advert",
    "spinner",
    "loader",
    "loading",
    "default",
    "rating",
    "share",
    "facebook",
    "twitter",
    "theme",
    "promo",
    "ads",
    "wp-includes",
];

pub static NEGATIVE_LEAD_IMAGE_URL_HINTS_REGEX: LazyLock<Regex> = LazyLock::new(|| {
    RegexBuilder::new(&NEGATIVE_LEAD_IMAGE_URL_HINTS.join("|"))
        .case_insensitive(true)
        .build()
        .expect("NEGATIVE_LEAD_IMAGE_URL_HINTS regex")
});

pub const PHOTO_HINTS: &[&str] = &["figure", "photo", "image", "caption"];
pub static PHOTO_HINTS_REGEX: LazyLock<Regex> = LazyLock::new(|| {
    RegexBuilder::new(&PHOTO_HINTS.join("|"))
        .case_insensitive(true)
        .build()
        .expect("PHOTO_HINTS_REGEX regex")
});

pub static GIF_REGEX: LazyLock<Regex> = LazyLock::new(|| {
    RegexBuilder::new(r#"\.gif(\?.*)?$"#)
        .case_insensitive(true)
        .build()
        .expect("GIF_REGEX")
});
pub static JPG_REGEX: LazyLock<Regex> = LazyLock::new(|| {
    RegexBuilder::new(r#"\.jpe?g(\?.*)?$"#)
        .case_insensitive(true)
        .build()
        .expect("JPG_REGEX")
});