article_scraper 3.0.0-alpha1

Scrap article contents from the web. Powered by fivefilters full text feed configurations & mozilla readability.
Documentation
//! Thin helpers on top of dom_query: tag-name checks, text content, tree walking and the one
//! cross-document operation the crate needs.
//!
//! dom_query's tree is an arena indexed by `NodeId`. Every `NodeRef`-level method that takes
//! another node (`append_child`, `insert_before`, `replace_with`, …) only receives that id, so
//! passing a node from a different document compiles and silently corrupts the target tree.
//! Everything in the pipeline works inside one document except [`move_into`], which copies
//! through `Selection::append_selection`. The helpers here debug-assert that both nodes share a
//! tree everywhere else.
//!
//! dom_query's iterators (`children_it`, `descendants_it`, `ancestors_it`) hold a borrow of the
//! tree's `RefCell` until they are dropped, so mutating the tree inside such a loop panics with
//! `BorrowMutError`. Collect the nodes first, as the helpers here do by returning `Vec`.

use dom_query::{Document, LocalName, NodeRef, Selection};

/// `true` if both nodes belong to the same tree.
pub fn same_tree(a: &NodeRef, b: &NodeRef) -> bool {
    std::ptr::eq(a.tree, b.tree)
}

/// The local tag name of an element in uppercase, or an empty string for every other node.
///
/// Allocates; to compare tag names use [`tag_name_is`] or [`tag_name_in`] instead.
pub fn tag_name(node: &NodeRef) -> String {
    node.qual_name_ref()
        .map(|name| name.local.as_ref().to_ascii_uppercase())
        .unwrap_or_default()
}

/// The local tag name of an element as html5ever stores it (lowercase for HTML elements),
/// `None` for every other node. Cheaper than [`tag_name`] when one walk checks several names.
pub fn local_name(node: &NodeRef) -> Option<LocalName> {
    node.qual_name_ref().map(|name| name.local.clone())
}

/// Compares the local tag name of an element ASCII-case-insensitively. `false` for non-elements.
///
/// html5ever lowercases HTML names but keeps SVG names such as `foreignObject`, so the
/// comparison is case-insensitive either way.
pub fn tag_name_is(node: &NodeRef, name: &str) -> bool {
    node.qual_name_ref()
        .is_some_and(|qual_name| qual_name.local.as_ref().eq_ignore_ascii_case(name))
}

/// `true` if the local tag name of an element is one of `names`, compared as in
/// [`tag_name_is`]. `false` for non-elements.
pub fn tag_name_in(node: &NodeRef, names: &[&str]) -> bool {
    node.qual_name_ref().is_some_and(|qual_name| {
        let local = qual_name.local.as_ref();
        names.iter().any(|name| local.eq_ignore_ascii_case(name))
    })
}

/// The text content of the node and its descendants (the text itself for text nodes).
pub fn text(node: &NodeRef) -> String {
    node.text().to_string()
}

/// The value of an attribute, `None` for missing attributes and non-elements.
pub fn attr(node: &NodeRef, name: &str) -> Option<String> {
    node.attr(name).map(|value| value.to_string())
}

/// All attributes as `(name, value)` pairs, in document order.
pub fn attrs(node: &NodeRef) -> Vec<(String, String)> {
    node.attrs()
        .into_iter()
        .map(|attr| (attr.name.local.to_string(), attr.value.to_string()))
        .collect()
}

/// `true` for a text node consisting only of whitespace.
pub fn is_whitespace_text(node: &NodeRef) -> bool {
    node.is_text() && node.text().trim().is_empty()
}

/// The nodes matching a CSS selector, in document order.
///
/// # Panics
///
/// Panics on an invalid selector. Only use it with selectors written in the code.
pub fn select<'a>(document: &'a Document, css: &str) -> Vec<NodeRef<'a>> {
    document.select(css).nodes().to_vec()
}

/// The descendant elements (not `node` itself) with the given tag name, in document order.
/// `"*"` matches every element.
pub fn elements_by_tag_name<'a>(node: &NodeRef<'a>, tag: &str) -> Vec<NodeRef<'a>> {
    let all_tags = tag == "*";
    node.descendants_it()
        .filter(|descendant| descendant.is_element() && (all_tags || tag_name_is(descendant, tag)))
        .collect()
}

/// The first descendant element with the given tag name, in document order.
pub fn first_element_by_tag_name<'a>(node: &NodeRef<'a>, tag: &str) -> Option<NodeRef<'a>> {
    node.descendants_it()
        .find(|descendant| tag_name_is(descendant, tag))
}

/// `true` if an element with one of the tag names is a descendant of `node`.
pub fn has_any_descendant_tag(node: &NodeRef, tag_names: &[&str]) -> bool {
    node.descendants_it()
        .any(|descendant| tag_name_in(&descendant, tag_names))
}

/// `true` if a descendant of `node` is a text node with non-whitespace content or an element with
/// one of the tag names. One walk that stops at the first hit, without building the text.
pub fn has_text_or_descendant_tag(node: &NodeRef, tag_names: &[&str]) -> bool {
    node.descendants_it().any(|descendant| {
        tag_name_in(&descendant, tag_names)
            || (descendant.is_text() && !descendant.text().trim().is_empty())
    })
}

/// `true` if one of the nodes in `nodes` is an ancestor of `node`.
pub fn has_ancestor_in(node: &NodeRef, nodes: &[NodeRef]) -> bool {
    node.ancestors_it(None)
        .any(|ancestor| nodes.iter().any(|candidate| candidate.id == ancestor.id))
}

/// The next node in document order, staying inside the subtree of `root`: the first child,
/// otherwise the next sibling of the node or of the nearest ancestor below `root`. `None` once
/// the subtree is exhausted. With `ignore_self_and_kids` the node's children are skipped.
pub fn next_node_within<'a>(
    node: &NodeRef<'a>,
    root: &NodeRef<'a>,
    ignore_self_and_kids: bool,
) -> Option<NodeRef<'a>> {
    if !ignore_self_and_kids && let Some(first_child) = node.first_child() {
        return Some(first_child);
    }

    let mut node = *node;
    while node.id != root.id {
        if let Some(next_sibling) = node.next_sibling() {
            return Some(next_sibling);
        }
        node = node.parent()?;
    }

    None
}

/// Removes `node` from its parent and returns the node that follows it inside `root`.
pub fn remove_and_next_within<'a>(node: &NodeRef<'a>, root: &NodeRef<'a>) -> Option<NodeRef<'a>> {
    let next_node = next_node_within(node, root, true);
    node.remove_from_parent();
    next_node
}

/// Appends `child` to `parent`. Both must belong to the same document.
pub fn append_child(parent: &NodeRef, child: &NodeRef) {
    debug_assert!(same_tree(parent, child), "append_child across documents");
    parent.append_child(child);
}

/// Replaces `old` with `new` in the tree. Both must belong to the same document.
pub fn replace_with(old: &NodeRef, new: &NodeRef) {
    debug_assert!(same_tree(old, new), "replace_with across documents");
    old.replace_with(new);
}

/// Moves `nodes` to the end of `target`, which may be in another document.
///
/// The nodes are detached from their document first and then deep-copied into the target's
/// tree, so a node together with one of its descendants ends up as siblings, not copied twice.
/// Nodes in the same document are moved as well (copied and the originals left detached).
pub fn move_into(target: &NodeRef, nodes: Vec<NodeRef>) {
    if nodes.is_empty() {
        return;
    }
    Selection::from(*target).append_selection(&Selection::from(nodes));
}

/// A new document holding only an empty `<article>` element, see [`article_root`].
pub fn new_article_document() -> Document {
    Document::from("<html><head></head><body><article></article></body></html>")
}

/// The `<article>` element of a document created with [`new_article_document`].
pub fn article_root(document: &Document) -> Option<NodeRef<'_>> {
    document
        .body()
        .and_then(|body| body.first_element_child())
        .filter(|node| tag_name_is(node, "article"))
}

#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn tag_names_are_case_insensitive() {
        let document = Document::from("<div><svg><foreignObject></foreignObject></svg>text</div>");
        let div = select(&document, "div")[0];
        assert_eq!(tag_name(&div), "DIV");
        assert!(tag_name_is(&div, "div") && tag_name_is(&div, "DIV"));
        assert!(tag_name_in(&div, &["p", "div"]) && !tag_name_in(&div, &["p"]));
        let foreign = first_element_by_tag_name(&div, "foreignobject").unwrap();
        assert_eq!(tag_name(&foreign), "FOREIGNOBJECT");
        let text_node = div.last_child().unwrap();
        assert!(text_node.is_text());
        assert_eq!(tag_name(&text_node), "");
        assert!(!tag_name_is(&text_node, ""));
        assert!(!tag_name_in(&text_node, &[""]));
    }

    #[test]
    fn walks_the_subtree_only() {
        let document = Document::from("<div id=a><p>1<b>2</b></p><p>3</p></div><div id=b>4</div>");
        let root = select(&document, "#a")[0];
        let mut visited = Vec::new();
        let mut node = Some(root);
        while let Some(current) = node {
            visited.push(if current.is_text() {
                text(&current)
            } else {
                tag_name(&current)
            });
            node = next_node_within(&current, &root, false);
        }
        assert_eq!(visited, ["DIV", "P", "1", "B", "2", "P", "3"]);

        // Removing the first <p> continues with its sibling, not with its children.
        let first_p = select(&document, "p")[0];
        let next = remove_and_next_within(&first_p, &root).unwrap();
        assert_eq!(text(&next), "3");
        assert_eq!(select(&document, "p").len(), 1);
    }

    #[test]
    fn moves_nodes_across_documents() {
        let source = Document::from("<table><tr><td>cell</td></tr></table><p>rest</p>");
        let target = new_article_document();
        let article = article_root(&target).unwrap();

        // A lone <tr> only survives as a node copy; fragment parsing would drop it.
        let tr = select(&source, "tr");
        move_into(&article, tr);

        assert_eq!(
            article.html().as_ref(),
            "<article><tr><td>cell</td></tr></article>"
        );
        assert_eq!(select(&source, "tr").len(), 0);
        assert!(select(&source, "p").len() == 1);
    }

    #[test]
    fn nested_nodes_are_moved_once() {
        let source = Document::from("<div class=x><p>outer</p><div class=x>inner</div></div>");
        let target = new_article_document();
        let article = article_root(&target).unwrap();

        let divs = select(&source, ".x");
        assert!(has_ancestor_in(&divs[1], &divs));
        move_into(&article, divs);

        let html = article.html();
        assert_eq!(html.matches("inner").count(), 1, "{html}");
        // The nested match became a sibling of its former container.
        assert_eq!(select(&target, "article > div").len(), 2, "{html}");
    }
}