article_scraper 3.0.0-alpha1

Scrap article contents from the web. Powered by fivefilters full text feed configurations & mozilla readability.
Documentation
//! Selectors from site configs, compiled when the config is loaded.

use std::fmt;

use dom_query::{Document, NodeRef};
use dom_query_xpath::{Value, XNode, XPath};

/// A selector from a site config (`title:`, `body:`, `strip:`, …).
///
/// Only XPath exists so far. CSS-based configs (ftr-site-config#1979) will add a variant; the
/// extraction code only ever sees `Selector`.
#[derive(Clone, Debug, PartialEq)]
#[non_exhaustive]
pub enum Selector {
    XPath(XPath),
}

#[derive(Debug, thiserror::Error)]
pub enum SelectorError {
    #[error("invalid XPath: {0}")]
    XPath(#[from] dom_query_xpath::ParseError),
    #[error("evaluating the selector failed: {0}")]
    Eval(#[from] dom_query_xpath::EvalError),
}

impl Selector {
    pub fn xpath(expression: &str) -> Result<Self, SelectorError> {
        Ok(Self::XPath(XPath::parse(expression)?))
    }

    /// The selector as written in the config.
    pub fn as_str(&self) -> &str {
        match self {
            Self::XPath(xpath) => xpath.as_str(),
        }
    }

    /// Evaluates the selector against a whole document.
    pub fn evaluate<'a>(&self, document: &'a Document) -> Result<Value<'a>, SelectorError> {
        self.evaluate_at(&document.root())
    }

    /// Evaluates the selector with `context` as the context node.
    pub fn evaluate_at<'a>(&self, context: &NodeRef<'a>) -> Result<Value<'a>, SelectorError> {
        match self {
            Self::XPath(xpath) => Ok(xpath.evaluate(context)?),
        }
    }

    /// The nodes the selector matches in a document, in document order.
    ///
    /// An XPath expression that returns a string, number or boolean instead of a node-set
    /// yields no nodes. Where a string makes sense (`title`, `author`, `date`, page links) the
    /// callers use [`Selector::evaluate`] instead.
    pub fn select<'a>(&self, document: &'a Document) -> Result<Vec<XNode<'a>>, SelectorError> {
        match self.evaluate(document)? {
            Value::NodeSet(nodes) => Ok(nodes),
            other => {
                tracing::debug!(selector = %self, result = other.type_name(), "Selector does not return nodes");
                Ok(Vec::new())
            }
        }
    }
}

impl fmt::Display for Selector {
    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
        f.write_str(self.as_str())
    }
}

/// The selector language of a config repository.
///
/// `ftr-site-config` writes every selector as XPath 1.0. A CSS-based repository
/// (ftr-site-config#1979) only needs its own implementation, which can also decide per value
/// whether it is CSS or an XPath fallback.
pub trait SelectorSyntax: Sync {
    fn parse(&self, value: &str) -> Result<Selector, SelectorError>;
}

/// `ftr-site-config`: every value is XPath 1.0.
pub struct XPathSyntax;

impl SelectorSyntax for XPathSyntax {
    fn parse(&self, value: &str) -> Result<Selector, SelectorError> {
        Selector::xpath(value)
    }
}

#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn selector_is_send_sync() {
        fn assert_send_sync<T: Send + Sync + Clone + fmt::Debug>() {}
        assert_send_sync::<Selector>();
        // for the CSS variant
        assert_send_sync::<dom_query::Matcher>();
    }

    #[test]
    fn invalid_xpath_is_rejected() {
        assert!(XPathSyntax.parse("//div[").is_err());
        assert!(XPathSyntax.parse("contains(@class='x')").is_err());
        assert_eq!(
            XPathSyntax.parse("//a/@href").unwrap().as_str(),
            "//a/@href"
        );
    }

    #[test]
    fn select_returns_nodes_in_document_order() {
        let document = Document::from(r#"<p id="1">a</p><p id="2">b</p><a href="/x">l</a>"#);
        let selector = Selector::xpath("//a | //p").unwrap();
        let nodes = selector.select(&document).unwrap();
        let texts = nodes.iter().map(XNode::string_value).collect::<Vec<_>>();
        assert_eq!(texts, ["a", "b", "l"]);

        let href = Selector::xpath("//a/@href").unwrap();
        let nodes = href.select(&document).unwrap();
        assert!(nodes[0].is_attribute());
        assert_eq!(nodes[0].string_value(), "/x");
    }

    #[test]
    fn string_results_yield_no_nodes() {
        let document = Document::from("<title>a | b</title>");
        let selector = Selector::xpath("substring-before(//title, ' | ')").unwrap();
        assert!(selector.select(&document).unwrap().is_empty());
        assert_eq!(
            selector.evaluate(&document).unwrap(),
            Value::String("a".into())
        );
    }
}