article_scraper 3.0.0-alpha1

Scrap article contents from the web. Powered by fivefilters full text feed configurations & mozilla readability.
Documentation
//! Compare HTML documents by structure instead of by exact serialization.
//!
//! Both sides are parsed with dom_query (html5ever) and turned into a canonical text form:
//! - attributes are sorted by name
//! - runs of whitespace in text are collapsed to a single space and trimmed,
//!   text nodes that are only whitespace are dropped (except inside `<pre>` & co.)
//! - the doctype is ignored
//!
//! This keeps the tests stable across serializers that write the same tree differently.

use dom_query::{Document, NodeData, NodeRef};
use std::fmt::Write;

const WHITESPACE_PRESERVING: &[&str] = &["pre", "textarea", "listing", "plaintext"];

const CONTEXT_LINES: usize = 5;

/// Set `UPDATE_FIXTURES=1` to overwrite `expected.html` files with the current output instead of
/// comparing against them. Combine it with a test filter to update only the fixtures that were
/// accepted in the review report (`cargo test -- --ignored fixture_report`).
pub fn update_fixtures() -> bool {
    std::env::var_os("UPDATE_FIXTURES").is_some_and(|value| value == "1")
}

/// Compares `actual` with the fixture at `expected_path`, or overwrites the fixture when
/// [`update_fixtures`] is set.
#[track_caller]
pub fn check_fixture(expected_path: &str, actual: &str) {
    if update_fixtures() {
        std::fs::write(expected_path, actual)
            .unwrap_or_else(|error| panic!("Failed to write {expected_path}: {error}"));
        return;
    }

    let expected = std::fs::read_to_string(expected_path)
        .unwrap_or_else(|error| panic!("Failed to read {expected_path}: {error}"));
    assert_html_eq(&expected, actual);
}

/// Like [`check_fixture`], for plain text compared exactly.
#[track_caller]
pub fn check_text_fixture(expected_path: &str, actual: &str) {
    if update_fixtures() {
        std::fs::write(expected_path, actual)
            .unwrap_or_else(|error| panic!("Failed to write {expected_path}: {error}"));
        return;
    }

    let expected = std::fs::read_to_string(expected_path)
        .unwrap_or_else(|error| panic!("Failed to read {expected_path}: {error}"));
    assert_eq!(expected, actual, "{expected_path} differs");
}

#[track_caller]
pub fn assert_html_eq(expected: &str, actual: &str) {
    let expected_canonical = canonicalize(expected);
    let actual_canonical = canonicalize(actual);

    if expected_canonical == actual_canonical {
        return;
    }

    let expected_lines = expected_canonical.lines().collect::<Vec<_>>();
    let actual_lines = actual_canonical.lines().collect::<Vec<_>>();
    let first_diff = expected_lines
        .iter()
        .zip(actual_lines.iter())
        .position(|(e, a)| e != a)
        .unwrap_or(expected_lines.len().min(actual_lines.len()));

    let excerpt = |lines: &[&str]| {
        let start = first_diff.saturating_sub(CONTEXT_LINES);
        let end = (first_diff + CONTEXT_LINES + 1).min(lines.len());
        lines[start..end]
            .iter()
            .enumerate()
            .map(|(i, line)| {
                let marker = if start + i == first_diff { ">" } else { " " };
                format!("{marker}{:>5} {line}\n", start + i + 1)
            })
            .collect::<String>()
    };

    panic!(
        "HTML documents differ (canonical form, first difference at line {}):\n\nexpected:\n{}\nactual:\n{}",
        first_diff + 1,
        excerpt(&expected_lines),
        excerpt(&actual_lines),
    );
}

pub fn canonicalize(html: &str) -> String {
    let document = Document::from(html);
    let mut out = String::new();
    write_node(&document.root(), 0, false, &mut out);
    out
}

fn write_node(node: &NodeRef, depth: usize, preserve_whitespace: bool, out: &mut String) {
    let indent = "  ".repeat(depth);
    let Some(data) = node.query(|node| node.data.clone()) else {
        return;
    };

    match data {
        NodeData::Document | NodeData::Fragment => {
            write_children(node, depth, preserve_whitespace, out)
        }
        NodeData::Doctype { .. } | NodeData::ProcessingInstruction { .. } => {}
        NodeData::Text { contents } => {
            let text = if preserve_whitespace {
                contents.to_string()
            } else {
                normalize_whitespace(&contents)
            };
            if !text.is_empty() {
                _ = writeln!(out, "{indent}{text:?}");
            }
        }
        NodeData::Comment { contents } => {
            _ = writeln!(
                out,
                "{indent}<!-- {:?} -->",
                normalize_whitespace(&contents)
            );
        }
        NodeData::Element(element) => {
            let mut attrs = element
                .attrs
                .iter()
                .map(|attr| {
                    let name = match &attr.name.prefix {
                        Some(prefix) => format!("{prefix}:{}", attr.name.local),
                        None => attr.name.local.to_string(),
                    };
                    (name, attr.value.to_string())
                })
                .collect::<Vec<_>>();
            attrs.sort();

            _ = write!(out, "{indent}<{}", element.name.local);
            for (name, value) in attrs {
                _ = write!(out, " {name}={value:?}");
            }
            _ = writeln!(out, ">");

            let preserve_whitespace =
                preserve_whitespace || WHITESPACE_PRESERVING.contains(&element.name.local.as_ref());
            write_children(node, depth + 1, preserve_whitespace, out);
            if let Some(template_contents) = element.template_contents {
                let template_contents = NodeRef::new(template_contents, node.tree);
                write_children(&template_contents, depth + 1, preserve_whitespace, out);
            }
        }
    }
}

fn write_children(node: &NodeRef, depth: usize, preserve_whitespace: bool, out: &mut String) {
    for child in node.children_it(false) {
        write_node(&child, depth, preserve_whitespace, out);
    }
}

/// Only ASCII whitespace counts, as in HTML. `&nbsp;` stays significant.
fn normalize_whitespace(text: &str) -> String {
    text.split_ascii_whitespace().collect::<Vec<_>>().join(" ")
}

#[cfg(test)]
mod tests {
    use super::{assert_html_eq, canonicalize};

    #[test]
    fn attribute_order_is_ignored() {
        assert_html_eq(
            r#"<img src="a.jpg" alt="b" title="c">"#,
            r#"<img title="c" alt="b" src="a.jpg">"#,
        );
    }

    #[test]
    fn whitespace_is_ignored() {
        assert_html_eq(
            "<div>\n  <p>foo   bar</p>\n\n  <p> baz </p>\n</div>",
            "<div><p>foo bar</p><p>baz</p></div>",
        );
    }

    #[test]
    fn serialization_details_are_ignored() {
        assert_html_eq(
            r#"<!DOCTYPE html PUBLIC "-//W3C//DTD HTML 4.0 Transitional//EN" "http://www.w3.org/TR/REC-html40/loose.dtd">
<html><body><p title="&quot;x&quot;">a&nbsp;b<br/></p><source src="x"></source></body></html>"#,
            "<!DOCTYPE html><p title='\"x\"'>a\u{a0}b<br></p><source src=x>",
        );
    }

    #[test]
    fn pre_keeps_whitespace() {
        assert_ne!(
            canonicalize("<pre>a\n  b</pre>"),
            canonicalize("<pre>a b</pre>")
        );
    }

    #[test]
    fn nbsp_is_not_whitespace() {
        assert_ne!(canonicalize("<p>a&nbsp;b</p>"), canonicalize("<p>a b</p>"));
    }

    #[test]
    fn template_contents_are_compared() {
        assert_ne!(
            canonicalize("<template><p>a</p></template>"),
            canonicalize("<template><p>b</p></template>")
        );
    }

    #[test]
    #[should_panic(expected = "HTML documents differ")]
    fn different_content_fails() {
        assert_html_eq("<p>foo</p>", "<p>bar</p>");
    }

    #[test]
    #[should_panic(expected = "HTML documents differ")]
    fn different_attribute_value_fails() {
        assert_html_eq(r#"<a href="a">x</a>"#, r#"<a href="b">x</a>"#);
    }
}