use dom_query::{Document, NodeData, NodeRef};
use std::fmt::Write;
const WHITESPACE_PRESERVING: &[&str] = &["pre", "textarea", "listing", "plaintext"];
const CONTEXT_LINES: usize = 5;
pub fn update_fixtures() -> bool {
std::env::var_os("UPDATE_FIXTURES").is_some_and(|value| value == "1")
}
#[track_caller]
pub fn check_fixture(expected_path: &str, actual: &str) {
if update_fixtures() {
std::fs::write(expected_path, actual)
.unwrap_or_else(|error| panic!("Failed to write {expected_path}: {error}"));
return;
}
let expected = std::fs::read_to_string(expected_path)
.unwrap_or_else(|error| panic!("Failed to read {expected_path}: {error}"));
assert_html_eq(&expected, actual);
}
#[track_caller]
pub fn check_text_fixture(expected_path: &str, actual: &str) {
if update_fixtures() {
std::fs::write(expected_path, actual)
.unwrap_or_else(|error| panic!("Failed to write {expected_path}: {error}"));
return;
}
let expected = std::fs::read_to_string(expected_path)
.unwrap_or_else(|error| panic!("Failed to read {expected_path}: {error}"));
assert_eq!(expected, actual, "{expected_path} differs");
}
#[track_caller]
pub fn assert_html_eq(expected: &str, actual: &str) {
let expected_canonical = canonicalize(expected);
let actual_canonical = canonicalize(actual);
if expected_canonical == actual_canonical {
return;
}
let expected_lines = expected_canonical.lines().collect::<Vec<_>>();
let actual_lines = actual_canonical.lines().collect::<Vec<_>>();
let first_diff = expected_lines
.iter()
.zip(actual_lines.iter())
.position(|(e, a)| e != a)
.unwrap_or(expected_lines.len().min(actual_lines.len()));
let excerpt = |lines: &[&str]| {
let start = first_diff.saturating_sub(CONTEXT_LINES);
let end = (first_diff + CONTEXT_LINES + 1).min(lines.len());
lines[start..end]
.iter()
.enumerate()
.map(|(i, line)| {
let marker = if start + i == first_diff { ">" } else { " " };
format!("{marker}{:>5} {line}\n", start + i + 1)
})
.collect::<String>()
};
panic!(
"HTML documents differ (canonical form, first difference at line {}):\n\nexpected:\n{}\nactual:\n{}",
first_diff + 1,
excerpt(&expected_lines),
excerpt(&actual_lines),
);
}
pub fn canonicalize(html: &str) -> String {
let document = Document::from(html);
let mut out = String::new();
write_node(&document.root(), 0, false, &mut out);
out
}
fn write_node(node: &NodeRef, depth: usize, preserve_whitespace: bool, out: &mut String) {
let indent = " ".repeat(depth);
let Some(data) = node.query(|node| node.data.clone()) else {
return;
};
match data {
NodeData::Document | NodeData::Fragment => {
write_children(node, depth, preserve_whitespace, out)
}
NodeData::Doctype { .. } | NodeData::ProcessingInstruction { .. } => {}
NodeData::Text { contents } => {
let text = if preserve_whitespace {
contents.to_string()
} else {
normalize_whitespace(&contents)
};
if !text.is_empty() {
_ = writeln!(out, "{indent}{text:?}");
}
}
NodeData::Comment { contents } => {
_ = writeln!(
out,
"{indent}<!-- {:?} -->",
normalize_whitespace(&contents)
);
}
NodeData::Element(element) => {
let mut attrs = element
.attrs
.iter()
.map(|attr| {
let name = match &attr.name.prefix {
Some(prefix) => format!("{prefix}:{}", attr.name.local),
None => attr.name.local.to_string(),
};
(name, attr.value.to_string())
})
.collect::<Vec<_>>();
attrs.sort();
_ = write!(out, "{indent}<{}", element.name.local);
for (name, value) in attrs {
_ = write!(out, " {name}={value:?}");
}
_ = writeln!(out, ">");
let preserve_whitespace =
preserve_whitespace || WHITESPACE_PRESERVING.contains(&element.name.local.as_ref());
write_children(node, depth + 1, preserve_whitespace, out);
if let Some(template_contents) = element.template_contents {
let template_contents = NodeRef::new(template_contents, node.tree);
write_children(&template_contents, depth + 1, preserve_whitespace, out);
}
}
}
}
fn write_children(node: &NodeRef, depth: usize, preserve_whitespace: bool, out: &mut String) {
for child in node.children_it(false) {
write_node(&child, depth, preserve_whitespace, out);
}
}
fn normalize_whitespace(text: &str) -> String {
text.split_ascii_whitespace().collect::<Vec<_>>().join(" ")
}
#[cfg(test)]
mod tests {
use super::{assert_html_eq, canonicalize};
#[test]
fn attribute_order_is_ignored() {
assert_html_eq(
r#"<img src="a.jpg" alt="b" title="c">"#,
r#"<img title="c" alt="b" src="a.jpg">"#,
);
}
#[test]
fn whitespace_is_ignored() {
assert_html_eq(
"<div>\n <p>foo bar</p>\n\n <p> baz </p>\n</div>",
"<div><p>foo bar</p><p>baz</p></div>",
);
}
#[test]
fn serialization_details_are_ignored() {
assert_html_eq(
r#"<!DOCTYPE html PUBLIC "-//W3C//DTD HTML 4.0 Transitional//EN" "http://www.w3.org/TR/REC-html40/loose.dtd">
<html><body><p title=""x"">a b<br/></p><source src="x"></source></body></html>"#,
"<!DOCTYPE html><p title='\"x\"'>a\u{a0}b<br></p><source src=x>",
);
}
#[test]
fn pre_keeps_whitespace() {
assert_ne!(
canonicalize("<pre>a\n b</pre>"),
canonicalize("<pre>a b</pre>")
);
}
#[test]
fn nbsp_is_not_whitespace() {
assert_ne!(canonicalize("<p>a b</p>"), canonicalize("<p>a b</p>"));
}
#[test]
fn template_contents_are_compared() {
assert_ne!(
canonicalize("<template><p>a</p></template>"),
canonicalize("<template><p>b</p></template>")
);
}
#[test]
#[should_panic(expected = "HTML documents differ")]
fn different_content_fails() {
assert_html_eq("<p>foo</p>", "<p>bar</p>");
}
#[test]
#[should_panic(expected = "HTML documents differ")]
fn different_attribute_value_fails() {
assert_html_eq(r#"<a href="a">x</a>"#, r#"<a href="b">x</a>"#);
}
}