Skip to main content

blitz_html/
html_sink.rs

1//! An implementation for Html5ever's sink trait, allowing us to parse HTML into a DOM.
2
3use html5ever::ParseOpts;
4use html5ever::tokenizer::TokenizerOpts;
5use html5ever::tree_builder::TreeBuilderOpts;
6use std::borrow::Cow;
7use std::cell::{Cell, Ref, RefCell, RefMut};
8
9use blitz_dom::node::Attribute;
10use blitz_dom::{DocumentMutator, HtmlParserProvider, NodeId};
11use html5ever::{
12    QualName,
13    tendril::{StrTendril, TendrilSink},
14    tree_builder::{ElementFlags, NodeOrText, QuirksMode, TreeSink},
15};
16
17/// Convert an html5ever Attribute, which uses tendril for its value, to a blitz
18/// Attribute, which interns its value.
19///
20/// This is the highest-volume construction site in the engine: every attribute
21/// of every element of every parsed document arrives here. It is therefore
22/// where interning pays off most, because a document's repeated `class` strings
23/// collapse as the tree is built rather than after it.
24fn html5ever_to_blitz_attr(attr: html5ever::Attribute) -> Attribute {
25    Attribute {
26        name: attr.name,
27        value: attr.value.as_ref().into(),
28    }
29}
30
31#[derive(Copy, Clone, Default, Debug)]
32pub struct HtmlProvider;
33
34impl HtmlParserProvider for HtmlProvider {
35    fn parse_inner_html<'m2, 'doc2>(
36        &self,
37        mutr: &'m2 mut DocumentMutator<'doc2>,
38        element_id: NodeId,
39        html: &str,
40    ) {
41        DocumentHtmlParser::parse_inner_html_into_mutator(mutr, element_id, html);
42    }
43
44    fn parse_document(
45        &self,
46        html: &str,
47        config: blitz_dom::DocumentConfig,
48    ) -> Box<dyn blitz_dom::Document> {
49        Box::new(crate::HtmlDocument::from_html(html, config))
50    }
51}
52
53pub struct DocumentHtmlParser<'m, 'doc> {
54    document_mutator: RefCell<&'m mut DocumentMutator<'doc>>,
55
56    /// Errors that occurred during parsing.
57    pub errors: RefCell<Vec<Cow<'static, str>>>,
58
59    /// The document's quirks mode.
60    pub quirks_mode: Cell<QuirksMode>,
61    pub is_xml: bool,
62}
63
64impl<'m, 'doc> DocumentHtmlParser<'m, 'doc> {
65    #[track_caller]
66    /// Get a mutable borrow of the DocumentMutator
67    fn mutr(&self) -> RefMut<'_, &'m mut DocumentMutator<'doc>> {
68        self.document_mutator.borrow_mut()
69    }
70}
71
72impl<'m, 'doc> DocumentHtmlParser<'m, 'doc> {
73    pub fn new(mutr: &'m mut DocumentMutator<'doc>) -> DocumentHtmlParser<'m, 'doc> {
74        DocumentHtmlParser {
75            document_mutator: RefCell::new(mutr),
76            errors: RefCell::new(Vec::new()),
77            quirks_mode: Cell::new(QuirksMode::NoQuirks),
78            is_xml: false,
79        }
80    }
81
82    /// Detects documents without an XML or DOCTYPE declaration whose root `<html>` element
83    /// declares the XHTML namespace (e.g. `<html xmlns="http://www.w3.org/1999/xhtml">`)
84    fn root_element_has_xhtml_namespace(html: &str) -> bool {
85        let rest = html.trim_start_matches('\u{feff}').trim_start();
86        let Some(rest) = rest.strip_prefix("<html") else {
87            return false;
88        };
89        let Some(tag_end) = rest.find('>') else {
90            return false;
91        };
92        rest[..tag_end].contains("xmlns=\"http://www.w3.org/1999/xhtml\"")
93            || rest[..tag_end].contains("xmlns='http://www.w3.org/1999/xhtml'")
94    }
95
96    pub fn parse_into_mutator<'a, 'd>(mutr: &'a mut DocumentMutator<'d>, html: &str) {
97        let mut sink = DocumentHtmlParser::new(mutr);
98
99        let is_xhtml_doc = html.starts_with("<?xml")
100            || html.starts_with("<!DOCTYPE") && {
101                let first_line = html.lines().next().unwrap();
102                first_line.contains("XHTML") || first_line.contains("xhtml")
103            }
104            || Self::root_element_has_xhtml_namespace(html);
105
106        if is_xhtml_doc {
107            // Parse as XHTML
108            sink.is_xml = true;
109            xml5ever::driver::parse_document(sink, Default::default())
110                .from_utf8()
111                .read_from(&mut html.as_bytes())
112                .unwrap();
113        } else {
114            // Parse as HTML
115            sink.is_xml = false;
116            let opts = ParseOpts {
117                tokenizer: TokenizerOpts::default(),
118                tree_builder: TreeBuilderOpts {
119                    exact_errors: false,
120                    scripting_enabled: false, // Enables parsing of <noscript> tags
121                    iframe_srcdoc: false,
122                    drop_doctype: true,
123                    quirks_mode: QuirksMode::NoQuirks,
124                },
125            };
126            html5ever::parse_document(sink, opts)
127                .from_utf8()
128                .read_from(&mut html.as_bytes())
129                .unwrap();
130        }
131    }
132
133    pub fn parse_inner_html_into_mutator<'a, 'd>(
134        mutr: &'a mut DocumentMutator<'d>,
135        element_id: NodeId,
136        html: &str,
137    ) {
138        let sink = DocumentHtmlParser::new(mutr);
139
140        let opts = ParseOpts {
141            tokenizer: TokenizerOpts::default(),
142            tree_builder: TreeBuilderOpts {
143                exact_errors: false,
144                scripting_enabled: false, // Enables parsing of <noscript> tags
145                iframe_srcdoc: false,
146                drop_doctype: true,
147                quirks_mode: QuirksMode::NoQuirks,
148            },
149        };
150        html5ever::driver::parse_fragment_for_element(sink, opts, element_id, false, None)
151            .from_utf8()
152            .read_from(&mut html.as_bytes())
153            .unwrap();
154
155        // html5ever creates a new fragment root node under the document node and parses the nodes into that fragment root.
156        // So here we move the children of the fragment root to element_id and then drop the fragment root.
157        let document_id = mutr.doc.root_node().id;
158        let fragment_root_id = mutr.last_child_id(document_id).unwrap();
159        let child_ids = mutr.child_ids(fragment_root_id);
160        mutr.append_children(element_id, &child_ids);
161        mutr.remove_and_drop_node(fragment_root_id);
162    }
163}
164
165impl<'m, 'doc> TreeSink for DocumentHtmlParser<'m, 'doc> {
166    type Output = ();
167
168    // we use the ID of the nodes in the tree as the handle
169    type Handle = NodeId;
170
171    type ElemName<'a>
172        = Ref<'a, QualName>
173    where
174        Self: 'a;
175
176    fn finish(self) -> Self::Output {
177        #[cfg(feature = "tracing")]
178        for error in self.errors.borrow().iter() {
179            tracing::error!("{error}");
180        }
181    }
182
183    fn parse_error(&self, msg: Cow<'static, str>) {
184        self.errors.borrow_mut().push(msg);
185    }
186
187    fn get_document(&self) -> Self::Handle {
188        self.document_mutator.borrow().doc.root_node().id
189    }
190
191    fn elem_name<'a>(&'a self, target: &'a Self::Handle) -> Self::ElemName<'a> {
192        Ref::map(self.document_mutator.borrow(), |docm| {
193            docm.element_name(*target)
194                .expect("TreeSink::elem_name called on a node which is not an element!")
195        })
196    }
197
198    fn create_element(
199        &self,
200        name: QualName,
201        attrs: Vec<html5ever::Attribute>,
202        _flags: ElementFlags,
203    ) -> Self::Handle {
204        let attrs = attrs.into_iter().map(html5ever_to_blitz_attr).collect();
205        self.mutr().create_element(name, attrs)
206    }
207
208    fn create_comment(&self, text: StrTendril) -> Self::Handle {
209        self.mutr().create_comment_node(&text)
210    }
211
212    fn create_pi(&self, _target: StrTendril, _data: StrTendril) -> Self::Handle {
213        self.mutr().create_comment_node("")
214    }
215
216    fn append(&self, parent_id: &Self::Handle, child: NodeOrText<Self::Handle>) {
217        match child {
218            NodeOrText::AppendNode(id) => self.mutr().append_children(*parent_id, &[id]),
219            // If content to append is text, first attempt to append it to the last child of parent.
220            // Else create a new text node and append it to the parent
221            NodeOrText::AppendText(text) => {
222                let last_child_id = self.mutr().last_child_id(*parent_id);
223                let has_appended = if let Some(id) = last_child_id {
224                    self.mutr().append_text_to_node(id, &text).is_ok()
225                } else {
226                    false
227                };
228                if !has_appended {
229                    let new_child_id = self.mutr().create_text_node(&text);
230                    self.mutr().append_children(*parent_id, &[new_child_id]);
231                }
232            }
233        }
234    }
235
236    // Note: The tree builder promises we won't have a text node after the insertion point.
237    // https://github.com/servo/html5ever/blob/main/rcdom/lib.rs#L338
238    fn append_before_sibling(&self, sibling_id: &Self::Handle, new_node: NodeOrText<Self::Handle>) {
239        match new_node {
240            NodeOrText::AppendNode(id) => self.mutr().insert_nodes_before(*sibling_id, &[id]),
241            // If content to append is text, first attempt to append it to the node before sibling_node
242            // Else create a new text node and insert it before sibling_node
243            NodeOrText::AppendText(text) => {
244                let previous_sibling_id = self.mutr().previous_sibling_id(*sibling_id);
245                let has_appended = if let Some(id) = previous_sibling_id {
246                    self.mutr().append_text_to_node(id, &text).is_ok()
247                } else {
248                    false
249                };
250                if !has_appended {
251                    let new_child_id = self.mutr().create_text_node(&text);
252                    self.mutr()
253                        .insert_nodes_before(*sibling_id, &[new_child_id]);
254                }
255            }
256        };
257    }
258
259    fn append_based_on_parent_node(
260        &self,
261        element: &Self::Handle,
262        prev_element: &Self::Handle,
263        child: NodeOrText<Self::Handle>,
264    ) {
265        if self.mutr().node_has_parent(*element) {
266            self.append_before_sibling(element, child);
267        } else {
268            self.append(prev_element, child);
269        }
270    }
271
272    fn append_doctype_to_document(
273        &self,
274        _name: StrTendril,
275        _public_id: StrTendril,
276        _system_id: StrTendril,
277    ) {
278        // Ignore. We don't care about the DOCTYPE for now.
279    }
280
281    fn get_template_contents(&self, target: &Self::Handle) -> Self::Handle {
282        // TODO: implement templates properly. This should allow to function like regular elements.
283        *target
284    }
285
286    fn same_node(&self, x: &Self::Handle, y: &Self::Handle) -> bool {
287        x == y
288    }
289
290    fn set_quirks_mode(&self, mode: QuirksMode) {
291        self.quirks_mode.set(mode);
292    }
293
294    fn add_attrs_if_missing(&self, target: &Self::Handle, attrs: Vec<html5ever::Attribute>) {
295        let attrs = attrs.into_iter().map(html5ever_to_blitz_attr).collect();
296        self.mutr().add_attrs_if_missing(*target, attrs);
297    }
298
299    fn remove_from_parent(&self, target: &Self::Handle) {
300        self.mutr().remove_node(*target);
301    }
302
303    fn reparent_children(&self, old_parent_id: &Self::Handle, new_parent_id: &Self::Handle) {
304        self.mutr()
305            .reparent_children(*old_parent_id, *new_parent_id);
306    }
307}
308
309#[test]
310fn parses_some_html() {
311    use blitz_dom::{BaseDocument, DocumentConfig};
312
313    let html = "<!DOCTYPE html><html><body><h1>hello world</h1></body></html>";
314    let mut doc = BaseDocument::new(DocumentConfig::default());
315    let mut mutr = doc.mutate();
316    let sink = DocumentHtmlParser::new(&mut mutr);
317
318    html5ever::parse_document(sink, Default::default())
319        .from_utf8()
320        .read_from(&mut html.as_bytes())
321        .unwrap();
322
323    drop(mutr);
324    doc.print_tree()
325
326    // Now our tree should have some nodes in it
327}