Skip to main content

rdom_parser/
parser.rs

1//! Recursive-descent HTML-ish parser.
2//!
3//! Consumes a template string, emits `Dom<Ext>` tree under a mount
4//! NodeId. Supports:
5//!
6//! - Start tags (`<tag>`), end tags (`</tag>`), self-closing (`<br/>`)
7//! - Void elements (`<br>`, `<hr>`, `<img>`, …) auto-close without `/>`
8//! - Attributes: `name="value"` / `name='value'` / `name=value` / `name`
9//! - Text with entity decoding: the common named references
10//!   (`&amp; &lt; &copy; &mdash; …`) plus `&#NNN;` / `&#xHH;`; U+0000,
11//!   surrogates and out-of-range code points decode to U+FFFD
12//! - A `<` not followed by an ASCII letter, `/`, `!`, or `?` is text
13//!   (HTML §13.2.5.6 tag-open state), so `a < b` needs no escaping
14//! - `<style>` / `<script>` bodies are RAWTEXT (no tags, no entities)
15//!   and `<textarea>` / `<title>` bodies are RCDATA (entities only),
16//!   each ending at its own case-insensitive end tag (HTML §13.2.5.3–6)
17//! - Comments: `<!-- … -->` preserved as Comment nodes
18//! - `<!DOCTYPE …>` is consumed and produces no node; any other `<!…>`
19//!   or `<?…>` is a bogus comment kept as a Comment node (§13.2.5.41)
20//! - A newline right after `<textarea>` is dropped (§13.2.6.4.7)
21//! - Case-insensitive tag names (tags are normalized to lowercase)
22//!
23//! Out of scope: CDATA, namespace prefixes, processing instructions,
24//! tree-construction error recovery (a mismatched or missing end tag is
25//! a hard error — this is a template parser, not a browser).
26
27use rdom_core::{Dom, NodeId};
28
29use crate::error::{ParseError, Result};
30
31/// HTML5 void tag set — never have children, always self-close.
32/// Mirror of `rdom_core::markup::VOID_TAGS` (we don't depend on that
33/// private constant, so redeclare here).
34const VOID_TAGS: &[&str] = &[
35    "area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", "param", "source",
36    "track", "wbr", "vr",
37];
38
39fn is_void_tag(tag: &str) -> bool {
40    VOID_TAGS.contains(&tag)
41}
42
43/// Parse `template` into a fresh `Dom<Ext>` with a Fragment root. The
44/// returned ids are the top-level children of the fragment.
45pub fn parse<Ext>(template: &str) -> Result<(Dom<Ext>, Vec<NodeId>)>
46where
47    Ext: Default + 'static,
48{
49    let mut dom = Dom::new();
50    let root = dom.root();
51    let ids = parse_into(&mut dom, template, root)?;
52    Ok((dom, ids))
53}
54
55/// Parse `template` and append the parsed tree under `mount`. Returns
56/// the ids of the top-level parsed nodes (direct children appended to
57/// `mount`). Does not alter existing children of `mount`.
58pub fn parse_into<Ext>(dom: &mut Dom<Ext>, template: &str, mount: NodeId) -> Result<Vec<NodeId>>
59where
60    Ext: Default + 'static,
61{
62    let mut p = Parser::new(template);
63    let ids = p.parse_nodes(dom, mount)?;
64    if !p.eof() {
65        // `parse_nodes` stops at `</…`; at the top level nothing is open,
66        // so a stray end tag is an error rather than silent truncation.
67        return Err(p
68            .err("unexpected closing tag at top level")
69            .with_hint("nothing is open here — remove the end tag or open its element"));
70    }
71    Ok(ids)
72}
73
74// ─── Internal parser ────────────────────────────────────────────────
75
76struct Parser<'a> {
77    src: &'a str,
78    bytes: &'a [u8],
79    pos: usize,
80    line: u32,
81    col: u32,
82}
83
84impl<'a> Parser<'a> {
85    fn new(src: &'a str) -> Self {
86        Self {
87            src,
88            bytes: src.as_bytes(),
89            pos: 0,
90            line: 1,
91            col: 1,
92        }
93    }
94
95    // ── Cursor ────────────────────────────────────────────────────
96
97    fn eof(&self) -> bool {
98        self.pos >= self.bytes.len()
99    }
100
101    fn peek(&self) -> Option<u8> {
102        self.bytes.get(self.pos).copied()
103    }
104
105    fn peek_at(&self, offset: usize) -> Option<u8> {
106        self.bytes.get(self.pos + offset).copied()
107    }
108
109    fn starts_with(&self, needle: &str) -> bool {
110        self.src[self.pos..].starts_with(needle)
111    }
112
113    fn advance(&mut self) -> Option<u8> {
114        let b = self.peek()?;
115        self.pos += 1;
116        if b == b'\n' {
117            self.line += 1;
118            self.col = 1;
119        } else {
120            self.col += 1;
121        }
122        Some(b)
123    }
124
125    fn advance_n(&mut self, n: usize) {
126        for _ in 0..n {
127            if self.advance().is_none() {
128                break;
129            }
130        }
131    }
132
133    fn skip_ws(&mut self) {
134        while let Some(b) = self.peek() {
135            if b.is_ascii_whitespace() {
136                self.advance();
137            } else {
138                break;
139            }
140        }
141    }
142
143    fn err(&self, msg: impl Into<String>) -> ParseError {
144        ParseError::new(msg, self.line, self.col, self.pos)
145    }
146
147    // ── Top-level: parse children of `parent` ─────────────────────
148
149    fn parse_nodes<Ext>(&mut self, dom: &mut Dom<Ext>, parent: NodeId) -> Result<Vec<NodeId>>
150    where
151        Ext: Default + 'static,
152    {
153        let mut out = Vec::new();
154        loop {
155            if self.eof() {
156                break;
157            }
158            if self.starts_with("</") {
159                // Bubble up to the containing element parser.
160                break;
161            }
162            if self.starts_with("<!--") {
163                let id = self.parse_comment(dom, parent)?;
164                out.push(id);
165                continue;
166            }
167            if self.starts_with("<!") {
168                if self.src[self.pos + 2..]
169                    .get(..7)
170                    .is_some_and(|k| k.eq_ignore_ascii_case("DOCTYPE"))
171                {
172                    // `<!DOCTYPE html>`: consumed, no node (rdom has no
173                    // DocumentType node — see DIVERGENCES §HTML parsing).
174                    self.skip_declaration();
175                } else {
176                    // Any other `<!…>` is a bogus comment (§13.2.5.42).
177                    let id = self.parse_bogus_comment(dom, parent)?;
178                    out.push(id);
179                }
180                continue;
181            }
182            if self.starts_with("<?") {
183                // `<?…>` is a bogus comment (§13.2.5.6 "?" branch).
184                let id = self.parse_bogus_comment(dom, parent)?;
185                out.push(id);
186                continue;
187            }
188            if self.peek() == Some(b'<') && self.peek_at(1).is_some_and(|b| b.is_ascii_alphabetic())
189            {
190                let id = self.parse_element(dom, parent)?;
191                out.push(id);
192                continue;
193            }
194            // Plain text until the next tag open. A `<` not followed by
195            // an ASCII letter, `/`, `!`, or `?` is text (§13.2.5.6).
196            let id = self.parse_text(dom, parent)?;
197            if let Some(id) = id {
198                out.push(id);
199            }
200        }
201        Ok(out)
202    }
203
204    // ── Comment ────────────────────────────────────────────────────
205
206    fn parse_comment<Ext>(&mut self, dom: &mut Dom<Ext>, parent: NodeId) -> Result<NodeId>
207    where
208        Ext: Default + 'static,
209    {
210        // Consume `<!--`.
211        self.advance_n(4);
212        let start = self.pos;
213        loop {
214            if self.eof() {
215                return Err(self
216                    .err("unterminated comment")
217                    .with_hint("missing `-->` closing"));
218            }
219            if self.starts_with("-->") {
220                let data = &self.src[start..self.pos];
221                self.advance_n(3);
222                let id = dom.create_comment(data);
223                dom.append_child(parent, id)
224                    .map_err(|e| self.err(format!("failed to append comment: {:?}", e)))?;
225                return Ok(id);
226            }
227            self.advance();
228        }
229    }
230
231    // ── Text ───────────────────────────────────────────────────────
232
233    /// Consume chars until next `<`, decode entities, emit a Text node.
234    /// Returns `None` when the captured text is empty (no Text node
235    /// created). UTF-8-safe — we slice by byte but the boundaries are
236    /// always on valid char boundaries because we advance byte-at-a-time
237    /// only via `self.advance()` which respects the source encoding.
238    fn parse_text<Ext>(&mut self, dom: &mut Dom<Ext>, parent: NodeId) -> Result<Option<NodeId>>
239    where
240        Ext: Default + 'static,
241    {
242        let mut out = String::new();
243        loop {
244            // Collect consecutive raw (non-'<', non-'&') bytes as a
245            // UTF-8 slice from the source.
246            let slice_start = self.pos;
247            while let Some(b) = self.peek() {
248                if b == b'&' || (b == b'<' && self.at_tag_open()) {
249                    break;
250                }
251                self.advance();
252            }
253            if slice_start < self.pos {
254                out.push_str(&self.src[slice_start..self.pos]);
255            }
256            match self.peek() {
257                None | Some(b'<') => break,
258                Some(b'&') => {
259                    out.push_str(&self.parse_entity()?);
260                }
261                _ => unreachable!(),
262            }
263        }
264        if out.is_empty() {
265            return Ok(None);
266        }
267        let id = dom.create_text_node(&out);
268        dom.append_child(parent, id)
269            .map_err(|e| self.err(format!("failed to append text: {:?}", e)))?;
270        Ok(Some(id))
271    }
272
273    /// Is the `<` at the cursor a real tag open (HTML §13.2.5.6)? Only
274    /// when followed by an ASCII letter, `/`, `!`, or `?`.
275    fn at_tag_open(&self) -> bool {
276        self.peek() == Some(b'<')
277            && self
278                .peek_at(1)
279                .is_some_and(|b| b.is_ascii_alphabetic() || matches!(b, b'/' | b'!' | b'?'))
280    }
281
282    /// HTML §13.2.5.41 bogus comment state: everything from just after
283    /// the `<` up to the next `>` becomes a Comment node's data
284    /// (`<?xml version="1.0"?>` → `?xml version="1.0"?`).
285    fn parse_bogus_comment<Ext>(&mut self, dom: &mut Dom<Ext>, parent: NodeId) -> Result<NodeId>
286    where
287        Ext: Default + 'static,
288    {
289        self.advance(); // '<'
290        let start = self.pos;
291        while let Some(b) = self.peek() {
292            if b == b'>' {
293                break;
294            }
295            self.advance();
296        }
297        let data = self.src[start..self.pos].to_string();
298        if self.peek() == Some(b'>') {
299            self.advance();
300        }
301        let id = dom.create_comment(&data);
302        dom.append_child(parent, id)
303            .map_err(|e| self.err(format!("failed to append comment: {:?}", e)))?;
304        Ok(id)
305    }
306
307    /// Consume a `<!DOCTYPE …>` declaration through its closing `>`, or
308    /// to EOF.
309    fn skip_declaration(&mut self) {
310        while let Some(b) = self.advance() {
311            if b == b'>' {
312                return;
313            }
314        }
315    }
316
317    /// Byte offset of the `</tag` (ASCII-case-insensitive, followed by
318    /// whitespace, `/`, or `>`) that ends a RAWTEXT / RCDATA element,
319    /// searching from the cursor. `None` at EOF.
320    fn find_end_tag(&self, tag_lc: &str) -> Option<usize> {
321        let hay = &self.bytes[self.pos..];
322        let needle_len = 2 + tag_lc.len();
323        let mut i = 0;
324        while i + needle_len <= hay.len() {
325            if hay[i] == b'<' && hay[i + 1] == b'/' {
326                let name = &hay[i + 2..i + needle_len];
327                if name.eq_ignore_ascii_case(tag_lc.as_bytes()) {
328                    let after = hay.get(i + needle_len).copied();
329                    if after.is_none_or(|b| b.is_ascii_whitespace() || b == b'>' || b == b'/') {
330                        return Some(self.pos + i);
331                    }
332                }
333            }
334            i += 1;
335        }
336        None
337    }
338
339    /// Parse the body of a RAWTEXT (`<style>`, `<script>`) or RCDATA
340    /// (`<textarea>`, `<title>`) element as a single text node, then
341    /// consume its end tag. RCDATA decodes character references; RAWTEXT
342    /// takes the bytes verbatim (HTML §13.2.5.3–6).
343    fn parse_special_text<Ext>(
344        &mut self,
345        dom: &mut Dom<Ext>,
346        element: NodeId,
347        tag_lc: &str,
348        decode_entities: bool,
349    ) -> Result<()>
350    where
351        Ext: Default + 'static,
352    {
353        let Some(end) = self.find_end_tag(tag_lc) else {
354            return Err(self
355                .err(format!("missing closing tag for <{}>", tag_lc))
356                .with_hint(format!("add </{}> to close", tag_lc)));
357        };
358        let mut raw = &self.src[self.pos..end];
359        // HTML §13.2.6.4.7: a newline immediately after `<textarea>` is
360        // ignored (the same rule HTML applies to `<pre>` / `<listing>`).
361        if tag_lc == "textarea" {
362            raw = raw
363                .strip_prefix("\r\n")
364                .or_else(|| raw.strip_prefix('\n'))
365                .unwrap_or(raw);
366        }
367        let text = if decode_entities {
368            decode_character_references(raw)
369        } else {
370            raw.to_string()
371        };
372        if !text.is_empty() {
373            let id = dom.create_text_node(&text);
374            dom.append_child(element, id)
375                .map_err(|e| self.err(format!("failed to append text: {:?}", e)))?;
376        }
377        // Walk (not jump) so line / column stay right for later errors.
378        self.advance_n(end - self.pos);
379        self.advance_n(2 + tag_lc.len()); // `</tag`
380        self.skip_ws();
381        if self.peek() != Some(b'>') {
382            return Err(self
383                .err(format!("expected `>` in </{}>", tag_lc))
384                .with_hint("no attributes on closing tags"));
385        }
386        self.advance();
387        Ok(())
388    }
389
390    // ── Entity ─────────────────────────────────────────────────────
391
392    fn parse_entity(&mut self) -> Result<String> {
393        // We've seen '&'. One scanner for the text path and the RCDATA
394        // path (`scan_reference`), so both agree on what a reference
395        // looks like; unknown or malformed input keeps the `&` literal
396        // (lenient, as browsers flush the raw characters).
397        debug_assert_eq!(self.peek(), Some(b'&'));
398        match scan_reference(&self.src[self.pos + 1..]) {
399            Some((decoded, consumed)) => {
400                self.advance_n(1 + consumed);
401                Ok(decoded)
402            }
403            None => {
404                self.advance(); // consume the '&'
405                Ok("&".to_string())
406            }
407        }
408    }
409
410    // ── Element ────────────────────────────────────────────────────
411
412    fn parse_element<Ext>(&mut self, dom: &mut Dom<Ext>, parent: NodeId) -> Result<NodeId>
413    where
414        Ext: Default + 'static,
415    {
416        debug_assert_eq!(self.peek(), Some(b'<'));
417        self.advance(); // '<'
418
419        let tag = self.parse_tag_name()?;
420        let tag_lc = tag.to_ascii_lowercase();
421
422        let element = dom.create_element(&tag_lc);
423
424        // Parse attributes until '>' or '/>'.
425        loop {
426            self.skip_ws();
427            match self.peek() {
428                None => {
429                    return Err(self
430                        .err(format!("unexpected EOF inside <{}>", tag_lc))
431                        .with_hint("missing closing `>`"));
432                }
433                Some(b'>') => {
434                    self.advance();
435                    break;
436                }
437                Some(b'/') => {
438                    // Self-closing.
439                    self.advance();
440                    self.skip_ws();
441                    if self.peek() != Some(b'>') {
442                        return Err(self
443                            .err(format!("expected `>` after `/` in <{}/>", tag_lc))
444                            .with_hint("self-closing syntax is `/>`"));
445                    }
446                    self.advance();
447                    dom.append_child(parent, element)
448                        .map_err(|e| self.err(format!("failed to append <{}>: {:?}", tag_lc, e)))?;
449                    return Ok(element);
450                }
451                Some(_) => {
452                    self.parse_attribute(dom, element)?;
453                }
454            }
455        }
456
457        // Void tag? Done.
458        if is_void_tag(&tag_lc) {
459            dom.append_child(parent, element)
460                .map_err(|e| self.err(format!("failed to append <{}>: {:?}", tag_lc, e)))?;
461            return Ok(element);
462        }
463
464        // RAWTEXT / RCDATA elements take their body as one text node
465        // up to their own end tag.
466        match tag_lc.as_str() {
467            "style" | "script" => {
468                self.parse_special_text(dom, element, &tag_lc, false)?;
469                dom.append_child(parent, element)
470                    .map_err(|e| self.err(format!("failed to append <{}>: {:?}", tag_lc, e)))?;
471                return Ok(element);
472            }
473            "textarea" | "title" => {
474                self.parse_special_text(dom, element, &tag_lc, true)?;
475                dom.append_child(parent, element)
476                    .map_err(|e| self.err(format!("failed to append <{}>: {:?}", tag_lc, e)))?;
477                return Ok(element);
478            }
479            _ => {}
480        }
481
482        // Parse children, then expect </tag>.
483        self.parse_nodes(dom, element)?;
484
485        if !self.starts_with("</") {
486            return Err(self
487                .err(format!("missing closing tag for <{}>", tag_lc))
488                .with_hint(format!("add </{}> to close", tag_lc)));
489        }
490        self.advance_n(2); // '</'
491
492        let close_tag = self.parse_tag_name()?;
493        if close_tag.to_ascii_lowercase() != tag_lc {
494            return Err(self
495                .err(format!(
496                    "mismatched closing tag: found </{}>, expected </{}>",
497                    close_tag, tag_lc
498                ))
499                .with_hint("tags must be properly nested"));
500        }
501        self.skip_ws();
502        if self.peek() != Some(b'>') {
503            return Err(self
504                .err(format!("expected `>` in </{}>", tag_lc))
505                .with_hint("no attributes on closing tags"));
506        }
507        self.advance();
508
509        dom.append_child(parent, element)
510            .map_err(|e| self.err(format!("failed to append <{}>: {:?}", tag_lc, e)))?;
511        Ok(element)
512    }
513
514    fn parse_tag_name(&mut self) -> Result<String> {
515        let start = self.pos;
516        while let Some(b) = self.peek() {
517            if b.is_ascii_alphanumeric() || b == b'-' || b == b'_' {
518                self.advance();
519            } else {
520                break;
521            }
522        }
523        if start == self.pos {
524            return Err(self
525                .err("expected tag name")
526                .with_hint("tag names start with a letter"));
527        }
528        Ok(self.src[start..self.pos].to_string())
529    }
530
531    fn parse_attribute<Ext>(&mut self, dom: &mut Dom<Ext>, element: NodeId) -> Result<()>
532    where
533        Ext: Default + 'static,
534    {
535        let name = self.parse_attr_name()?;
536        self.skip_ws();
537
538        let value = if self.peek() == Some(b'=') {
539            self.advance();
540            self.skip_ws();
541            Some(self.parse_attr_value()?)
542        } else {
543            None
544        };
545
546        match value {
547            Some(v) => {
548                // Classes are normalized into the classList; other
549                // attrs go into the attribute map.
550                if name.eq_ignore_ascii_case("class") {
551                    for token in v.split_ascii_whitespace() {
552                        dom.add_class(element, token)
553                            .map_err(|e| self.err(format!("failed to add class: {:?}", e)))?;
554                    }
555                } else {
556                    dom.set_attribute(element, &name, &v)
557                        .map_err(|e| self.err(format!("failed to set attribute: {:?}", e)))?;
558                }
559            }
560            None => {
561                // Boolean attribute.
562                dom.set_attribute(element, &name, "")
563                    .map_err(|e| self.err(format!("failed to set attribute: {:?}", e)))?;
564            }
565        }
566        Ok(())
567    }
568
569    fn parse_attr_name(&mut self) -> Result<String> {
570        let start = self.pos;
571        while let Some(b) = self.peek() {
572            if b.is_ascii_alphanumeric() || b == b'-' || b == b'_' || b == b':' {
573                self.advance();
574            } else {
575                break;
576            }
577        }
578        if start == self.pos {
579            return Err(self.err("expected attribute name"));
580        }
581        Ok(self.src[start..self.pos].to_string())
582    }
583
584    fn parse_attr_value(&mut self) -> Result<String> {
585        let first = self.peek();
586        match first {
587            Some(b'"') => self.parse_quoted(b'"'),
588            Some(b'\'') => self.parse_quoted(b'\''),
589            Some(_) => self.parse_unquoted(),
590            None => Err(self.err("unexpected EOF in attribute value")),
591        }
592    }
593
594    fn parse_quoted(&mut self, quote: u8) -> Result<String> {
595        self.advance(); // opening quote
596        let mut out = String::new();
597        loop {
598            // Collect consecutive raw bytes up to the next quote or &.
599            let slice_start = self.pos;
600            while let Some(b) = self.peek() {
601                if b == quote || b == b'&' {
602                    break;
603                }
604                self.advance();
605            }
606            if slice_start < self.pos {
607                out.push_str(&self.src[slice_start..self.pos]);
608            }
609            match self.peek() {
610                None => {
611                    return Err(self
612                        .err(format!(
613                            "unterminated attribute value (expected `{}`)",
614                            quote as char
615                        ))
616                        .with_hint("missing closing quote"));
617                }
618                Some(b) if b == quote => {
619                    self.advance();
620                    return Ok(out);
621                }
622                Some(b'&') => {
623                    out.push_str(&self.parse_entity()?);
624                }
625                _ => unreachable!(),
626            }
627        }
628    }
629
630    fn parse_unquoted(&mut self) -> Result<String> {
631        let mut out = String::new();
632        loop {
633            let slice_start = self.pos;
634            while let Some(b) = self.peek() {
635                if b.is_ascii_whitespace() || b == b'>' || b == b'/' || b == b'&' {
636                    break;
637                }
638                self.advance();
639            }
640            if slice_start < self.pos {
641                out.push_str(&self.src[slice_start..self.pos]);
642            }
643            match self.peek() {
644                Some(b'&') => out.push_str(&self.parse_entity()?),
645                _ => break,
646            }
647        }
648        if out.is_empty() {
649            return Err(self
650                .err("empty unquoted attribute value")
651                .with_hint("use \"\" or '' for empty value"));
652        }
653        Ok(out)
654    }
655}
656
657// ─── Entity decoding ────────────────────────────────────────────────
658
659/// Decode an entity body (chars between `&` and `;`). Returns `None`
660/// for unrecognized input — caller emits the `&` literal.
661fn decode_entity_body(body: &str) -> Option<String> {
662    if let Some(rest) = body.strip_prefix('#') {
663        // Digits only (no sign); a value that overflows `u32` is still
664        // "a number above U+10FFFF" and yields U+FFFD (§13.2.5.80).
665        let (digits, radix) = match rest.strip_prefix('x').or_else(|| rest.strip_prefix('X')) {
666            Some(hex) => (hex, 16),
667            None => (rest, 10),
668        };
669        if digits.is_empty() || !digits.bytes().all(|b| (b as char).is_digit(radix)) {
670            return None;
671        }
672        let n = u32::from_str_radix(digits, radix).unwrap_or(u32::MAX);
673        // HTML §13.2.5.80: U+0000, surrogates, and code points above
674        // U+10FFFF are parse errors that yield U+FFFD.
675        let c = match n {
676            0 | 0xD800..=0xDFFF => '\u{FFFD}',
677            _ => char::from_u32(n).unwrap_or('\u{FFFD}'),
678        };
679        return Some(c.to_string());
680    }
681    NAMED_REFERENCES
682        .binary_search_by(|(name, _)| (*name).cmp(body))
683        .ok()
684        .map(|i| NAMED_REFERENCES[i].1.to_string())
685}
686
687/// Scan a character reference whose `&` has just been consumed:
688/// `after_amp` starts right after it. Returns the decoded text and the
689/// number of bytes to consume (body plus `;`) when `after_amp` starts
690/// with 1..=16 name / digit / `#` / hex-marker characters followed by
691/// `;` and the body decodes; `None` otherwise (caller keeps `&`).
692fn scan_reference(after_amp: &str) -> Option<(String, usize)> {
693    let bytes = after_amp.as_bytes();
694    let mut end = None;
695    for (i, &b) in bytes.iter().enumerate().take(17) {
696        if b == b';' {
697            end = Some(i);
698            break;
699        }
700        if !(b.is_ascii_alphanumeric() || b == b'#') {
701            return None;
702        }
703    }
704    let end = end?;
705    if end == 0 {
706        return None;
707    }
708    let decoded = decode_entity_body(&after_amp[..end])?;
709    Some((decoded, end + 1))
710}
711
712/// Decode every `&…;` reference in `text` (RCDATA bodies) with the same
713/// scanner the text path uses.
714fn decode_character_references(text: &str) -> String {
715    let mut out = String::with_capacity(text.len());
716    let mut rest = text;
717    while let Some(amp) = rest.find('&') {
718        out.push_str(&rest[..amp]);
719        let after = &rest[amp + 1..];
720        match scan_reference(after) {
721            Some((decoded, consumed)) => {
722                out.push_str(&decoded);
723                rest = &after[consumed..];
724            }
725            None => {
726                out.push('&');
727                rest = after;
728            }
729        }
730    }
731    out.push_str(rest);
732    out
733}
734
735/// The named character references a template author is likely to type.
736/// Sorted by name for binary search; the full HTML table has 2 231
737/// entries and is not worth its size for a TUI template language.
738const NAMED_REFERENCES: &[(&str, &str)] = &[
739    ("AElig", "\u{C6}"),
740    ("Aacute", "\u{C1}"),
741    ("Agrave", "\u{C0}"),
742    ("Auml", "\u{C4}"),
743    ("Ccedil", "\u{C7}"),
744    ("Dagger", "\u{2021}"),
745    ("Eacute", "\u{C9}"),
746    ("Egrave", "\u{C8}"),
747    ("Ntilde", "\u{D1}"),
748    ("Oslash", "\u{D8}"),
749    ("Ouml", "\u{D6}"),
750    ("Prime", "\u{2033}"),
751    ("Uuml", "\u{DC}"),
752    ("aacute", "\u{E1}"),
753    ("acute", "\u{B4}"),
754    ("aelig", "\u{E6}"),
755    ("agrave", "\u{E0}"),
756    ("amp", "&"),
757    ("apos", "'"),
758    ("aring", "\u{E5}"),
759    ("asymp", "\u{2248}"),
760    ("auml", "\u{E4}"),
761    ("bdquo", "\u{201E}"),
762    ("brvbar", "\u{A6}"),
763    ("bull", "\u{2022}"),
764    ("ccedil", "\u{E7}"),
765    ("cedil", "\u{B8}"),
766    ("cent", "\u{A2}"),
767    ("check", "\u{2713}"),
768    ("clubs", "\u{2663}"),
769    ("copy", "\u{A9}"),
770    ("crarr", "\u{21B5}"),
771    ("curren", "\u{A4}"),
772    ("dagger", "\u{2020}"),
773    ("darr", "\u{2193}"),
774    ("deg", "\u{B0}"),
775    ("diams", "\u{2666}"),
776    ("divide", "\u{F7}"),
777    ("eacute", "\u{E9}"),
778    ("egrave", "\u{E8}"),
779    ("equiv", "\u{2261}"),
780    ("euro", "\u{20AC}"),
781    ("frac12", "\u{BD}"),
782    ("frac14", "\u{BC}"),
783    ("frac34", "\u{BE}"),
784    ("ge", "\u{2265}"),
785    ("gt", ">"),
786    ("harr", "\u{2194}"),
787    ("hearts", "\u{2665}"),
788    ("hellip", "\u{2026}"),
789    ("iexcl", "\u{A1}"),
790    ("infin", "\u{221E}"),
791    ("iquest", "\u{BF}"),
792    ("laquo", "\u{AB}"),
793    ("larr", "\u{2190}"),
794    ("ldquo", "\u{201C}"),
795    ("le", "\u{2264}"),
796    ("loz", "\u{25CA}"),
797    ("lsaquo", "\u{2039}"),
798    ("lsquo", "\u{2018}"),
799    ("lt", "<"),
800    ("macr", "\u{AF}"),
801    ("mdash", "\u{2014}"),
802    ("micro", "\u{B5}"),
803    ("middot", "\u{B7}"),
804    ("minus", "\u{2212}"),
805    ("nbsp", "\u{A0}"),
806    ("ndash", "\u{2013}"),
807    ("ne", "\u{2260}"),
808    ("not", "\u{AC}"),
809    ("ntilde", "\u{F1}"),
810    ("oslash", "\u{F8}"),
811    ("ouml", "\u{F6}"),
812    ("para", "\u{B6}"),
813    ("permil", "\u{2030}"),
814    ("plusmn", "\u{B1}"),
815    ("pound", "\u{A3}"),
816    ("prime", "\u{2032}"),
817    ("quot", "\""),
818    ("raquo", "\u{BB}"),
819    ("rarr", "\u{2192}"),
820    ("rdquo", "\u{201D}"),
821    ("reg", "\u{AE}"),
822    ("rsaquo", "\u{203A}"),
823    ("rsquo", "\u{2019}"),
824    ("sbquo", "\u{201A}"),
825    ("sect", "\u{A7}"),
826    ("shy", "\u{AD}"),
827    ("spades", "\u{2660}"),
828    ("sup2", "\u{B2}"),
829    ("sup3", "\u{B3}"),
830    ("szlig", "\u{DF}"),
831    ("times", "\u{D7}"),
832    ("trade", "\u{2122}"),
833    ("uarr", "\u{2191}"),
834    ("uml", "\u{A8}"),
835    ("uuml", "\u{FC}"),
836    ("yen", "\u{A5}"),
837];
838
839#[cfg(test)]
840mod tests {
841    use super::*;
842
843    fn parse_str(s: &str) -> (Dom<()>, Vec<NodeId>) {
844        parse(s).unwrap()
845    }
846
847    /// `decode_entity_body` binary-searches the table, so it must stay
848    /// byte-sorted and free of duplicates.
849    #[test]
850    fn named_reference_table_is_sorted_and_unique() {
851        for w in NAMED_REFERENCES.windows(2) {
852            assert!(
853                w[0].0 < w[1].0,
854                "{:?} must sort before {:?}",
855                w[0].0,
856                w[1].0
857            );
858        }
859    }
860
861    // ── Basic elements ───────────────────────────────────────────────
862
863    #[test]
864    fn empty_element() {
865        let (dom, ids) = parse_str("<div></div>");
866        assert_eq!(ids.len(), 1);
867        let n = dom.node(ids[0]);
868        assert_eq!(n.tag_name(), Some("div"));
869        assert_eq!(n.child_nodes().count(), 0);
870    }
871
872    #[test]
873    fn self_closing_element() {
874        let (dom, ids) = parse_str("<br/>");
875        assert_eq!(ids.len(), 1);
876        assert_eq!(dom.node(ids[0]).tag_name(), Some("br"));
877    }
878
879    #[test]
880    fn self_closing_with_space() {
881        let (dom, ids) = parse_str("<br />");
882        assert_eq!(dom.node(ids[0]).tag_name(), Some("br"));
883    }
884
885    #[test]
886    fn void_element_auto_closes() {
887        // `<br>` without `/>` still treated as void.
888        let (dom, ids) = parse_str("<br>");
889        assert_eq!(ids.len(), 1);
890        assert_eq!(dom.node(ids[0]).tag_name(), Some("br"));
891    }
892
893    #[test]
894    fn multiple_void_elements() {
895        let (dom, ids) = parse_str("<br><hr><img>");
896        assert_eq!(ids.len(), 3);
897        assert_eq!(dom.node(ids[0]).tag_name(), Some("br"));
898        assert_eq!(dom.node(ids[1]).tag_name(), Some("hr"));
899        assert_eq!(dom.node(ids[2]).tag_name(), Some("img"));
900    }
901
902    #[test]
903    fn case_insensitive_tag_names() {
904        let (dom, ids) = parse_str("<DIV></div>");
905        assert_eq!(dom.node(ids[0]).tag_name(), Some("div"));
906    }
907
908    // ── Nested elements ──────────────────────────────────────────────
909
910    #[test]
911    fn nested_elements() {
912        let (dom, ids) = parse_str("<div><span></span></div>");
913        let outer = ids[0];
914        assert_eq!(dom.node(outer).child_nodes().count(), 1);
915        let inner = dom.node(outer).first_child().unwrap().id();
916        assert_eq!(dom.node(inner).tag_name(), Some("span"));
917    }
918
919    #[test]
920    fn deeply_nested() {
921        let (dom, ids) = parse_str("<a><b><c><d></d></c></b></a>");
922        let mut cur = ids[0];
923        for tag in &["a", "b", "c", "d"] {
924            assert_eq!(dom.node(cur).tag_name(), Some(*tag));
925            cur = dom.node(cur).first_child().map(|n| n.id()).unwrap_or(cur);
926        }
927    }
928
929    // ── Text content ─────────────────────────────────────────────────
930
931    #[test]
932    fn text_node() {
933        let (dom, ids) = parse_str("<div>hello</div>");
934        let child = dom.node(ids[0]).first_child().unwrap();
935        assert_eq!(child.node_value(), Some("hello"));
936    }
937
938    #[test]
939    fn mixed_content() {
940        let (dom, ids) = parse_str("<div>before <b>mid</b> after</div>");
941        let div = ids[0];
942        let children: Vec<_> = dom.node(div).child_nodes().collect();
943        assert_eq!(children.len(), 3);
944        assert_eq!(children[0].node_value(), Some("before "));
945        assert_eq!(children[1].tag_name(), Some("b"));
946        assert_eq!(children[2].node_value(), Some(" after"));
947    }
948
949    #[test]
950    fn text_at_top_level() {
951        let (dom, ids) = parse_str("hello <span>world</span>");
952        assert_eq!(ids.len(), 2);
953        let root = dom.root();
954        let first = dom.node(root).first_child().unwrap();
955        assert_eq!(first.node_value(), Some("hello "));
956    }
957
958    // ── Attributes ──────────────────────────────────────────────────
959
960    #[test]
961    fn double_quoted_attr() {
962        let (dom, ids) = parse_str(r#"<div id="main"></div>"#);
963        assert_eq!(dom.node(ids[0]).get_attribute("id"), Some("main"));
964    }
965
966    #[test]
967    fn single_quoted_attr() {
968        let (dom, ids) = parse_str("<div id='main'></div>");
969        assert_eq!(dom.node(ids[0]).get_attribute("id"), Some("main"));
970    }
971
972    #[test]
973    fn unquoted_attr() {
974        let (dom, ids) = parse_str("<div id=main></div>");
975        assert_eq!(dom.node(ids[0]).get_attribute("id"), Some("main"));
976    }
977
978    #[test]
979    fn boolean_attr() {
980        let (dom, ids) = parse_str("<input disabled>");
981        assert_eq!(dom.node(ids[0]).get_attribute("disabled"), Some(""));
982        assert!(dom.node(ids[0]).has_attribute("disabled"));
983    }
984
985    #[test]
986    fn multiple_attrs() {
987        let (dom, ids) = parse_str(r#"<div id="x" role="banner" data-n="5"></div>"#);
988        let n = dom.node(ids[0]);
989        assert_eq!(n.get_attribute("id"), Some("x"));
990        assert_eq!(n.get_attribute("role"), Some("banner"));
991        assert_eq!(n.get_attribute("data-n"), Some("5"));
992    }
993
994    #[test]
995    fn class_attr_populates_classlist() {
996        let (dom, ids) = parse_str(r#"<div class="a b c"></div>"#);
997        let n = dom.node(ids[0]);
998        assert!(n.has_class("a"));
999        assert!(n.has_class("b"));
1000        assert!(n.has_class("c"));
1001    }
1002
1003    #[test]
1004    fn attr_name_case_preserved() {
1005        // Unlike tag names, we preserve attribute name case.
1006        let (dom, ids) = parse_str(r#"<div dataFoo="bar"></div>"#);
1007        assert_eq!(dom.node(ids[0]).get_attribute("dataFoo"), Some("bar"));
1008    }
1009
1010    #[test]
1011    fn whitespace_around_attrs() {
1012        let (dom, ids) = parse_str("<div  id=main  role=banner  ></div>");
1013        assert_eq!(dom.node(ids[0]).get_attribute("id"), Some("main"));
1014        assert_eq!(dom.node(ids[0]).get_attribute("role"), Some("banner"));
1015    }
1016
1017    #[test]
1018    fn attr_name_with_hyphens_and_colons() {
1019        let (dom, ids) = parse_str(r#"<div data-x="1" aria:label="y"></div>"#);
1020        assert_eq!(dom.node(ids[0]).get_attribute("data-x"), Some("1"));
1021        assert_eq!(dom.node(ids[0]).get_attribute("aria:label"), Some("y"));
1022    }
1023
1024    // ── Entities ─────────────────────────────────────────────────────
1025
1026    #[test]
1027    fn entity_amp() {
1028        let (dom, ids) = parse_str("<div>a &amp; b</div>");
1029        let child = dom.node(ids[0]).first_child().unwrap();
1030        assert_eq!(child.node_value(), Some("a & b"));
1031    }
1032
1033    #[test]
1034    fn entity_lt_gt_quot_apos() {
1035        let (dom, ids) = parse_str("<div>&lt;tag&gt; &quot;q&quot; &apos;a&apos;</div>");
1036        let child = dom.node(ids[0]).first_child().unwrap();
1037        assert_eq!(child.node_value(), Some("<tag> \"q\" 'a'"));
1038    }
1039
1040    #[test]
1041    fn entity_decimal_numeric() {
1042        let (dom, ids) = parse_str("<div>&#65;&#66;</div>");
1043        let child = dom.node(ids[0]).first_child().unwrap();
1044        assert_eq!(child.node_value(), Some("AB"));
1045    }
1046
1047    #[test]
1048    fn entity_hex_numeric() {
1049        let (dom, ids) = parse_str("<div>&#x41;&#X42;</div>");
1050        let child = dom.node(ids[0]).first_child().unwrap();
1051        assert_eq!(child.node_value(), Some("AB"));
1052    }
1053
1054    #[test]
1055    fn entity_in_attr_value() {
1056        let (dom, ids) = parse_str(r#"<div title="a &amp; b"></div>"#);
1057        assert_eq!(dom.node(ids[0]).get_attribute("title"), Some("a & b"));
1058    }
1059
1060    #[test]
1061    fn unknown_entity_preserved_as_literal_amp() {
1062        // `&unknown;` → '&' literal + "unknown;" as text
1063        let (dom, ids) = parse_str("<div>&xyz;</div>");
1064        let child = dom.node(ids[0]).first_child().unwrap();
1065        // We emit '&' and leave the rest to parse as text.
1066        assert_eq!(child.node_value(), Some("&xyz;"));
1067    }
1068
1069    #[test]
1070    fn entity_nbsp() {
1071        let (dom, ids) = parse_str("<div>a&nbsp;b</div>");
1072        let child = dom.node(ids[0]).first_child().unwrap();
1073        assert_eq!(child.node_value(), Some("a\u{A0}b"));
1074    }
1075
1076    // ── Comments ─────────────────────────────────────────────────────
1077
1078    #[test]
1079    fn comment_preserved() {
1080        let (dom, ids) = parse_str("<!-- hello -->");
1081        assert_eq!(ids.len(), 1);
1082        let c = dom.node(ids[0]);
1083        assert_eq!(c.node_type(), rdom_core::NodeType::Comment);
1084        assert_eq!(c.data(), Some(" hello "));
1085    }
1086
1087    #[test]
1088    fn comment_inside_element() {
1089        let (dom, ids) = parse_str("<div><!-- note -->body</div>");
1090        let div = ids[0];
1091        let children: Vec<_> = dom.node(div).child_nodes().collect();
1092        assert_eq!(children.len(), 2);
1093        assert_eq!(children[0].node_type(), rdom_core::NodeType::Comment);
1094        assert_eq!(children[1].node_value(), Some("body"));
1095    }
1096
1097    // ── Errors ───────────────────────────────────────────────────────
1098
1099    #[test]
1100    fn error_mismatched_tags() {
1101        let err = parse::<()>("<div></span>").unwrap_err();
1102        assert!(err.msg.contains("mismatched"));
1103    }
1104
1105    #[test]
1106    fn error_missing_close() {
1107        let err = parse::<()>("<div>").unwrap_err();
1108        assert!(err.msg.contains("missing closing"));
1109    }
1110
1111    #[test]
1112    fn error_unterminated_comment() {
1113        let err = parse::<()>("<!-- never ends").unwrap_err();
1114        assert!(err.msg.contains("unterminated"));
1115    }
1116
1117    #[test]
1118    fn error_unterminated_attr_value() {
1119        let err = parse::<()>(r#"<div id="abc>"#).unwrap_err();
1120        assert!(err.msg.contains("unterminated"));
1121    }
1122
1123    #[test]
1124    fn error_position_reported() {
1125        let err = parse::<()>("<div>\n<span></p>\n</div>").unwrap_err();
1126        // Mismatched </p> is on line 2.
1127        assert_eq!(err.line, 2);
1128    }
1129
1130    #[test]
1131    fn error_has_hint() {
1132        let err = parse::<()>("<div>").unwrap_err();
1133        assert!(err.hint.is_some());
1134    }
1135
1136    // ── parse_into API ───────────────────────────────────────────────
1137
1138    #[test]
1139    fn parse_into_appends_to_mount() {
1140        let mut dom: Dom<()> = Dom::new();
1141        let mount = dom.create_element("body");
1142        let root = dom.root();
1143        dom.append_child(root, mount).unwrap();
1144
1145        let ids = parse_into(&mut dom, "<h1>Title</h1><p>Body</p>", mount).unwrap();
1146        assert_eq!(ids.len(), 2);
1147        assert_eq!(dom.node(mount).child_nodes().count(), 2);
1148    }
1149
1150    // ── Complex templates ────────────────────────────────────────────
1151
1152    #[test]
1153    fn realistic_template() {
1154        let t = r#"
1155            <div class="card" id="hero">
1156              <h1>Welcome</h1>
1157              <p>Hello &amp; welcome to <strong>rdom</strong>.</p>
1158              <br/>
1159              <!-- TODO: add icon -->
1160              <button disabled>OK</button>
1161            </div>
1162        "#;
1163        let (dom, ids) = parse::<()>(t).unwrap();
1164        // Top-level: the outer div (plus potentially whitespace-only
1165        // text around it — we preserve all whitespace).
1166        let div_id = ids
1167            .iter()
1168            .find(|&&id| dom.node(id).tag_name() == Some("div"))
1169            .copied()
1170            .unwrap();
1171        let div = dom.node(div_id);
1172        assert!(div.has_class("card"));
1173        assert_eq!(div.get_attribute("id"), Some("hero"));
1174
1175        // Find <h1> inside.
1176        let h1 = div
1177            .child_nodes()
1178            .find(|c| c.tag_name() == Some("h1"))
1179            .unwrap();
1180        assert_eq!(
1181            dom.node(h1.id()).first_child().unwrap().node_value(),
1182            Some("Welcome")
1183        );
1184
1185        // The <button disabled> element.
1186        let btn = div
1187            .child_nodes()
1188            .find(|c| c.tag_name() == Some("button"))
1189            .unwrap();
1190        assert!(dom.node(btn.id()).has_attribute("disabled"));
1191    }
1192
1193    // ── Round-trip ───────────────────────────────────────────────────
1194
1195    #[test]
1196    fn round_trip_simple() {
1197        let src = "<div><span>hi</span></div>";
1198        let (dom, ids) = parse::<()>(src).unwrap();
1199        let out = dom.outer_markup(ids[0]);
1200        assert_eq!(out, src);
1201    }
1202
1203    #[test]
1204    fn round_trip_with_attrs() {
1205        let src = r#"<div data-x="1" id="main"><p></p></div>"#;
1206        let (dom, ids) = parse::<()>(src).unwrap();
1207        let out = dom.outer_markup(ids[0]);
1208        // Attributes sort alphabetically in outer_markup, matching input order.
1209        assert_eq!(out, src);
1210    }
1211
1212    #[test]
1213    fn round_trip_void_element() {
1214        let src = "<hr/>";
1215        let (dom, ids) = parse::<()>(src).unwrap();
1216        let out = dom.outer_markup(ids[0]);
1217        assert_eq!(out, "<hr/>");
1218    }
1219
1220    #[test]
1221    fn round_trip_entities_escaped() {
1222        let src = "<div>a &amp; b &lt;c&gt;</div>";
1223        let (dom, ids) = parse::<()>(src).unwrap();
1224        let out = dom.outer_markup(ids[0]);
1225        assert_eq!(out, src);
1226    }
1227
1228    // ── Whitespace preservation ──────────────────────────────────────
1229
1230    #[test]
1231    fn whitespace_preserved_in_text() {
1232        let (dom, ids) = parse_str("<p>  hello   world  </p>");
1233        let child = dom.node(ids[0]).first_child().unwrap();
1234        assert_eq!(child.node_value(), Some("  hello   world  "));
1235    }
1236
1237    #[test]
1238    fn newlines_preserved() {
1239        let (dom, ids) = parse_str("<pre>line1\nline2</pre>");
1240        let child = dom.node(ids[0]).first_child().unwrap();
1241        assert_eq!(child.node_value(), Some("line1\nline2"));
1242    }
1243
1244    // ── Many children ────────────────────────────────────────────────
1245
1246    #[test]
1247    fn many_children() {
1248        let src: String = (0..50).map(|_| "<li>x</li>").collect();
1249        let (dom, ids) = parse::<()>(&format!("<ul>{}</ul>", src)).unwrap();
1250        let ul = ids[0];
1251        assert_eq!(dom.node(ul).child_element_count(), 50);
1252    }
1253
1254    // ── Empty template ───────────────────────────────────────────────
1255
1256    #[test]
1257    fn empty_template() {
1258        let (_, ids) = parse_str("");
1259        assert!(ids.is_empty());
1260    }
1261
1262    #[test]
1263    fn whitespace_only_template() {
1264        let (dom, ids) = parse_str("   \n  ");
1265        // A single text node containing the whitespace.
1266        assert_eq!(ids.len(), 1);
1267        let c = dom.node(ids[0]);
1268        assert_eq!(c.node_type(), rdom_core::NodeType::Text);
1269    }
1270
1271    // ── Tag name chars ───────────────────────────────────────────────
1272
1273    #[test]
1274    fn hyphenated_tag() {
1275        let (dom, ids) = parse_str("<tree-item></tree-item>");
1276        assert_eq!(dom.node(ids[0]).tag_name(), Some("tree-item"));
1277    }
1278
1279    #[test]
1280    fn underscore_tag() {
1281        let (dom, ids) = parse_str("<my_element></my_element>");
1282        assert_eq!(dom.node(ids[0]).tag_name(), Some("my_element"));
1283    }
1284
1285    // ── Siblings + lack of whitespace ────────────────────────────────
1286
1287    #[test]
1288    fn adjacent_elements() {
1289        let (dom, ids) = parse_str("<a></a><b></b>");
1290        assert_eq!(ids.len(), 2);
1291        assert_eq!(dom.node(ids[0]).tag_name(), Some("a"));
1292        assert_eq!(dom.node(ids[1]).tag_name(), Some("b"));
1293    }
1294}