Skip to main content

socketry_markdown/mdast/
headings.rs

1// Released under the MIT License.
2// Copyright, 2026, by Samuel Williams.
3
4//! Helpers for extracting headings from an mdast tree.
5use super::Node;
6use alloc::{
7    collections::{BTreeMap, BTreeSet},
8    string::{String, ToString},
9    vec::Vec,
10};
11
12/// Options for extracting headings from a document.
13#[derive(Clone, Copy, Debug, Eq, PartialEq)]
14pub struct HeadingOptions {
15    /// The shallowest heading level to include (between `1` and `6`).
16    pub min_level: u8,
17    /// The deepest heading level to include (between `1` and `6`).
18    pub max_level: u8,
19}
20
21impl Default for HeadingOptions {
22    fn default() -> Self {
23        Self {
24            min_level: 1,
25            max_level: 6,
26        }
27    }
28}
29
30/// A heading extracted from an mdast tree.
31#[derive(Clone, Debug, Eq, PartialEq)]
32pub struct HeadingEntry<'a> {
33    /// The original heading node.
34    pub node: &'a Node,
35    /// The heading level, between `1` and `6`.
36    pub level: u8,
37    /// The heading's text content.
38    pub text: String,
39    /// A unique, lowercase anchor derived from the heading text.
40    pub anchor: String,
41}
42
43/// Headings extracted from a Markdown document.
44///
45/// This helper is useful for building a table of contents. It walks the tree
46/// in document order, converts each heading to plain text, and assigns unique
47/// anchors to repeated headings. Anchor assignment includes headings filtered
48/// out by [`HeadingOptions`], so the remaining anchors still match rendered
49/// headings when [`CompileOptions::heading_ids`][crate::CompileOptions::heading_ids]
50/// is enabled.
51#[derive(Clone, Debug, Default, Eq, PartialEq)]
52pub struct Headings<'a> {
53    entries: Vec<HeadingEntry<'a>>,
54}
55
56impl<'a> Headings<'a> {
57    /// Extract all headings from a document tree.
58    #[must_use]
59    pub fn extract(root: &'a Node) -> Self {
60        Self::extract_with_options(root, &HeadingOptions::default())
61    }
62
63    /// Extract headings from a document tree with a level range.
64    #[must_use]
65    pub fn extract_with_options(root: &'a Node, options: &HeadingOptions) -> Self {
66        let mut result = Self::default();
67        let mut anchors = AnchorGenerator::default();
68
69        root.walk(|node| {
70            if let Node::Heading(heading) = node {
71                let text = chomp_line_ending(&node.text_content()).to_string();
72                let anchor = anchors.anchor_for(&text);
73
74                if heading.depth >= options.min_level && heading.depth <= options.max_level {
75                    result.entries.push(HeadingEntry {
76                        node,
77                        level: heading.depth,
78                        text,
79                        anchor,
80                    });
81                }
82            }
83        });
84
85        result
86    }
87
88    /// Return the extracted entries as a slice.
89    #[must_use]
90    pub fn as_slice(&self) -> &[HeadingEntry<'a>] {
91        &self.entries
92    }
93
94    /// Iterate over the extracted entries in document order.
95    pub fn iter(&self) -> core::slice::Iter<'_, HeadingEntry<'a>> {
96        self.entries.iter()
97    }
98
99    /// Return the number of extracted headings.
100    #[must_use]
101    pub fn len(&self) -> usize {
102        self.entries.len()
103    }
104
105    /// Whether the document contains no headings in the selected level range.
106    #[must_use]
107    pub fn is_empty(&self) -> bool {
108        self.entries.is_empty()
109    }
110}
111
112impl<'a, 'b> IntoIterator for &'b Headings<'a> {
113    type Item = &'b HeadingEntry<'a>;
114    type IntoIter = core::slice::Iter<'b, HeadingEntry<'a>>;
115
116    fn into_iter(self) -> Self::IntoIter {
117        self.entries.iter()
118    }
119}
120
121impl<'a> IntoIterator for Headings<'a> {
122    type Item = HeadingEntry<'a>;
123    type IntoIter = alloc::vec::IntoIter<HeadingEntry<'a>>;
124
125    fn into_iter(self) -> Self::IntoIter {
126        self.entries.into_iter()
127    }
128}
129
130#[derive(Default)]
131struct AnchorGenerator {
132    next_suffix: BTreeMap<String, usize>,
133    used: BTreeSet<String>,
134}
135
136impl AnchorGenerator {
137    fn anchor_for(&mut self, text: &str) -> String {
138        let base = slug(text);
139        let mut suffix = self.next_suffix.get(&base).copied().unwrap_or(2);
140        let mut anchor = base.clone();
141
142        while self.used.contains(&anchor) {
143            anchor = alloc::format!("{base}-{suffix}");
144            suffix += 1;
145        }
146
147        self.next_suffix.insert(base, suffix);
148        self.used.insert(anchor.clone());
149        anchor
150    }
151}
152
153fn slug(text: &str) -> String {
154    let mut result = String::new();
155    let mut previous_was_whitespace = false;
156    let lowercase = text.to_lowercase();
157
158    for character in lowercase.chars() {
159        if character.is_whitespace() {
160            if !previous_was_whitespace {
161                result.push('-');
162            }
163            previous_was_whitespace = true;
164        } else {
165            result.push(character);
166            previous_was_whitespace = false;
167        }
168    }
169
170    result
171}
172
173fn chomp_line_ending(text: &str) -> &str {
174    if let Some(text) = text.strip_suffix("\r\n") {
175        text
176    } else if let Some(text) = text.strip_suffix('\n') {
177        text
178    } else if let Some(text) = text.strip_suffix('\r') {
179        text
180    } else {
181        text
182    }
183}
184
185#[cfg(test)]
186mod tests {
187    use super::{chomp_line_ending, slug, HeadingOptions, Headings};
188    use crate::mdast::{Heading, Node, Root, Text};
189    use alloc::{string::String, vec, vec::Vec};
190
191    fn heading(depth: u8, text: &str) -> Node {
192        Node::Heading(Heading {
193            position: None,
194            depth,
195            children: vec![Node::Text(Text {
196                value: String::from(text),
197                position: None,
198            })],
199        })
200    }
201
202    fn root(children: Vec<Node>) -> Node {
203        Node::Root(Root {
204            position: None,
205            children,
206        })
207    }
208
209    #[test]
210    fn extracts_headings_with_filtered_unique_anchors() {
211        let document = root(vec![
212            heading(1, "Title"),
213            heading(2, "Title"),
214            heading(3, "Other"),
215            heading(6, "Title"),
216        ]);
217        let headings = Headings::extract_with_options(
218            &document,
219            &HeadingOptions {
220                min_level: 2,
221                max_level: 3,
222            },
223        );
224
225        assert_eq!(headings.len(), 2);
226        assert!(!headings.is_empty());
227        assert_eq!(headings.as_slice()[0].text, "Title");
228        assert_eq!(headings.as_slice()[0].anchor, "title-2");
229        assert_eq!(headings.as_slice()[1].text, "Other");
230        assert_eq!(headings.as_slice()[1].anchor, "other");
231        assert_eq!(headings.iter().count(), 2);
232        assert_eq!((&headings).into_iter().count(), 2);
233        assert_eq!(
234            headings
235                .into_iter()
236                .map(|entry| entry.anchor)
237                .collect::<Vec<_>>(),
238            ["title-2", "other"]
239        );
240
241        let empty = Headings::extract_with_options(
242            &document,
243            &HeadingOptions {
244                min_level: 4,
245                max_level: 3,
246            },
247        );
248        assert!(empty.is_empty());
249        assert_eq!(empty.len(), 0);
250        assert_eq!(empty.as_slice(), &[]);
251    }
252
253    #[test]
254    fn extracts_all_headings_and_resolves_explicit_slug_collisions() {
255        let document = root(vec![
256            heading(1, "Heading"),
257            heading(2, "Heading"),
258            heading(3, "Heading-2"),
259            heading(4, "Heading"),
260        ]);
261        let headings = Headings::extract(&document);
262
263        assert_eq!(headings.len(), 4);
264        assert_eq!(
265            headings
266                .iter()
267                .map(|entry| entry.anchor.as_str())
268                .collect::<Vec<_>>(),
269            ["heading", "heading-2", "heading-2-2", "heading-3"]
270        );
271        assert_eq!(
272            headings.iter().map(|entry| entry.level).collect::<Vec<_>>(),
273            [1, 2, 3, 4]
274        );
275        assert_eq!(
276            headings
277                .iter()
278                .map(|entry| entry.node)
279                .collect::<Vec<_>>()
280                .len(),
281            4
282        );
283    }
284
285    #[test]
286    fn normalizes_slug_whitespace_and_chomps_line_endings() {
287        assert_eq!(slug("  Crème\r\nTea \t"), "-crème-tea-");
288        assert_eq!(slug(""), "");
289        assert_eq!(chomp_line_ending("line\r\n"), "line");
290        assert_eq!(chomp_line_ending("line\n"), "line");
291        assert_eq!(chomp_line_ending("line\r"), "line");
292        assert_eq!(chomp_line_ending("line"), "line");
293    }
294}