Skip to main content

docling_core/
markdown.rs

1//! Markdown serializer for [`DoclingDocument`].
2
3use crate::document::{DoclingDocument, Node, Table};
4
5/// How pictures are rendered (mirrors docling-core's `ImageRefMode`).
6#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
7pub enum ImageMode {
8    /// `<!-- image -->` (docling's default, and the only mode without image data).
9    #[default]
10    Placeholder,
11    /// `![Image](data:<mime>;base64,…)` — self-contained.
12    Embedded,
13    /// `![Image](<artifacts>/image_NNNNNN.<ext>)`; the bytes are returned for the
14    /// caller to write.
15    Referenced,
16}
17
18/// Serializer state threaded through the render walk.
19struct Ctx {
20    strict: bool,
21    /// Emit compact `| a | b |` tables instead of the padded GitHub serializer.
22    compact_tables: bool,
23    images: ImageMode,
24    artifacts_dir: String,
25    /// (relative path, bytes) for each referenced image — written by the caller.
26    artifacts: Vec<(String, Vec<u8>)>,
27    pic_index: usize,
28}
29
30/// Render a document to a Markdown string (pictures as placeholders).
31///
32/// `strict` selects the serializer-level behaviours that differ between
33/// docling-legacy output and cleaner Markdown — currently the code-fence
34/// language (legacy drops it, strict keeps it).
35pub fn to_markdown(doc: &DoclingDocument, strict: bool) -> String {
36    to_markdown_images(doc, strict, ImageMode::Placeholder, "artifacts").0
37}
38
39/// Render to Markdown with an explicit picture [`ImageMode`]. Returns the
40/// Markdown and, for [`ImageMode::Referenced`], the `(path, bytes)` of each image
41/// the caller should write (relative to the Markdown file).
42pub fn to_markdown_images(
43    doc: &DoclingDocument,
44    strict: bool,
45    images: ImageMode,
46    artifacts_dir: &str,
47) -> (String, Vec<(String, Vec<u8>)>) {
48    let mut ctx = Ctx {
49        strict,
50        compact_tables: doc.compact_tables,
51        images,
52        artifacts_dir: artifacts_dir.to_string(),
53        artifacts: Vec::new(),
54        pic_index: 0,
55    };
56    let mut blocks: Vec<String> = Vec::new();
57    render(&doc.nodes, &mut blocks, &mut ctx);
58    let mut body = blocks.join("\n\n");
59    // Strict mode only: turn recovered source hyperlinks into Markdown links.
60    // docling's standard pipeline drops them, so doing this in legacy mode would
61    // diverge from docling — hence strict-only, leaving conformance output intact.
62    if strict && !doc.links.is_empty() {
63        body = apply_links(&body, &doc.links);
64    }
65    let md = if body.is_empty() {
66        String::new()
67    } else {
68        format!("{body}\n")
69    };
70    (md, ctx.artifacts)
71}
72
73/// Wrap each recovered link's anchor text in Markdown `[anchor](href)`. Anchors
74/// arrive cleaned (curly quotes/dashes already normalized) but un-escaped, so we
75/// match against the body's HTML-escaped (`&`/`<`/`>`) form, the way prose nodes
76/// were serialized. Links are consumed in document order from a moving cursor, so
77/// a repeated anchor (e.g. two "issues") links its successive occurrences rather
78/// than all pointing at the first. An anchor that can't be located is skipped
79/// (its text may have been split across a line wrap or table cell).
80fn apply_links(body: &str, links: &[(String, String)]) -> String {
81    let mut out = body.to_string();
82    let mut cursor = 0usize;
83    for (anchor, href) in links {
84        let anchor = anchor
85            .replace('&', "&amp;")
86            .replace('<', "&lt;")
87            .replace('>', "&gt;");
88        if anchor.is_empty() {
89            continue;
90        }
91        if let Some(rel) = out[cursor..].find(&anchor) {
92            let at = cursor + rel;
93            // Don't relink inside an already-emitted `](` Markdown link target.
94            let replacement = format!("[{anchor}]({href})");
95            out.replace_range(at..at + anchor.len(), &replacement);
96            cursor = at + replacement.len();
97        }
98    }
99    out
100}
101
102/// Like [`apply_links`] but over a single chunk, consuming from a shared queue so
103/// the same `[anchor](href)` rewriting can be applied incrementally as Markdown is
104/// streamed out. Each queued link is matched (in document order) against `chunk`
105/// and rewritten in place; a link whose anchor is not in this chunk is carried
106/// forward in the queue for a later chunk. Anchors are recovered in document
107/// order and a chunk is always a contiguous run of whole blocks, so this
108/// reproduces [`apply_links`]' single moving cursor: the link lands in whichever
109/// chunk contains its anchor, identically to the buffered path. (A link whose
110/// anchor never appears is carried to the end and dropped — the same no-op
111/// `apply_links` performs for an unlocatable anchor.)
112fn apply_links_chunk(chunk: &str, queue: &mut Vec<(String, String)>) -> String {
113    let mut out = chunk.to_string();
114    let mut cursor = 0usize;
115    let mut carried: Vec<(String, String)> = Vec::new();
116    for (anchor_raw, href) in std::mem::take(queue) {
117        let anchor = anchor_raw
118            .replace('&', "&amp;")
119            .replace('<', "&lt;")
120            .replace('>', "&gt;");
121        if anchor.is_empty() {
122            continue;
123        }
124        if let Some(rel) = out[cursor..].find(&anchor) {
125            let at = cursor + rel;
126            let replacement = format!("[{anchor}]({href})");
127            out.replace_range(at..at + anchor.len(), &replacement);
128            cursor = at + replacement.len();
129        } else {
130            // Not in this chunk; try again when its block is flushed.
131            carried.push((anchor_raw, href));
132        }
133    }
134    *queue = carried;
135    out
136}
137
138/// Incremental Markdown serializer: feed finalized, in-document-order batches of
139/// [`Node`]s and receive Markdown chunks whose concatenation is **byte-identical**
140/// to [`to_markdown_images`] over the same nodes. This is the streaming
141/// counterpart of the buffered serializer — used to emit a document's Markdown in
142/// chunks (e.g. page by page, as the parallel PDF pipeline finishes pages) instead
143/// of building the whole string up front.
144///
145/// Only [`ImageMode::Placeholder`] and [`ImageMode::Embedded`] are streamable:
146/// [`ImageMode::Referenced`] needs a side-channel for the image bytes, which only
147/// the buffered [`to_markdown_images`] provides.
148///
149/// Each [`push`](Self::push) must contain whole blocks in reading order: a caller
150/// must not split a run of list items across two pushes (the run would render as
151/// two separate lists). Finalized PDF page batches already satisfy this.
152pub struct MarkdownStreamer {
153    strict: bool,
154    images: ImageMode,
155    compact_tables: bool,
156    /// Whether any non-empty chunk has been emitted yet (drives `\n\n` joins and
157    /// the trailing newline).
158    emitted_any: bool,
159    /// Recovered links not yet placed (strict mode), consumed in document order.
160    links: Vec<(String, String)>,
161}
162
163impl MarkdownStreamer {
164    /// Create a streamer. `compact_tables` mirrors [`DoclingDocument::compact_tables`].
165    pub fn new(strict: bool, images: ImageMode, compact_tables: bool) -> Self {
166        debug_assert!(
167            images != ImageMode::Referenced,
168            "referenced image mode is not streamable; use to_markdown_images"
169        );
170        Self {
171            strict,
172            images,
173            compact_tables,
174            emitted_any: false,
175            links: Vec::new(),
176        }
177    }
178
179    /// Render one finalized batch of nodes (plus any links recovered from the same
180    /// span, in document order) into the next Markdown chunk. Returns an empty
181    /// string when the batch produces no output (e.g. empty tables/pictures), in
182    /// which case nothing should be written.
183    pub fn push(&mut self, nodes: &[Node], links: &[(String, String)]) -> String {
184        self.links.extend(links.iter().cloned());
185        let mut ctx = Ctx {
186            strict: self.strict,
187            compact_tables: self.compact_tables,
188            images: self.images,
189            // Referenced mode is rejected at construction, so the artifact sink is
190            // never touched.
191            artifacts_dir: String::new(),
192            artifacts: Vec::new(),
193            pic_index: 0,
194        };
195        let mut blocks: Vec<String> = Vec::new();
196        render(nodes, &mut blocks, &mut ctx);
197        if blocks.is_empty() {
198            return String::new();
199        }
200        let mut body = blocks.join("\n\n");
201        if self.strict && !self.links.is_empty() {
202            body = apply_links_chunk(&body, &mut self.links);
203        }
204        let chunk = if self.emitted_any {
205            format!("\n\n{body}")
206        } else {
207            body
208        };
209        self.emitted_any = true;
210        chunk
211    }
212
213    /// Emit the trailing newline that finishes the document (empty if no content
214    /// was produced). Call exactly once, after the final [`push`](Self::push).
215    pub fn finish(self) -> String {
216        if self.emitted_any {
217            "\n".to_string()
218        } else {
219            String::new()
220        }
221    }
222}
223
224/// In `strict` mode, rewrite inline text for readability rather than byte-for-byte
225/// docling fidelity: undo the legacy `\_` underscore escaping, and tighten stray
226/// spaces around punctuation (`[ 37 , 36 ]` → `[37, 36]`, `( x )` → `(x)`). This
227/// cleans up both the PDF backend's glyph-split spacing and the space the legacy
228/// emphasis serialization leaves before punctuation (`*a* ,` → `*a*,`).
229/// Legacy/default output keeps docling's spacing untouched. Only inline text
230/// nodes pass through here — code blocks and table cells are left alone.
231fn strict_text(text: &str, strict: bool) -> String {
232    if !strict {
233        return text.to_string();
234    }
235    text.replace("\\_", "_")
236        .replace(" ,", ",")
237        .replace(" .", ".")
238        .replace(" ;", ";")
239        .replace(" )", ")")
240        .replace("( ", "(")
241        .replace(" ]", "]")
242        .replace("[ ", "[")
243}
244
245fn render(nodes: &[Node], blocks: &mut Vec<String>, ctx: &mut Ctx) {
246    let mut i = 0;
247    while i < nodes.len() {
248        match &nodes[i] {
249            Node::ListItem { .. } => {
250                let start = i;
251                i += 1;
252                loop {
253                    match nodes.get(i) {
254                        Some(Node::ListItem { .. }) => i += 1,
255                        // An empty paragraph between two list items is absorbed
256                        // into the run — docling keeps such a ListGroup
257                        // contiguous rather than splitting it.
258                        Some(Node::Paragraph { text })
259                            if text.is_empty()
260                                && matches!(nodes.get(i + 1), Some(Node::ListItem { .. })) =>
261                        {
262                            i += 1
263                        }
264                        _ => break,
265                    }
266                }
267                render_list_run(&nodes[start..i], blocks, ctx.strict);
268            }
269            other => {
270                render_one(other, blocks, ctx);
271                i += 1;
272            }
273        }
274    }
275}
276
277/// Render a contiguous run of list items.
278///
279/// Ordered items use their explicit `number`. A new sibling list (marked by
280/// `first_in_list`) at the same depth is separated by a blank line, matching
281/// docling-core's serializer.
282fn render_list_run(items: &[Node], blocks: &mut Vec<String>, strict: bool) {
283    let mut lines: Vec<String> = Vec::new();
284    // Per level, the previous item's (ordered, number) so we can detect a new
285    // sibling list.
286    let mut prev: Vec<Option<(bool, u64)>> = Vec::new();
287
288    for item in items {
289        let Node::ListItem {
290            ordered,
291            number,
292            first_in_list,
293            text,
294            level,
295            marker: _,
296            location: _,
297        } = item
298        else {
299            continue;
300        };
301        let level = *level as usize;
302
303        // Returning to a shallower level ends the deeper sibling lists.
304        prev.truncate(level + 1);
305        while prev.len() <= level {
306            prev.push(None);
307        }
308
309        // A new sibling list at the same depth gets a blank line: the kind flips
310        // (`<ul>`↔`<ol>`), an ordered run breaks (`1, 2` then `42`), or the
311        // backend flagged a fresh list (e.g. Markdown's bullet changing `-`→`*`).
312        if let Some((prev_ordered, prev_number)) = prev[level] {
313            let new_list = *first_in_list
314                || prev_ordered != *ordered
315                || (*ordered && *number != prev_number + 1);
316            if new_list {
317                lines.push(String::new());
318            }
319        }
320
321        let indent = "    ".repeat(level);
322        let marker = if *ordered {
323            format!("{number}.")
324        } else {
325            "-".to_string()
326        };
327        lines.push(format!("{indent}{marker} {}", strict_text(text, strict)));
328        prev[level] = Some((*ordered, *number));
329    }
330
331    blocks.push(lines.join("\n"));
332}
333
334fn render_one(node: &Node, blocks: &mut Vec<String>, ctx: &mut Ctx) {
335    match node {
336        Node::Heading { level, text } => {
337            let hashes = "#".repeat((*level).clamp(1, 6) as usize);
338            blocks.push(format!("{hashes} {}", strict_text(text, ctx.strict)));
339        }
340        // An empty body paragraph (docling's blank-line text item) contributes
341        // nothing to Markdown — only DocLang/JSON keep it.
342        Node::Paragraph { text } if text.is_empty() => {}
343        Node::Paragraph { text } => blocks.push(strict_text(text, ctx.strict)),
344        Node::Code { language, text } => {
345            // Legacy docling never emits a language on the fence; strict keeps it.
346            let lang = match language {
347                Some(l) if ctx.strict => l.as_str(),
348                _ => "",
349            };
350            blocks.push(format!("```{lang}\n{text}\n```"));
351        }
352        Node::Table(table) => {
353            let rendered = render_table(table, ctx.compact_tables);
354            if !rendered.is_empty() {
355                blocks.push(rendered);
356            }
357        }
358        Node::Picture { caption, image } => {
359            if let Some(cap) = caption {
360                if !cap.is_empty() {
361                    blocks.push(cap.clone());
362                }
363            }
364            blocks.push(picture_marker(image.as_ref(), ctx));
365        }
366        // A chart renders like a picture placeholder (its data table is
367        // DocLang-only); no image payload.
368        Node::Chart { .. } => blocks.push(picture_marker(None, ctx)),
369        // A DocLang-only node is omitted from Markdown.
370        Node::DoclangOnly(_) => {}
371        Node::Group { children, .. } => render(children, blocks, ctx),
372        Node::FieldRegion { items } => {
373            // docling renders the region container (which carries no text of its
374            // own) as a `<!-- missing-text -->` marker, then each field item the
375            // same way, followed by that item's marker/key/value as separate
376            // paragraphs.
377            blocks.push(MISSING_TEXT.to_string());
378            for item in items {
379                blocks.push(MISSING_TEXT.to_string());
380                for part in [&item.marker, &item.key, &item.value].into_iter().flatten() {
381                    blocks.push(strict_text(part, ctx.strict));
382                }
383            }
384        }
385        // A rich inline group renders exactly like a paragraph of its Markdown
386        // text — the structured runs are DocLang-only.
387        Node::InlineGroup { md_text, .. } => blocks.push(strict_text(md_text, ctx.strict)),
388        // Furniture (page headers/footers, HTML `<title>`) is excluded from
389        // Markdown by default, mirroring docling.
390        Node::Furniture(_) => {}
391        // Layout provenance is DocLang-only; render the wrapped node.
392        Node::Located { inner, .. } => render_one(inner, blocks, ctx),
393        // Page breaks are DocLang-only; docling omits them from Markdown.
394        Node::PageBreak => {}
395        // Handled by the run-merging branch in `render`.
396        Node::ListItem { .. } => unreachable!("list items are rendered in runs"),
397    }
398}
399
400/// docling's placeholder for a structural node (a field region / item) that has
401/// no text of its own.
402const MISSING_TEXT: &str = "<!-- missing-text -->";
403
404/// The Markdown for a picture under the active [`ImageMode`]; Referenced mode also
405/// records the bytes in `ctx.artifacts` for the caller to write.
406fn picture_marker(image: Option<&crate::PictureImage>, ctx: &mut Ctx) -> String {
407    match (ctx.images, image) {
408        (ImageMode::Embedded, Some(img)) => format!("![Image]({})", img.data_uri()),
409        (ImageMode::Referenced, Some(img)) => {
410            let path = format!(
411                "{}/image_{:06}.{}",
412                ctx.artifacts_dir,
413                ctx.pic_index,
414                ext_for(&img.mimetype)
415            );
416            ctx.pic_index += 1;
417            ctx.artifacts.push((path.clone(), img.data.clone()));
418            format!("![Image]({path})")
419        }
420        // Placeholder, or any mode with no extracted image.
421        _ => "<!-- image -->".to_string(),
422    }
423}
424
425fn ext_for(mimetype: &str) -> &str {
426    match mimetype {
427        "image/jpeg" => "jpg",
428        "image/gif" => "gif",
429        "image/webp" => "webp",
430        "image/bmp" => "bmp",
431        "image/tiff" => "tif",
432        _ => "png",
433    }
434}
435
436/// Render a table. `compact` selects between two serializers:
437///
438/// - **padded** (default) — docling-core's `tabulate(tablefmt="github")`: columns
439///   are padded to a fixed width (header width + a minimum padding of 2, or the
440///   widest data cell); numeric columns (every data cell parses as a number) are
441///   right-aligned, others left-aligned; separators are plain dashes of
442///   `width + 2`. Matches current published docling (DOCX/HTML conformance).
443/// - **compact** — `| a | b |` cells with single-dash `| - | - |` separators, no
444///   width padding. Matches the committed PDF groundtruth corpus, which predates
445///   the padded serializer.
446///
447/// Each cell is first escaped (`\n` → space, `|` → `&#124;`) so it can't break the
448/// table. Row 0 is the header.
449/// Whether a table cell counts as a number for column alignment, matching
450/// `tabulate`'s detection: an ordinary float/int (`f64`-parseable, covering
451/// `1e2`/`inf`/`+1.5`) **or** a thousands-separated number like `7,015`.
452fn is_number_cell(t: &str) -> bool {
453    t.parse::<f64>().is_ok() || is_thousands_number(t)
454}
455
456/// A number with comma thousands-separators, per `tabulate`'s
457/// `_float_with_thousands_separators` regex
458/// (`^(([+-]?[0-9]{1,3})(?:,([0-9]{3}))*)?(?(1)\.[0-9]*|\.[0-9]+)?$`): the
459/// integer part is 1–3 digits then any number of `,ddd` groups; the fraction is
460/// optional (and, without an integer part, must have at least one digit).
461fn is_thousands_number(t: &str) -> bool {
462    let b = t.as_bytes();
463    let mut i = 0;
464    let start = i;
465    if i < b.len() && (b[i] == b'+' || b[i] == b'-') {
466        i += 1;
467    }
468    // First digit chunk: 1–3 digits.
469    let d0 = i;
470    while i < b.len() && b[i].is_ascii_digit() && i - d0 < 3 {
471        i += 1;
472    }
473    let has_int = i > d0;
474    if has_int {
475        // Subsequent `,ddd` groups (exactly three digits each).
476        while i + 3 < b.len() + 1
477            && b.get(i) == Some(&b',')
478            && b.get(i + 1).is_some_and(u8::is_ascii_digit)
479            && b.get(i + 2).is_some_and(u8::is_ascii_digit)
480            && b.get(i + 3).is_some_and(u8::is_ascii_digit)
481        {
482            i += 4;
483        }
484    } else {
485        // A sign only counts with an integer part.
486        i = start;
487    }
488    // Optional fraction.
489    if i < b.len() && b[i] == b'.' {
490        i += 1;
491        let f0 = i;
492        while i < b.len() && b[i].is_ascii_digit() {
493            i += 1;
494        }
495        if !has_int && i == f0 {
496            return false; // `.` with no digits and no integer part
497        }
498    } else if !has_int {
499        return false; // neither integer nor fractional part
500    }
501    i == b.len()
502}
503
504fn render_table(table: &Table, compact: bool) -> String {
505    if table.rows.is_empty() {
506        return String::new();
507    }
508    let num_cols = table.rows.iter().map(Vec::len).max().unwrap_or(0);
509    if num_cols == 0 {
510        return String::new();
511    }
512
513    // Escaped, rectangular grid (ragged rows padded with empty cells). `tabulate`
514    // strips data cells of surrounding whitespace but leaves the header row as-is.
515    let grid: Vec<Vec<String>> = table
516        .rows
517        .iter()
518        .enumerate()
519        .map(|(r, row)| {
520            (0..num_cols)
521                .map(|c| {
522                    let cell = escape_cell(row.get(c).map(String::as_str).unwrap_or(""));
523                    if r == 0 {
524                        cell
525                    } else {
526                        cell.trim().to_string()
527                    }
528                })
529                .collect()
530        })
531        .collect();
532
533    if compact {
534        // Compact: cells joined by " | ", no padding, single-dash separators.
535        let render_row = |r: usize| -> String { format!("| {} |", grid[r].join(" | ")) };
536        let mut lines = Vec::with_capacity(grid.len() + 1);
537        lines.push(render_row(0));
538        let sep: Vec<&str> = (0..num_cols).map(|_| "-").collect();
539        lines.push(format!("| {} |", sep.join(" | ")));
540        for r in 1..grid.len() {
541            lines.push(render_row(r));
542        }
543        return lines.join("\n");
544    }
545
546    // Display width (Unicode scalar count — good enough for now).
547    let dw = |s: &str| s.chars().count();
548    let data_rows = 1..grid.len();
549
550    // A column is right-aligned when at least one data cell is numeric and every
551    // non-empty data cell is numeric — matching `tabulate`'s column typing, where
552    // empty cells are "missing" (ignored) and a number may carry thousands
553    // separators (`7,015`), which a plain `f64` parse rejects.
554    let right: Vec<bool> = (0..num_cols)
555        .map(|c| {
556            let mut any = false;
557            for r in data_rows.clone() {
558                let t = grid[r][c].trim();
559                if t.is_empty() {
560                    continue;
561                }
562                if !is_number_cell(t) {
563                    return false;
564                }
565                any = true;
566            }
567            any
568        })
569        .collect();
570
571    // Column width = max(header_width + MIN_PADDING(2), max data-cell width).
572    let width: Vec<usize> = (0..num_cols)
573        .map(|c| {
574            let mut w = dw(&grid[0][c]) + 2;
575            for r in data_rows.clone() {
576                w = w.max(dw(&grid[r][c]));
577            }
578            w
579        })
580        .collect();
581
582    let fmt_cell = |s: &str, c: usize| -> String {
583        let pad = " ".repeat(width[c].saturating_sub(dw(s)));
584        let body = if right[c] {
585            format!("{pad}{s}")
586        } else {
587            format!("{s}{pad}")
588        };
589        format!(" {body} ")
590    };
591    let render_row = |r: usize| -> String {
592        let cells: Vec<String> = (0..num_cols).map(|c| fmt_cell(&grid[r][c], c)).collect();
593        format!("|{}|", cells.join("|"))
594    };
595
596    let mut lines = Vec::with_capacity(grid.len() + 1);
597    lines.push(render_row(0));
598    let sep: Vec<String> = (0..num_cols).map(|c| "-".repeat(width[c] + 2)).collect();
599    lines.push(format!("|{}|", sep.join("|")));
600    for r in data_rows {
601        lines.push(render_row(r));
602    }
603    lines.join("\n")
604}
605
606/// Escape a table cell so it can't break the markdown table: newlines become
607/// spaces and pipes become the `&#124;` HTML entity (matches docling-core).
608fn escape_cell(s: &str) -> String {
609    s.replace('\n', " ").replace('|', "&#124;")
610}
611
612#[cfg(test)]
613mod tests {
614    use super::*;
615
616    #[test]
617    fn renders_headings_paragraphs_and_lists() {
618        let mut doc = DoclingDocument::new("demo");
619        doc.add_heading(1, "Title");
620        doc.add_paragraph("Hello world.");
621        doc.push(Node::ListItem {
622            ordered: false,
623            number: 1,
624            first_in_list: true,
625            text: "first".into(),
626            level: 0,
627            marker: None,
628            location: None,
629        });
630        doc.push(Node::ListItem {
631            ordered: false,
632            number: 2,
633            first_in_list: false,
634            text: "second".into(),
635            level: 0,
636            marker: None,
637            location: None,
638        });
639        let md = doc.export_to_markdown();
640        assert_eq!(md, "# Title\n\nHello world.\n\n- first\n- second\n");
641    }
642
643    #[test]
644    fn strict_renders_recovered_links_legacy_does_not() {
645        let mut doc = DoclingDocument::new("cv");
646        doc.add_paragraph("Find me on LinkedIn or GitHub.");
647        doc.links = vec![
648            ("LinkedIn".into(), "https://www.linkedin.com/in/x/".into()),
649            ("GitHub".into(), "https://github.com/x/".into()),
650        ];
651        // Legacy/docling mode: links are left untouched (conformance preserved).
652        assert_eq!(doc.export_to_markdown(), "Find me on LinkedIn or GitHub.\n");
653        // Strict mode: anchors become Markdown links.
654        assert_eq!(
655            doc.export_to_markdown_with(true),
656            "Find me on [LinkedIn](https://www.linkedin.com/in/x/) or [GitHub](https://github.com/x/).\n"
657        );
658    }
659
660    #[test]
661    fn strict_links_match_escaped_anchor_and_consume_in_order() {
662        let mut doc = DoclingDocument::new("d");
663        // The PDF assembler HTML-escapes prose, so by serialization time the body
664        // already carries `&amp;`; the anchor is stored un-escaped. The matcher must
665        // escape the anchor to find it. Two identical anchors link in document order.
666        doc.add_paragraph("AI &amp; ML here, and issues here, then issues there.");
667        doc.links = vec![
668            ("AI & ML".into(), "https://a/".into()),
669            ("issues".into(), "https://first/".into()),
670            ("issues".into(), "https://second/".into()),
671        ];
672        assert_eq!(
673            doc.export_to_markdown_with(true),
674            "[AI &amp; ML](https://a/) here, and [issues](https://first/) here, then [issues](https://second/) there.\n"
675        );
676    }
677
678    #[test]
679    fn renders_compact_table() {
680        let mut doc = DoclingDocument::new("t");
681        // The compact form is opt-in (the PDF backend sets it); default output uses
682        // the padded GitHub serializer (covered by the regression fixtures).
683        doc.compact_tables = true;
684        doc.push(Node::Table(Table {
685            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
686            location: None,
687            structure: None,
688            cell_blocks: None,
689        }));
690        let md = doc.export_to_markdown();
691        assert_eq!(md, "| a | b |\n| - | - |\n| 1 | 2 |\n");
692    }
693
694    #[test]
695    fn renders_padded_github_table_by_default() {
696        let mut doc = DoclingDocument::new("t");
697        doc.push(Node::Table(Table {
698            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
699            location: None,
700            structure: None,
701            cell_blocks: None,
702        }));
703        let md = doc.export_to_markdown();
704        // Numeric data columns are right-aligned; columns padded to header+2.
705        assert_eq!(md, "|   a |   b |\n|-----|-----|\n|   1 |   2 |\n");
706    }
707
708    #[test]
709    fn strict_unescapes_inline_underscores_legacy_keeps_them() {
710        let mut doc = DoclingDocument::new("t");
711        doc.add_heading(1, "a\\_b");
712        doc.add_paragraph("x\\_y");
713        doc.push(Node::ListItem {
714            ordered: false,
715            number: 1,
716            first_in_list: true,
717            text: "i\\_j".into(),
718            level: 0,
719            marker: None,
720            location: None,
721        });
722        // Legacy reproduces docling's `\_` escaping byte-for-byte.
723        assert_eq!(doc.export_to_markdown(), "# a\\_b\n\nx\\_y\n\n- i\\_j\n");
724        // Strict prefers literal underscores (Rust-only readability mode).
725        assert_eq!(doc.export_to_markdown_with(true), "# a_b\n\nx_y\n\n- i_j\n");
726    }
727
728    /// Drive a document's nodes through [`MarkdownStreamer`] in the given page
729    /// splits and assert the concatenated chunks equal the buffered serializer.
730    fn assert_stream_matches(
731        doc: &DoclingDocument,
732        strict: bool,
733        images: ImageMode,
734        splits: &[usize],
735    ) {
736        let want = to_markdown_images(doc, strict, images, "artifacts").0;
737        let mut streamer = MarkdownStreamer::new(strict, images, doc.compact_tables);
738        let mut got = String::new();
739        let mut start = 0;
740        for &end in splits {
741            // Links only matter in strict mode; feed them all with the first batch
742            // that has content (document order is preserved by the queue).
743            let links = if start == 0 {
744                doc.links.as_slice()
745            } else {
746                &[]
747            };
748            got.push_str(&streamer.push(&doc.nodes[start..end], links));
749            start = end;
750        }
751        got.push_str(&streamer.push(
752            &doc.nodes[start..],
753            if start == 0 {
754                doc.links.as_slice()
755            } else {
756                &[]
757            },
758        ));
759        got.push_str(&streamer.finish());
760        assert_eq!(
761            got, want,
762            "streamed output diverged (splits={splits:?}, strict={strict})"
763        );
764    }
765
766    #[test]
767    fn streaming_is_byte_identical_to_buffered() {
768        let mut doc = DoclingDocument::new("d");
769        doc.add_heading(1, "Title");
770        doc.add_paragraph("First paragraph.");
771        doc.push(Node::ListItem {
772            ordered: false,
773            number: 1,
774            first_in_list: true,
775            text: "a".into(),
776            level: 0,
777            marker: None,
778            location: None,
779        });
780        doc.push(Node::ListItem {
781            ordered: false,
782            number: 2,
783            first_in_list: false,
784            text: "b".into(),
785            level: 0,
786            marker: None,
787            location: None,
788        });
789        doc.push(Node::Code {
790            language: Some("rust".into()),
791            text: "let x = 1;".into(),
792        });
793        doc.push(Node::Table(Table {
794            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
795            location: None,
796            structure: None,
797            cell_blocks: None,
798        }));
799        doc.push(Node::Picture {
800            caption: Some("Fig 1".into()),
801            image: None,
802        });
803        doc.add_paragraph("Last paragraph.");
804
805        // A run of list items must never straddle a split, so try splits that fall
806        // on safe block boundaries (the streaming PDF assembler guarantees this).
807        for &strict in &[false, true] {
808            for &images in &[ImageMode::Placeholder, ImageMode::Embedded] {
809                for splits in [&[][..], &[1][..], &[2][..], &[4][..], &[1, 4, 6][..]] {
810                    assert_stream_matches(&doc, strict, images, splits);
811                }
812            }
813        }
814    }
815
816    #[test]
817    fn streaming_applies_recovered_links_in_strict_mode() {
818        let mut doc = DoclingDocument::new("d");
819        doc.add_paragraph("See LinkedIn for details.");
820        doc.add_paragraph("And GitHub too.");
821        doc.links = vec![
822            ("LinkedIn".into(), "https://lnkd/".into()),
823            ("GitHub".into(), "https://gh/".into()),
824        ];
825        // The second anchor lives in the second block, so it must be carried across
826        // the page boundary and placed when that block streams out.
827        assert_stream_matches(&doc, true, ImageMode::Placeholder, &[1]);
828    }
829
830    #[test]
831    fn strict_tightens_punctuation_spacing_legacy_keeps_it() {
832        let mut doc = DoclingDocument::new("t");
833        doc.add_paragraph("see [ 37 , 36 ] and ( x ) .");
834        // Legacy keeps docling's spacing byte-for-byte.
835        assert_eq!(doc.export_to_markdown(), "see [ 37 , 36 ] and ( x ) .\n");
836        // Strict tightens punctuation for readable Markdown.
837        assert_eq!(doc.export_to_markdown_with(true), "see [37, 36] and (x).\n");
838    }
839}