Skip to main content

docling_core/
markdown.rs

1//! Markdown serializer for [`DoclingDocument`].
2
3use crate::document::{DoclingDocument, Node, Table};
4
5/// How pictures are rendered (mirrors docling-core's `ImageRefMode`).
6#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
7pub enum ImageMode {
8    /// `<!-- image -->` (docling's default, and the only mode without image data).
9    #[default]
10    Placeholder,
11    /// `![Image](data:<mime>;base64,…)` — self-contained.
12    Embedded,
13    /// `![Image](<artifacts>/image_NNNNNN.<ext>)`; the bytes are returned for the
14    /// caller to write.
15    Referenced,
16}
17
18/// Serializer state threaded through the render walk.
19struct Ctx {
20    strict: bool,
21    /// Emit compact `| a | b |` tables instead of the padded GitHub serializer.
22    compact_tables: bool,
23    images: ImageMode,
24    artifacts_dir: String,
25    /// (relative path, bytes) for each referenced image — written by the caller.
26    artifacts: Vec<(String, Vec<u8>)>,
27    pic_index: usize,
28    /// Rendering the block content of a rich table cell (docling-core 2.94's
29    /// `in_table_cell`, docling-core#540): a heading has no valid Markdown
30    /// form inside a table, so it renders as plain text without `#` markers.
31    in_table_cell: bool,
32    /// docling-core's `MarkdownParams.page_break_placeholder`: the text that
33    /// separates two pages ([`DoclingDocument::page_break_placeholder`]).
34    /// `None` omits page breaks, docling's default.
35    page_break: Option<String>,
36    /// A page boundary has been crossed since the last rendered block, so the
37    /// next block is preceded by the placeholder. docling yields its
38    /// `_PageBreakNode` between two *items* whose `prov.page_no` differ, so a
39    /// boundary before the first block or after the last one emits nothing,
40    /// and a run of empty pages collapses into a single break.
41    pending_page_break: bool,
42    /// Whether any block has been rendered yet — for a streamer, across every
43    /// earlier push too — which is what makes a boundary a *pending* break.
44    emitted_any: bool,
45}
46
47/// Render a document to a Markdown string (pictures as placeholders).
48///
49/// `strict` selects the serializer-level behaviours that differ between
50/// docling-legacy output and cleaner Markdown — currently the code-fence
51/// language (legacy drops it, strict keeps it).
52pub fn to_markdown(doc: &DoclingDocument, strict: bool) -> String {
53    to_markdown_images(doc, strict, ImageMode::Placeholder, "artifacts").0
54}
55
56/// Render to Markdown with an explicit picture [`ImageMode`]. Returns the
57/// Markdown and, for [`ImageMode::Referenced`], the `(path, bytes)` of each image
58/// the caller should write (relative to the Markdown file).
59pub fn to_markdown_images(
60    doc: &DoclingDocument,
61    strict: bool,
62    images: ImageMode,
63    artifacts_dir: &str,
64) -> (String, Vec<(String, Vec<u8>)>) {
65    let mut ctx = Ctx {
66        strict,
67        compact_tables: doc.compact_tables,
68        images,
69        artifacts_dir: artifacts_dir.to_string(),
70        artifacts: Vec::new(),
71        pic_index: 0,
72        in_table_cell: false,
73        page_break: doc.page_break_placeholder.clone(),
74        pending_page_break: false,
75        emitted_any: false,
76    };
77    let mut blocks: Vec<String> = Vec::new();
78    render(&doc.nodes, &mut blocks, &mut ctx);
79    let mut body = blocks.join("\n\n");
80    // Strict mode only: turn recovered source hyperlinks into Markdown links.
81    // docling's standard pipeline drops them, so doing this in legacy mode would
82    // diverge from docling — hence strict-only, leaving conformance output intact.
83    if strict && !doc.links.is_empty() {
84        body = apply_links(&body, &doc.links);
85    }
86    let md = if body.is_empty() {
87        String::new()
88    } else {
89        format!("{body}\n")
90    };
91    (md, ctx.artifacts)
92}
93
94/// Render the block content of a *rich table cell* to Markdown — what
95/// docling-core's table serializer does for a `RichTableCell`
96/// (`doc_serializer.serialize(item, in_table_cell=True)`): the cell's
97/// paragraphs, lists and flattened nested tables render as in a document, but a
98/// heading loses its `#` markers (docling-core#540 — the Markdown spec has no
99/// headings inside tables). Pictures stay placeholders. The caller flattens the
100/// result into its cell text; the table serializer later turns the newlines
101/// into spaces.
102pub fn to_markdown_table_cell(doc: &DoclingDocument, strict: bool) -> String {
103    let mut ctx = Ctx {
104        strict,
105        compact_tables: doc.compact_tables,
106        images: ImageMode::Placeholder,
107        artifacts_dir: String::new(),
108        artifacts: Vec::new(),
109        pic_index: 0,
110        in_table_cell: true,
111        // A rich cell is one page's content; its sub-document carries no
112        // page boundaries and docling's `_iterate_items` runs the page-break
113        // scan over the document root only.
114        page_break: None,
115        pending_page_break: false,
116        emitted_any: false,
117    };
118    let mut blocks: Vec<String> = Vec::new();
119    render(&doc.nodes, &mut blocks, &mut ctx);
120    blocks.join("\n\n")
121}
122
123/// Wrap each recovered link's anchor text in Markdown `[anchor](href)`. Anchors
124/// arrive cleaned (curly quotes/dashes already normalized) but un-escaped, so we
125/// match against the body's HTML-escaped (`&`/`<`/`>`) form, the way prose nodes
126/// were serialized. Links are consumed in document order from a moving cursor, so
127/// a repeated anchor (e.g. two "issues") links its successive occurrences rather
128/// than all pointing at the first. An anchor that can't be located is skipped
129/// (its text may have been split across a line wrap or table cell).
130fn apply_links(body: &str, links: &[(String, String)]) -> String {
131    let mut out = body.to_string();
132    let mut cursor = 0usize;
133    for (anchor, href) in links {
134        let anchor = anchor
135            .replace('&', "&amp;")
136            .replace('<', "&lt;")
137            .replace('>', "&gt;");
138        if anchor.is_empty() {
139            continue;
140        }
141        if let Some(rel) = out[cursor..].find(&anchor) {
142            let at = cursor + rel;
143            // Don't relink inside an already-emitted `](` Markdown link target.
144            let replacement = format!("[{anchor}]({href})");
145            out.replace_range(at..at + anchor.len(), &replacement);
146            cursor = at + replacement.len();
147        }
148    }
149    out
150}
151
152/// Like [`apply_links`] but over a single chunk, consuming from a shared queue so
153/// the same `[anchor](href)` rewriting can be applied incrementally as Markdown is
154/// streamed out. Each queued link is matched (in document order) against `chunk`
155/// and rewritten in place; a link whose anchor is not in this chunk is carried
156/// forward in the queue for a later chunk. Anchors are recovered in document
157/// order and a chunk is always a contiguous run of whole blocks, so this
158/// reproduces [`apply_links`]' single moving cursor: the link lands in whichever
159/// chunk contains its anchor, identically to the buffered path. (A link whose
160/// anchor never appears is carried to the end and dropped — the same no-op
161/// `apply_links` performs for an unlocatable anchor.)
162fn apply_links_chunk(chunk: &str, queue: &mut Vec<(String, String)>) -> String {
163    let mut out = chunk.to_string();
164    let mut cursor = 0usize;
165    let mut carried: Vec<(String, String)> = Vec::new();
166    for (anchor_raw, href) in std::mem::take(queue) {
167        let anchor = anchor_raw
168            .replace('&', "&amp;")
169            .replace('<', "&lt;")
170            .replace('>', "&gt;");
171        if anchor.is_empty() {
172            continue;
173        }
174        if let Some(rel) = out[cursor..].find(&anchor) {
175            let at = cursor + rel;
176            let replacement = format!("[{anchor}]({href})");
177            out.replace_range(at..at + anchor.len(), &replacement);
178            cursor = at + replacement.len();
179        } else {
180            // Not in this chunk; try again when its block is flushed.
181            carried.push((anchor_raw, href));
182        }
183    }
184    *queue = carried;
185    out
186}
187
188/// Incremental Markdown serializer: feed finalized, in-document-order batches of
189/// [`Node`]s and receive Markdown chunks whose concatenation is **byte-identical**
190/// to [`to_markdown_images`] over the same nodes. This is the streaming
191/// counterpart of the buffered serializer — used to emit a document's Markdown in
192/// chunks (e.g. page by page, as the parallel PDF pipeline finishes pages) instead
193/// of building the whole string up front.
194///
195/// [`ImageMode::Placeholder`] and [`ImageMode::Embedded`] render inline.
196/// [`ImageMode::Referenced`] additionally hands each picture's bytes out through
197/// [`take_artifacts`](Self::take_artifacts) — construct with
198/// [`with_artifacts`](Self::with_artifacts) and drain after every push so the
199/// bytes can be written to disk as pages finish instead of accumulating for the
200/// whole document (issue #80's memory-bounded image handling).
201///
202/// Each [`push`](Self::push) must contain whole blocks in reading order: a caller
203/// must not split a run of list items across two pushes (the run would render as
204/// two separate lists). Finalized PDF page batches already satisfy this.
205pub struct MarkdownStreamer {
206    strict: bool,
207    images: ImageMode,
208    compact_tables: bool,
209    /// Whether any non-empty chunk has been emitted yet (drives `\n\n` joins and
210    /// the trailing newline).
211    emitted_any: bool,
212    /// Recovered links not yet placed (strict mode), consumed in document order.
213    links: Vec<(String, String)>,
214    /// Referenced mode: the link prefix, the not-yet-drained `(path, bytes)`
215    /// artifacts, and the running image number (continues across pushes so the
216    /// stream matches the buffered serializer's `image_000000…` numbering).
217    artifacts_dir: String,
218    artifacts: Vec<(String, Vec<u8>)>,
219    pic_index: usize,
220    /// [`DoclingDocument::page_break_placeholder`] and the boundary carried
221    /// over from the previous push (a page batch opens with its page marker,
222    /// so the break it implies is paid by that batch's first block).
223    page_break: Option<String>,
224    pending_page_break: bool,
225}
226
227impl MarkdownStreamer {
228    /// Create a streamer. `compact_tables` mirrors [`DoclingDocument::compact_tables`].
229    /// For [`ImageMode::Referenced`] use [`with_artifacts`](Self::with_artifacts).
230    pub fn new(strict: bool, images: ImageMode, compact_tables: bool) -> Self {
231        debug_assert!(
232            images != ImageMode::Referenced,
233            "referenced image mode needs an artifacts dir; use with_artifacts"
234        );
235        Self::with_artifacts(strict, images, compact_tables, "artifacts")
236    }
237
238    /// Like [`new`](Self::new) but with the artifacts link prefix, allowing
239    /// [`ImageMode::Referenced`]: pictures render as
240    /// `![Image](<artifacts_dir>/image_NNNNNN.<ext>)` and each push's image
241    /// bytes wait in [`take_artifacts`](Self::take_artifacts) for the caller to
242    /// write. The concatenated chunks and the artifact list match the buffered
243    /// [`to_markdown_images`] byte-for-byte.
244    pub fn with_artifacts(
245        strict: bool,
246        images: ImageMode,
247        compact_tables: bool,
248        artifacts_dir: &str,
249    ) -> Self {
250        Self {
251            strict,
252            images,
253            compact_tables,
254            emitted_any: false,
255            links: Vec::new(),
256            artifacts_dir: artifacts_dir.to_string(),
257            artifacts: Vec::new(),
258            pic_index: 0,
259            page_break: None,
260            pending_page_break: false,
261        }
262    }
263
264    /// Insert `placeholder` between pages, mirroring
265    /// [`DoclingDocument::page_break_placeholder`] for the buffered path (the
266    /// concatenated chunks stay byte-identical to it). `None` — the default —
267    /// omits page breaks. Set before the first [`push`](Self::push).
268    pub fn with_page_break_placeholder(mut self, placeholder: Option<String>) -> Self {
269        self.page_break = placeholder;
270        self
271    }
272
273    /// The `(relative path, bytes)` of images rendered by pushes since the last
274    /// drain ([`ImageMode::Referenced`] only — empty otherwise). Paths are
275    /// relative to the Markdown file, i.e. they start with the configured
276    /// artifacts dir.
277    pub fn take_artifacts(&mut self) -> Vec<(String, Vec<u8>)> {
278        std::mem::take(&mut self.artifacts)
279    }
280
281    /// Render one finalized batch of nodes (plus any links recovered from the same
282    /// span, in document order) into the next Markdown chunk. Returns an empty
283    /// string when the batch produces no output (e.g. empty tables/pictures), in
284    /// which case nothing should be written.
285    pub fn push(&mut self, nodes: &[Node], links: &[(String, String)]) -> String {
286        self.links.extend(links.iter().cloned());
287        let mut ctx = Ctx {
288            strict: self.strict,
289            compact_tables: self.compact_tables,
290            images: self.images,
291            artifacts_dir: std::mem::take(&mut self.artifacts_dir),
292            artifacts: std::mem::take(&mut self.artifacts),
293            pic_index: self.pic_index,
294            in_table_cell: false,
295            page_break: std::mem::take(&mut self.page_break),
296            pending_page_break: self.pending_page_break,
297            emitted_any: self.emitted_any,
298        };
299        let mut blocks: Vec<String> = Vec::new();
300        render(nodes, &mut blocks, &mut ctx);
301        self.artifacts_dir = std::mem::take(&mut ctx.artifacts_dir);
302        self.artifacts = std::mem::take(&mut ctx.artifacts);
303        self.pic_index = ctx.pic_index;
304        self.page_break = std::mem::take(&mut ctx.page_break);
305        self.pending_page_break = ctx.pending_page_break;
306        if blocks.is_empty() {
307            return String::new();
308        }
309        let mut body = blocks.join("\n\n");
310        if self.strict && !self.links.is_empty() {
311            body = apply_links_chunk(&body, &mut self.links);
312        }
313        let chunk = if self.emitted_any {
314            format!("\n\n{body}")
315        } else {
316            body
317        };
318        self.emitted_any = true;
319        chunk
320    }
321
322    /// Emit the trailing newline that finishes the document (empty if no content
323    /// was produced). Call exactly once, after the final [`push`](Self::push).
324    pub fn finish(self) -> String {
325        if self.emitted_any {
326            "\n".to_string()
327        } else {
328            String::new()
329        }
330    }
331}
332
333/// In `strict` mode, rewrite inline text for readability rather than byte-for-byte
334/// docling fidelity: undo the legacy `\_` underscore escaping, and tighten stray
335/// spaces around punctuation (`[ 37 , 36 ]` → `[37, 36]`, `( x )` → `(x)`). This
336/// cleans up both the PDF backend's glyph-split spacing and the space the legacy
337/// emphasis serialization leaves before punctuation (`*a* ,` → `*a*,`).
338/// Legacy/default output keeps docling's spacing untouched. Only inline text
339/// nodes pass through here — code blocks and table cells are left alone.
340fn strict_text(text: &str, strict: bool) -> String {
341    if !strict {
342        return text.to_string();
343    }
344    text.replace("\\_", "_")
345        .replace(" ,", ",")
346        .replace(" .", ".")
347        .replace(" ;", ";")
348        .replace(" )", ")")
349        .replace("( ", "(")
350        .replace(" ]", "]")
351        .replace("[ ", "[")
352}
353
354/// docling-core 2.92's `_md_line_breaks` (docling-core#721): a single `\n`
355/// inside an item's text becomes a GFM hard line break (`"  \n"`, two trailing
356/// spaces) so renderers honour it; a blank line (`\n\n`) is a paragraph break
357/// and stays as is — the document scope already joins blocks with `\n\n`.
358/// Applied to body text, list items and captions, never to code/formulas.
359fn md_line_breaks(text: &str) -> String {
360    if !text.contains('\n') {
361        return text.to_string();
362    }
363    text.split("\n\n")
364        .map(|para| para.replace('\n', "  \n"))
365        .collect::<Vec<_>>()
366        .join("\n\n")
367}
368
369/// Undo [`md_line_breaks`] on a rich table cell's flattened Markdown so the
370/// non-Markdown exports (JSON `text`, LaTeX cells) see the cell's raw line
371/// breaks, as docling's do — a rich cell's text is its Markdown serialization
372/// in our model, and the two trailing spaces are a Markdown-only marker.
373pub(crate) fn strip_hard_breaks(text: &str) -> String {
374    if text.contains("  \n") {
375        text.replace("  \n", "\n")
376    } else {
377        text.to_string()
378    }
379}
380
381/// docling-core's `_heading_line_breaks`: a GFM heading cannot span lines, so a
382/// newline inside heading text collapses to a space (`# Hello World`, not a
383/// broken `# Hello\nWorld`).
384fn heading_line_breaks(text: &str) -> String {
385    text.replace('\n', " ")
386}
387
388fn render(nodes: &[Node], blocks: &mut Vec<String>, ctx: &mut Ctx) {
389    let mut i = 0;
390    while i < nodes.len() {
391        let before = blocks.len();
392        match &nodes[i] {
393            // A page boundary: the explicit `PageBreak` (slides, DjVu / DocTags
394            // pages) or the `PageInfo` marker that opens every PDF page and
395            // spreadsheet sheet — a sheet boundary carries both, and the flag
396            // absorbs the pair into one break. docling's `_iterate_items`
397            // yields a `_PageBreakNode` only between two items on different
398            // pages (a group's leading item counts for the group), which is
399            // exactly "a boundary between two rendered blocks": nothing before
400            // the first block, nothing after the last, consecutive boundaries
401            // — empty or furniture-only pages — collapsed into one.
402            Node::PageBreak | Node::PageInfo { .. } => {
403                if ctx.page_break.is_some() && ctx.emitted_any {
404                    ctx.pending_page_break = true;
405                }
406                i += 1;
407            }
408            Node::ListItem { .. } => {
409                let start = i;
410                i += 1;
411                loop {
412                    match nodes.get(i) {
413                        Some(Node::ListItem { .. }) => i += 1,
414                        // An empty paragraph between two list items is absorbed
415                        // into the run — docling keeps such a ListGroup
416                        // contiguous rather than splitting it.
417                        Some(Node::Paragraph { text })
418                            if text.is_empty()
419                                && matches!(nodes.get(i + 1), Some(Node::ListItem { .. })) =>
420                        {
421                            i += 1
422                        }
423                        _ => break,
424                    }
425                }
426                render_list_run(&nodes[start..i], blocks, ctx.strict);
427            }
428            other => {
429                render_one(other, blocks, ctx);
430                i += 1;
431            }
432        }
433        if blocks.len() > before {
434            if ctx.pending_page_break {
435                // The placeholder is a block of its own, joined by the document
436                // delimiter like docling's `_PageBreakSerResult` part — an
437                // empty placeholder therefore leaves the doubled `\n\n`
438                // upstream leaves too.
439                if let Some(placeholder) = &ctx.page_break {
440                    blocks.insert(before, placeholder.clone());
441                }
442                ctx.pending_page_break = false;
443            }
444            ctx.emitted_any = true;
445        }
446    }
447}
448
449/// Render a contiguous run of list items.
450///
451/// Ordered items use their explicit `number`. A new sibling list (marked by
452/// `first_in_list`) at the same depth is separated by a blank line, matching
453/// docling-core's serializer.
454fn render_list_run(items: &[Node], blocks: &mut Vec<String>, strict: bool) {
455    let mut lines: Vec<String> = Vec::new();
456    // Whether a top-level item has been rendered yet — a fresh-list flag on
457    // the very first item opens nothing.
458    let mut any_top = false;
459
460    for item in items {
461        let Node::ListItem {
462            ordered,
463            number,
464            first_in_list,
465            text,
466            level,
467            marker: orig_marker,
468            location: _,
469            dclx: _,
470            href: _,
471            layer,
472        } = item
473        else {
474            continue;
475        };
476        // A non-body (furniture) list item is omitted from Markdown, matching
477        // docling's content-layer filtering.
478        if layer.is_some() {
479            continue;
480        }
481        let level = *level as usize;
482
483        // A new sibling list at the top level gets a blank line — and only the
484        // backend knows where one starts (`first_in_list`: Word's `numId`
485        // changing, an HTML `<ul>` closing, a Markdown bullet switching
486        // `-`→`*`). The serializer used to guess it from a kind flip or a
487        // number gap as well, which split lists docling keeps whole (an
488        // AsciiDoc `1.` … `5.`, mixed `*`/`1.` markers) — #385. Only at the
489        // top level: nested sibling groups are children of a list item, and
490        // docling joins an item's children without blank lines.
491        if level == 0 {
492            if any_top && *first_in_list {
493                lines.push(String::new());
494            }
495            any_top = true;
496        }
497
498        let indent = "    ".repeat(level);
499        // docling-core's `case_already_valid`: a marker of digits and a dot
500        // prints verbatim — Python's `\d+\.` admits every Unicode decimal
501        // digit, so a DOCX `decimalFullWidth` marker (`1.`, docling#4336)
502        // is kept as it is rather than renumbered in ASCII.
503        let verbatim = orig_marker.as_deref().filter(|m| {
504            m.strip_suffix('.')
505                .is_some_and(|d| !d.is_empty() && d.chars().all(char::is_numeric))
506        });
507        let marker = match verbatim {
508            Some(m) if *ordered => m.to_string(),
509            _ if *ordered => format!("{number}."),
510            _ => "-".to_string(),
511        };
512        lines.push(format!("{indent}{marker} {}", list_item_text(text, strict)));
513    }
514
515    // A run consisting only of furniture (content-layer-filtered) items yields no
516    // lines; pushing an empty block here would surface as a stray blank line.
517    if !lines.is_empty() {
518        blocks.push(lines.join("\n"));
519    }
520}
521
522/// A list item's Markdown body. The GFM hard-line-break rule (docling-core#721)
523/// applies to the item's own text; pictures the HTML backend folded into the
524/// item (`"\n[alt\n]<!-- image -->"` per `<img>` inside the `<li>`) are
525/// docling's picture *children* of the item, which its serializer prints after
526/// the item line with plain newlines — so a folded tail keeps its newlines
527/// unmarked. The tail is recognised structurally: every line after the first is
528/// an image marker or an alt caption directly followed by one.
529fn list_item_text(text: &str, strict: bool) -> String {
530    let escaped = strict_text(text, strict);
531    if let Some((own, tail)) = escaped.split_once('\n') {
532        if is_folded_child_tail(tail) {
533            return format!("{}\n{tail}", md_line_breaks(own));
534        }
535    }
536    md_line_breaks(&escaped)
537}
538
539/// Whether everything after a list item's own first line is a folded *child*
540/// block rather than a continuation of the item's text: an image marker
541/// (optionally preceded by its caption/alt line) or a fenced code block. The
542/// AsciiDoc backend indents such a block to the item's own depth (as
543/// docling-core's list serializer does for each part it emits), so a leading
544/// indent is ignored here.
545fn is_folded_child_tail(tail: &str) -> bool {
546    const MARKER: &str = "<!-- image -->";
547    const FENCE: &str = "```";
548    let mut lines = tail.split('\n').peekable();
549    let mut any = false;
550    while let Some(line) = lines.next() {
551        let line = line.trim_start();
552        if line == MARKER {
553            any = true;
554        } else if line == FENCE {
555            // Skip the block's body; an unclosed fence is not a folded child.
556            loop {
557                match lines.next() {
558                    Some(l) if l.trim_start() == FENCE => break,
559                    Some(_) => {}
560                    None => return false,
561                }
562            }
563            any = true;
564        } else if lines.next().map(str::trim_start) == Some(MARKER) {
565            any = true; // an alt caption line, then its marker
566        } else {
567            return false;
568        }
569    }
570    any
571}
572
573fn render_one(node: &Node, blocks: &mut Vec<String>, ctx: &mut Ctx) {
574    match node {
575        Node::Heading { level, text } => {
576            let text = heading_line_breaks(&strict_text(text, ctx.strict));
577            if ctx.in_table_cell {
578                // docling-core#540: no `#` markers inside a table cell.
579                blocks.push(text);
580            } else {
581                let hashes = "#".repeat((*level).clamp(1, 6) as usize);
582                blocks.push(format!("{hashes} {text}"));
583            }
584        }
585        // An empty body paragraph (docling's blank-line text item) contributes
586        // nothing to Markdown — only DocLang/JSON keep it.
587        Node::Paragraph { text } if text.is_empty() => {}
588        Node::Paragraph { text } => blocks.push(md_line_breaks(&strict_text(text, ctx.strict))),
589        // A standalone caption item renders like a text item; its hyperlink
590        // annotation becomes a Markdown link around the whole caption.
591        Node::Caption { text, .. } if text.is_empty() => {}
592        Node::Caption { text, href } => {
593            let body = md_line_breaks(&strict_text(text, ctx.strict));
594            blocks.push(match href {
595                Some(url) => format!("[{body}]({url})"),
596                None => body,
597            });
598        }
599        Node::CheckboxItem { checked, text } => {
600            let mark = if *checked { "- [x] " } else { "- [ ] " };
601            blocks.push(md_line_breaks(&strict_text(
602                &format!("{mark}{text}"),
603                ctx.strict,
604            )));
605        }
606        Node::Code {
607            language,
608            text,
609            pretty,
610            ..
611        } => {
612            // Legacy docling never emits a language on the fence; strict keeps it.
613            let lang = match language {
614                Some(l) if ctx.strict => l.as_str(),
615                _ => "",
616            };
617            // Strict prefers the line-preserving rendering when the backend
618            // supplied one (PDF); legacy stays on docling's flat `text`.
619            let body = match pretty {
620                Some(p) if ctx.strict => p.as_str(),
621                _ => text.as_str(),
622            };
623            blocks.push(format!("```{lang}\n{body}\n```"));
624        }
625        // A CodeFormula-decoded display formula renders as docling's `$$…$$`
626        // (the un-enriched pipeline emits a placeholder paragraph instead).
627        Node::Formula { latex, .. } => blocks.push(format!("$${latex}$$")),
628        Node::Table(table) => {
629            // docling renders a table's caption as a text line before the grid.
630            // `caption` is already escaped (backend convention), like a paragraph.
631            if let Some(cap) = &table.caption {
632                if !cap.is_empty() {
633                    blocks.push(md_line_breaks(&strict_text(cap, ctx.strict)));
634                }
635            }
636            let rendered = render_table(table, ctx.compact_tables);
637            if !rendered.is_empty() {
638                blocks.push(rendered);
639            }
640        }
641        // Classification predictions don't affect docling's Markdown output.
642        Node::Picture { caption, image, .. } => {
643            if let Some(cap) = caption {
644                if !cap.is_empty() {
645                    blocks.push(md_line_breaks(cap));
646                }
647            }
648            blocks.push(picture_marker(image.as_ref(), ctx));
649        }
650        // A chart renders as docling's picture-with-meta markdown: the caption,
651        // the placeholder, the humanized classification ("line_chart" ->
652        // "Line chart"), then the chart's data grid as a regular table.
653        Node::Chart {
654            kind,
655            table,
656            caption,
657            ..
658        } => {
659            if let Some(cap) = caption {
660                if !cap.is_empty() {
661                    blocks.push(md_line_breaks(cap));
662                }
663            }
664            blocks.push(picture_marker(None, ctx));
665            blocks.push(humanize_label(kind));
666            let rendered = render_table(table, false);
667            if !rendered.is_empty() {
668                blocks.push(rendered);
669            }
670        }
671        // A DocLang-only node is omitted from Markdown.
672        Node::DoclangOnly(_) => {}
673        // A group on a non-body layer (a hidden spreadsheet sheet) renders
674        // nothing, like every other non-body item.
675        Node::Group { layer: Some(_), .. } => {}
676        Node::Group { children, .. } => render(children, blocks, ctx),
677        Node::FieldRegion { items } => {
678            // The region container and each field item carry no text of their
679            // own; docling-core 2.93 (#724) serializes them to nothing (older
680            // releases emitted a `<!-- missing-text -->` marker for each), so
681            // only an item's marker/key/value appear, as separate paragraphs.
682            for item in items {
683                for part in [&item.marker, &item.key, &item.value].into_iter().flatten() {
684                    blocks.push(md_line_breaks(&strict_text(part, ctx.strict)));
685                }
686            }
687        }
688        // A rich inline group renders exactly like a paragraph of its Markdown
689        // text — the structured runs are DocLang-only.
690        Node::InlineGroup { md_text, .. } => {
691            blocks.push(md_line_breaks(&strict_text(md_text, ctx.strict)))
692        }
693        // A plain-text backend dump renders verbatim as a single block.
694        Node::TextDump(text) => {
695            if !text.is_empty() {
696                blocks.push(text.clone());
697            }
698        }
699        // Furniture (page headers/footers, HTML `<title>`) is excluded from
700        // Markdown by default, mirroring docling.
701        Node::Furniture { .. } => {}
702        Node::PageFurniture { .. } => {}
703        // A comment lives in the notes layer — omitted like other furniture;
704        // the annotation on a body item is JSON-only, so render the item.
705        Node::CommentSection { .. } => {}
706        Node::Commented { inner, .. } => render_one(inner, blocks, ctx),
707        // Layout provenance is DocLang-only; render the wrapped node.
708        Node::Located { inner, .. } | Node::Prov { inner, .. } => render_one(inner, blocks, ctx),
709        // Page breaks are DocLang-only; docling omits them from Markdown.
710        Node::PageBreak => {}
711        // Page markers feed the JSON export only.
712        Node::PageInfo { .. } => {}
713        // Runs of adjacent list items are merged by `render`; a stray single
714        // item (a hand-built document, or a `Located` wrapper around one)
715        // still renders as its own one-item list instead of panicking —
716        // `nodes` is public API, so every representable tree must serialize.
717        Node::ListItem { .. } => render_list_run(std::slice::from_ref(node), blocks, ctx.strict),
718    }
719}
720
721/// The Markdown for a picture under the active [`ImageMode`]; Referenced mode also
722/// records the bytes in `ctx.artifacts` for the caller to write.
723/// docling-core's `_humanize_text`: underscores to spaces, first letter
724/// capitalized ("line_chart" -> "Line chart").
725fn humanize_label(label: &str) -> String {
726    let text = label.replace('_', " ");
727    let mut chars = text.chars();
728    match chars.next() {
729        Some(f) => f.to_uppercase().collect::<String>() + chars.as_str(),
730        None => text,
731    }
732}
733
734fn picture_marker(image: Option<&crate::PictureImage>, ctx: &mut Ctx) -> String {
735    match (ctx.images, image) {
736        (ImageMode::Embedded, Some(img)) => format!("![Image]({})", img.data_uri()),
737        (ImageMode::Referenced, Some(img)) => {
738            let path = format!(
739                "{}/image_{:06}.{}",
740                ctx.artifacts_dir,
741                ctx.pic_index,
742                ext_for(&img.mimetype)
743            );
744            ctx.pic_index += 1;
745            ctx.artifacts.push((path.clone(), img.data.clone()));
746            format!("![Image]({})", escape_uri_path(&path))
747        }
748        // Placeholder, or any mode with no extracted image.
749        _ => "<!-- image -->".to_string(),
750    }
751}
752
753/// Encode a URL or filesystem path as a Markdown link destination —
754/// docling-core's `MarkdownPictureSerializer._escape_uri_path`
755/// (docling-core#698, 2.94). Handles URLs of any scheme as well as POSIX and
756/// Windows paths, keeps relative paths relative and never double-encodes:
757/// backslashes become `/` (a backslash is both the Windows separator and a
758/// Markdown escape), a UNC share `//host/…` and an absolute Windows path
759/// `C:/…` become RFC 8089 `file://` URLs (the one spelling a renderer cannot
760/// misread as a scheme-relative URL or a `C:` scheme), a URL keeps its
761/// scheme / authority / delimiters with only the components encoded, and
762/// everything else is percent-encoded as a path. `%` is kept so an
763/// already-encoded destination stays as it is; spaces and parentheses are
764/// encoded because they would end (or unbalance) a Markdown inline link.
765pub(crate) fn escape_uri_path(value: &str) -> String {
766    const KEEP: &str = "/%:@+,;=~$!&'*";
767    let s = value.replace('\\', "/");
768    if let Some(rest) = s.strip_prefix("//") {
769        // A fileshare: `file://<host>/<path>`, the host possibly empty.
770        let rest = rest.trim_start_matches('/');
771        let (host, tail) = rest.split_once('/').unwrap_or((rest, ""));
772        return format!("file://{host}{}", percent_quote(&format!("/{tail}"), KEEP));
773    }
774    let bytes = s.as_bytes();
775    if bytes.len() >= 3 && bytes[0].is_ascii_alphabetic() && bytes[1] == b':' && bytes[2] == b'/' {
776        // A Windows path with a drive letter: `file:///C:/…`.
777        return format!("file:///{}", percent_quote(&s, KEEP));
778    }
779    // A URL keeps its scheme, authority and delimiters; only its components are
780    // encoded. A single-character scheme cannot be real (it is a drive letter,
781    // handled above), so it is read as a path — like `urlsplit`.
782    if let Some((scheme, rest)) = s.split_once(':') {
783        let valid_scheme = scheme.len() > 1
784            && scheme.as_bytes()[0].is_ascii_alphabetic()
785            && scheme
786                .bytes()
787                .all(|b| b.is_ascii_alphanumeric() || matches!(b, b'+' | b'-' | b'.'));
788        if valid_scheme {
789            let (authority, rest) = match rest.strip_prefix("//") {
790                Some(r) => {
791                    let end = r.find(['/', '?', '#']).unwrap_or(r.len());
792                    (Some(&r[..end]), &r[end..])
793                }
794                None => (None, rest),
795            };
796            let (before_frag, fragment) = rest.split_once('#').unwrap_or((rest, ""));
797            let (path, query) = before_frag.split_once('?').unwrap_or((before_frag, ""));
798            let mut out = format!("{scheme}:");
799            if let Some(a) = authority {
800                out.push_str("//");
801                out.push_str(a);
802            }
803            out.push_str(&percent_quote(path, KEEP));
804            if !query.is_empty() {
805                out.push('?');
806                out.push_str(&percent_quote(query, KEEP));
807            }
808            if !fragment.is_empty() {
809                out.push('#');
810                out.push_str(&percent_quote(fragment, KEEP));
811            }
812            return out;
813        }
814    }
815    // A relative or root-relative local path.
816    percent_quote(&s, KEEP)
817}
818
819/// `urllib.parse.quote(s, safe)`: unreserved ASCII (`A–Z a–z 0–9 _ . - ~`) and
820/// the `safe` set stay, every other byte of the UTF-8 encoding becomes `%XX`.
821fn percent_quote(s: &str, safe: &str) -> String {
822    let mut out = String::with_capacity(s.len());
823    for &b in s.as_bytes() {
824        let keep = b.is_ascii_alphanumeric()
825            || matches!(b, b'_' | b'.' | b'-' | b'~')
826            || (b.is_ascii() && safe.contains(b as char));
827        if keep {
828            out.push(b as char);
829        } else {
830            out.push_str(&format!("%{b:02X}"));
831        }
832    }
833    out
834}
835
836fn ext_for(mimetype: &str) -> &str {
837    match mimetype {
838        "image/jpeg" => "jpg",
839        "image/gif" => "gif",
840        "image/webp" => "webp",
841        "image/bmp" => "bmp",
842        "image/tiff" => "tif",
843        _ => "png",
844    }
845}
846
847/// Render a table. `compact` selects between two serializers:
848///
849/// - **padded** (default) — docling-core's `tabulate(tablefmt="github")`: columns
850///   are padded to a fixed width (header width + a minimum padding of 2, or the
851///   widest data cell); numeric columns (every data cell parses as a number) are
852///   right-aligned, others left-aligned; separators are plain dashes of
853///   `width + 2`. Matches current published docling (DOCX/HTML conformance).
854/// - **compact** — `| a | b |` cells with single-dash `| - | - |` separators, no
855///   width padding. Matches the committed PDF groundtruth corpus, which predates
856///   the padded serializer.
857///
858/// Each cell is first escaped (`\n` → space, `|` → `&#124;`) so it can't break the
859/// table. The header row is the table's leading `column_header` block flattened
860/// to one row ([`Table::header_row_count`] + [`flatten_header_rows`],
861/// docling-core#723); alignment and widths are computed over the body rows.
862/// Whether a table cell counts as a number for column alignment, matching
863/// `tabulate`'s detection: an ordinary float/int (`f64`-parseable, covering
864/// `1e2`/`inf`/`+1.5`) **or** a thousands-separated number like `7,015`.
865fn is_number_cell(t: &str) -> bool {
866    t.parse::<f64>().is_ok() || is_thousands_number(t)
867}
868
869/// A number with comma thousands-separators, per `tabulate`'s
870/// `_float_with_thousands_separators` regex
871/// (`^(([+-]?[0-9]{1,3})(?:,([0-9]{3}))*)?(?(1)\.[0-9]*|\.[0-9]+)?$`): the
872/// integer part is 1–3 digits then any number of `,ddd` groups; the fraction is
873/// optional (and, without an integer part, must have at least one digit).
874fn is_thousands_number(t: &str) -> bool {
875    let b = t.as_bytes();
876    let mut i = 0;
877    let start = i;
878    if i < b.len() && (b[i] == b'+' || b[i] == b'-') {
879        i += 1;
880    }
881    // First digit chunk: 1–3 digits.
882    let d0 = i;
883    while i < b.len() && b[i].is_ascii_digit() && i - d0 < 3 {
884        i += 1;
885    }
886    let has_int = i > d0;
887    if has_int {
888        // Subsequent `,ddd` groups (exactly three digits each).
889        while i + 3 < b.len() + 1
890            && b.get(i) == Some(&b',')
891            && b.get(i + 1).is_some_and(u8::is_ascii_digit)
892            && b.get(i + 2).is_some_and(u8::is_ascii_digit)
893            && b.get(i + 3).is_some_and(u8::is_ascii_digit)
894        {
895            i += 4;
896        }
897    } else {
898        // A sign only counts with an integer part.
899        i = start;
900    }
901    // Optional fraction.
902    if i < b.len() && b[i] == b'.' {
903        i += 1;
904        let f0 = i;
905        while i < b.len() && b[i].is_ascii_digit() {
906            i += 1;
907        }
908        if !has_int && i == f0 {
909            return false; // `.` with no digits and no integer part
910        }
911    } else if !has_int {
912        return false; // neither integer nor fractional part
913    }
914    i == b.len()
915}
916
917/// The single GFM header row for a table: the leading header rows (see
918/// [`Table::header_row_count`]) flattened per column, texts joined with
919/// `" - "` after dropping consecutive duplicates — docling-core's
920/// `_flatten_header_rows` (docling-core#723). The duplicate rule is what
921/// keeps a row-spanning header from being joined to itself (the grid repeats
922/// its text into every row it covers); it is position-based, so two stacked
923/// levels sharing a label collapse too — GFM has one header row, and upstream
924/// accepts that loss. No header rows → one empty header cell per column.
925fn flatten_header_rows(header_rows: &[Vec<String>], num_cols: usize) -> Vec<String> {
926    (0..num_cols)
927        .map(|c| {
928            let mut parts: Vec<&str> = Vec::new();
929            for row in header_rows {
930                let text = row.get(c).map(String::as_str).unwrap_or("");
931                if !text.is_empty() && parts.last() != Some(&text) {
932                    parts.push(text);
933                }
934            }
935            parts.join(" - ")
936        })
937        .collect()
938}
939
940pub(crate) fn render_table(table: &Table, compact: bool) -> String {
941    if table.rows.is_empty() {
942        return String::new();
943    }
944    let num_cols = table.rows.iter().map(Vec::len).max().unwrap_or(0);
945    if num_cols == 0 {
946        return String::new();
947    }
948
949    // Escaped, rectangular grid (ragged rows padded with empty cells). The
950    // header block is resolved to the one row GFM allows (docling-core#723);
951    // `tabulate` strips data cells of surrounding whitespace but leaves the
952    // header texts as-is.
953    let num_headers = table.header_row_count().min(table.rows.len());
954    let escaped = |r: usize| -> Vec<String> {
955        (0..num_cols)
956            .map(|c| escape_cell(table.rows[r].get(c).map(String::as_str).unwrap_or("")))
957            .collect()
958    };
959    let header_rows: Vec<Vec<String>> = (0..num_headers).map(escaped).collect();
960    let header = flatten_header_rows(&header_rows, num_cols);
961    let body: Vec<Vec<String>> = (num_headers..table.rows.len())
962        .map(|r| {
963            escaped(r)
964                .into_iter()
965                .map(|c| c.trim().to_string())
966                .collect()
967        })
968        .collect();
969
970    if compact {
971        // Compact: cells joined by " | ", no padding, single-dash separators.
972        let render_row = |row: &[String]| -> String { format!("| {} |", row.join(" | ")) };
973        let mut lines = Vec::with_capacity(body.len() + 2);
974        lines.push(render_row(&header));
975        let sep: Vec<&str> = (0..num_cols).map(|_| "-").collect();
976        lines.push(format!("| {} |", sep.join(" | ")));
977        for row in &body {
978            lines.push(render_row(row));
979        }
980        return lines.join("\n");
981    }
982
983    // Display width (Unicode scalar count — good enough for now).
984    let dw = |s: &str| s.chars().count();
985
986    // A column is right-aligned when at least one body cell is numeric and every
987    // non-empty body cell is numeric — matching `tabulate`'s column typing, where
988    // empty cells are "missing" (ignored) and a number may carry thousands
989    // separators (`7,015`), which a plain `f64` parse rejects.
990    let right: Vec<bool> = (0..num_cols)
991        .map(|c| {
992            let mut any = false;
993            for row in &body {
994                let t = row[c].trim();
995                if t.is_empty() {
996                    continue;
997                }
998                if !is_number_cell(t) {
999                    return false;
1000                }
1001                any = true;
1002            }
1003            any
1004        })
1005        .collect();
1006
1007    // Column width = max(header_width + MIN_PADDING(2), max body-cell width).
1008    let width: Vec<usize> = (0..num_cols)
1009        .map(|c| {
1010            let mut w = dw(&header[c]) + 2;
1011            for row in &body {
1012                w = w.max(dw(&row[c]));
1013            }
1014            w
1015        })
1016        .collect();
1017
1018    let fmt_cell = |s: &str, c: usize| -> String {
1019        let pad = " ".repeat(width[c].saturating_sub(dw(s)));
1020        let body = if right[c] {
1021            format!("{pad}{s}")
1022        } else {
1023            format!("{s}{pad}")
1024        };
1025        format!(" {body} ")
1026    };
1027    let render_row = |row: &[String]| -> String {
1028        let cells: Vec<String> = (0..num_cols).map(|c| fmt_cell(&row[c], c)).collect();
1029        format!("|{}|", cells.join("|"))
1030    };
1031
1032    let mut lines = Vec::with_capacity(body.len() + 2);
1033    lines.push(render_row(&header));
1034    let sep: Vec<String> = (0..num_cols).map(|c| "-".repeat(width[c] + 2)).collect();
1035    lines.push(format!("|{}|", sep.join("|")));
1036    for row in &body {
1037        lines.push(render_row(row));
1038    }
1039    lines.join("\n")
1040}
1041
1042/// Escape a table cell so it can't break the markdown table: newlines become
1043/// spaces and pipes become the `&#124;` HTML entity (matches docling-core).
1044fn escape_cell(s: &str) -> String {
1045    s.replace('\n', " ").replace('|', "&#124;")
1046}
1047
1048#[cfg(test)]
1049mod tests {
1050    use super::*;
1051    use crate::{PictureImage, TableCell, TableStructure};
1052
1053    /// #385: where one list ends and the next begins is the backend's call
1054    /// (`first_in_list`), never the serializer's. An ordered run `1.` → `5.`
1055    /// is one list (an AsciiDoc numbered list around a nested one), and so are
1056    /// mixed bullet/ordered items the backend did not separate; only a flagged
1057    /// item opens a new list and earns the blank line.
1058    #[test]
1059    fn list_boundaries_come_from_the_backend_not_the_numbering() {
1060        let item = |ordered: bool, number: u64, first_in_list: bool, text: &str| Node::ListItem {
1061            ordered,
1062            number,
1063            first_in_list,
1064            text: text.into(),
1065            level: 0,
1066            marker: None,
1067            location: None,
1068            dclx: None,
1069            href: None,
1070            layer: None,
1071        };
1072        let md = |items: Vec<Node>| {
1073            let mut doc = DoclingDocument::new("t");
1074            for n in items {
1075                doc.push(n);
1076            }
1077            doc.export_to_markdown()
1078        };
1079        // A number gap alone is not a boundary.
1080        assert_eq!(
1081            md(vec![
1082                item(true, 1, true, "one"),
1083                item(true, 5, false, "five")
1084            ]),
1085            "1. one\n5. five\n"
1086        );
1087        // Nor is a kind flip the backend did not flag …
1088        assert_eq!(
1089            md(vec![
1090                item(false, 0, true, "bullet"),
1091                item(true, 1, false, "one"),
1092                item(false, 0, false, "bullet two"),
1093            ]),
1094            "- bullet\n1. one\n- bullet two\n"
1095        );
1096        // … while a flagged item is one, whatever its number says.
1097        assert_eq!(
1098            md(vec![
1099                item(true, 1, true, "a"),
1100                item(true, 2, false, "b"),
1101                item(true, 3, true, "new list, continuing count"),
1102            ]),
1103            "1. a\n2. b\n\n3. new list, continuing count\n"
1104        );
1105    }
1106
1107    #[test]
1108    fn renders_headings_paragraphs_and_lists() {
1109        let mut doc = DoclingDocument::new("demo");
1110        doc.add_heading(1, "Title");
1111        doc.add_paragraph("Hello world.");
1112        doc.push(Node::ListItem {
1113            ordered: false,
1114            number: 1,
1115            first_in_list: true,
1116            text: "first".into(),
1117            level: 0,
1118            marker: None,
1119            location: None,
1120            dclx: None,
1121            href: None,
1122            layer: None,
1123        });
1124        doc.push(Node::ListItem {
1125            ordered: false,
1126            number: 2,
1127            first_in_list: false,
1128            text: "second".into(),
1129            level: 0,
1130            marker: None,
1131            location: None,
1132            dclx: None,
1133            href: None,
1134            layer: None,
1135        });
1136        let md = doc.export_to_markdown();
1137        assert_eq!(md, "# Title\n\nHello world.\n\n- first\n- second\n");
1138    }
1139
1140    /// docling-core 2.92 (#721): a single newline inside an item's text is a
1141    /// GFM hard line break, a blank line stays a paragraph break, and a heading
1142    /// collapses its newline to a space. Nested-table dumps stay verbatim.
1143    #[test]
1144    fn single_newlines_become_gfm_hard_line_breaks() {
1145        let mut doc = DoclingDocument::new("t");
1146        doc.push(Node::Heading {
1147            level: 1,
1148            text: "Hello\nWorld".into(),
1149        });
1150        doc.push(Node::Paragraph {
1151            text: "line one\nline two\n\npara two".into(),
1152        });
1153        doc.push(Node::ListItem {
1154            ordered: false,
1155            number: 1,
1156            first_in_list: true,
1157            text: "item\ncontinued".into(),
1158            level: 0,
1159            marker: None,
1160            location: None,
1161            dclx: None,
1162            href: None,
1163            layer: None,
1164        });
1165        doc.push(Node::TextDump("A1 B1 \n\n\nC1".into()));
1166        assert_eq!(
1167            doc.export_to_markdown(),
1168            "# Hello World\n\nline one  \nline two\n\npara two\n\n- item  \ncontinued\n\nA1 B1 \n\n\nC1\n"
1169        );
1170    }
1171
1172    /// docling-core#540: inside a rich table cell a heading is plain text;
1173    /// docling-core#724: a field region renders only its items' key/value text.
1174    #[test]
1175    fn table_cell_mode_and_field_regions() {
1176        let mut doc = DoclingDocument::new("t");
1177        doc.push(Node::Heading {
1178            level: 2,
1179            text: "A  text".into(),
1180        });
1181        doc.push(Node::Paragraph {
1182            text: "body".into(),
1183        });
1184        assert_eq!(to_markdown_table_cell(&doc, false), "A  text\n\nbody");
1185        assert_eq!(doc.export_to_markdown(), "## A  text\n\nbody\n");
1186
1187        let mut doc = DoclingDocument::new("f");
1188        doc.push(Node::FieldRegion {
1189            items: vec![crate::FieldItem {
1190                marker: None,
1191                key: Some("Name:".into()),
1192                value: Some("John Doe".into()),
1193                value_kind: None,
1194            }],
1195        });
1196        assert_eq!(doc.export_to_markdown(), "Name:\n\nJohn Doe\n");
1197    }
1198
1199    #[test]
1200    fn strict_renders_recovered_links_legacy_does_not() {
1201        let mut doc = DoclingDocument::new("cv");
1202        doc.add_paragraph("Find me on LinkedIn or GitHub.");
1203        doc.links = vec![
1204            ("LinkedIn".into(), "https://www.linkedin.com/in/x/".into()),
1205            ("GitHub".into(), "https://github.com/x/".into()),
1206        ];
1207        // Legacy/docling mode: links are left untouched (conformance preserved).
1208        assert_eq!(doc.export_to_markdown(), "Find me on LinkedIn or GitHub.\n");
1209        // Strict mode: anchors become Markdown links.
1210        assert_eq!(
1211            doc.export_to_markdown_with(true),
1212            "Find me on [LinkedIn](https://www.linkedin.com/in/x/) or [GitHub](https://github.com/x/).\n"
1213        );
1214    }
1215
1216    #[test]
1217    fn strict_links_match_escaped_anchor_and_consume_in_order() {
1218        let mut doc = DoclingDocument::new("d");
1219        // The PDF assembler HTML-escapes prose, so by serialization time the body
1220        // already carries `&amp;`; the anchor is stored un-escaped. The matcher must
1221        // escape the anchor to find it. Two identical anchors link in document order.
1222        doc.add_paragraph("AI &amp; ML here, and issues here, then issues there.");
1223        doc.links = vec![
1224            ("AI & ML".into(), "https://a/".into()),
1225            ("issues".into(), "https://first/".into()),
1226            ("issues".into(), "https://second/".into()),
1227        ];
1228        assert_eq!(
1229            doc.export_to_markdown_with(true),
1230            "[AI &amp; ML](https://a/) here, and [issues](https://first/) here, then [issues](https://second/) there.\n"
1231        );
1232    }
1233
1234    /// docling-core#698: the referenced-image destination is percent-encoded —
1235    /// upstream's own case table (paths, Windows flavours, UNC, URLs) plus
1236    /// idempotency on the encoded result.
1237    #[test]
1238    fn referenced_image_destinations_are_escaped() {
1239        let cases = [
1240            (
1241                "doc_artifacts/image_000001_ab12.png",
1242                "doc_artifacts/image_000001_ab12.png",
1243            ),
1244            (
1245                "My Report_artifacts/img.png",
1246                "My%20Report_artifacts/img.png",
1247            ),
1248            ("artifacts/img (1).png", "artifacts/img%20%281%29.png"),
1249            ("100%_scale/a#b?c.png", "100%_scale/a%23b%3Fc.png"),
1250            ("/home/a b/img.png", "/home/a%20b/img.png"),
1251            (
1252                "My Report_artifacts\\img.png",
1253                "My%20Report_artifacts/img.png",
1254            ),
1255            (
1256                "C:/Users/me/My Docs/img.png",
1257                "file:///C:/Users/me/My%20Docs/img.png",
1258            ),
1259            ("C:\\Users\\me\\img.png", "file:///C:/Users/me/img.png"),
1260            (
1261                "//server/share/My Docs/img.png",
1262                "file://server/share/My%20Docs/img.png",
1263            ),
1264            ("\\\\server\\share\\img.png", "file://server/share/img.png"),
1265            ("file:///home/a b/img.png", "file:///home/a%20b/img.png"),
1266            (
1267                "s3://bucket/My Report_artifacts/img.png",
1268                "s3://bucket/My%20Report_artifacts/img.png",
1269            ),
1270            (
1271                "https://example.com:8080/a b.png?w=1&h=2#frag",
1272                "https://example.com:8080/a%20b.png?w=1&h=2#frag",
1273            ),
1274            (
1275                "https://example.com/img (1).png",
1276                "https://example.com/img%20%281%29.png",
1277            ),
1278            ("caf\u{e9}/im\u{e4}ge.png", "caf%C3%A9/im%C3%A4ge.png"),
1279        ];
1280        for (input, expected) in cases {
1281            assert_eq!(escape_uri_path(input), expected, "input {input:?}");
1282            assert_eq!(
1283                escape_uri_path(expected),
1284                expected,
1285                "idempotent {expected:?}"
1286            );
1287        }
1288        // The whole marker, through the referenced-image export.
1289        let mut doc = DoclingDocument::new("t");
1290        doc.push(Node::Picture {
1291            caption: None,
1292            caption_href: None,
1293            image: Some(PictureImage {
1294                mimetype: "image/png".into(),
1295                width: 1,
1296                height: 1,
1297                data: b"x".to_vec(),
1298            }),
1299            classification: None,
1300            caption_parent: Default::default(),
1301        });
1302        let (md, files) = doc
1303            .export_to_markdown_with_images(ImageMode::Referenced, "My Report (final)_artifacts");
1304        assert!(
1305            md.contains("![Image](My%20Report%20%28final%29_artifacts/image_000000.png)"),
1306            "got:\n{md}"
1307        );
1308        // The file path handed back for writing stays unescaped.
1309        assert_eq!(files[0].0, "My Report (final)_artifacts/image_000000.png");
1310    }
1311
1312    /// Pictures the HTML backend folds into a list item print after the item
1313    /// line with plain newlines; a `<br>` newline in the item's own text is
1314    /// still a GFM hard line break.
1315    #[test]
1316    fn folded_list_item_pictures_keep_plain_newlines() {
1317        assert_eq!(
1318            list_item_text("Step\n<!-- image -->", false),
1319            "Step\n<!-- image -->"
1320        );
1321        assert_eq!(
1322            list_item_text("Step\nAlt text\n<!-- image -->\n<!-- image -->", false),
1323            "Step\nAlt text\n<!-- image -->\n<!-- image -->"
1324        );
1325        assert_eq!(
1326            list_item_text("line one\nline two", false),
1327            "line one  \nline two"
1328        );
1329    }
1330
1331    /// docling-core#723: the header block is the leading run of rows on which a
1332    /// `column_header` cell starts, flattened per column with " - ".
1333    #[test]
1334    fn stacked_header_rows_flatten_into_one() {
1335        let mut t = Table {
1336            rows: vec![
1337                vec!["".into(), "% of Total".into(), "% of Total".into()],
1338                vec!["class".into(), "Train".into(), "Test".into()],
1339                vec!["Caption".into(), "2.04".into(), "1.77".into()],
1340            ],
1341            ..Default::default()
1342        };
1343        t.structure = Some(TableStructure {
1344            header_row: vec![true, true, false],
1345            col_continuation: vec![
1346                vec![false, false, true],
1347                vec![false, false, false],
1348                vec![false, false, false],
1349            ],
1350            ..Default::default()
1351        });
1352        assert_eq!(t.header_row_count(), 2);
1353        assert_eq!(
1354            render_table(&t, true),
1355            "| class | % of Total - Train | % of Total - Test |\n| - | - | - |\n| Caption | 2.04 | 1.77 |"
1356        );
1357        // padded: widths from the flattened header, alignment from body rows
1358        assert_eq!(
1359            render_table(&t, false),
1360            "| class   |   % of Total - Train |   % of Total - Test |\n\
1361             |---------|----------------------|---------------------|\n\
1362             | Caption |                 2.04 |                1.77 |"
1363        );
1364    }
1365
1366    /// A header spanning two rows is repeated into the second row by the grid;
1367    /// that row is not a header row unless another header cell starts there.
1368    #[test]
1369    fn vertically_spanning_header_does_not_extend_the_block() {
1370        let mut t = Table {
1371            rows: vec![
1372                vec!["Name".into(), "Value".into()],
1373                vec!["Name".into(), "1".into()],
1374                vec!["x".into(), "2".into()],
1375            ],
1376            ..Default::default()
1377        };
1378        t.structure = Some(TableStructure {
1379            col_header: vec![vec![true, true], vec![true, false], vec![false, false]],
1380            row_continuation: vec![vec![false, false], vec![true, false], vec![false, false]],
1381            ..Default::default()
1382        });
1383        assert_eq!(t.header_row_count(), 1);
1384        assert_eq!(
1385            render_table(&t, true),
1386            "| Name | Value |\n| - | - |\n| Name | 1 |\n| x | 2 |"
1387        );
1388    }
1389
1390    /// Flags that begin on a later row promote nothing: every row stays in the
1391    /// body under an empty header row (tabulate's `headers=["", ""]`).
1392    #[test]
1393    fn header_flags_not_on_row_zero_keep_all_rows_in_the_body() {
1394        let mut t = Table {
1395            rows: vec![
1396                vec!["1".into(), "2".into()],
1397                vec!["a".into(), "b".into()],
1398                vec!["333".into(), "4".into()],
1399            ],
1400            ..Default::default()
1401        };
1402        t.structure = Some(TableStructure {
1403            header_row: vec![false, true, false],
1404            ..Default::default()
1405        });
1406        assert_eq!(t.header_row_count(), 0);
1407        assert_eq!(
1408            render_table(&t, false),
1409            "|     |    |\n|-----|----|\n| 1   | 2  |\n| a   | b  |\n| 333 | 4  |"
1410        );
1411    }
1412
1413    /// A pivot table's row headers (`<th rowspan>`) carry `row_header`, not
1414    /// `column_header` (docling#4216), so the data row beside them is not
1415    /// pulled into the header block — what this port used to reach with a
1416    /// deviation now falls out of the flags themselves.
1417    #[test]
1418    fn pivot_row_headers_do_not_extend_the_header() {
1419        let mut t = Table {
1420            rows: vec![
1421                vec!["Year".into(), "Month".into()],
1422                vec!["2025".into(), "January".into()],
1423                vec!["2025".into(), "February".into()],
1424            ],
1425            ..Default::default()
1426        };
1427        t.structure = Some(TableStructure {
1428            col_header: vec![vec![true, true], vec![false, false], vec![false, false]],
1429            row_header: vec![vec![false, false], vec![true, false], vec![true, false]],
1430            row_continuation: vec![vec![false, false], vec![false, false], vec![true, false]],
1431            ..Default::default()
1432        });
1433        assert_eq!(t.header_row_count(), 1);
1434        assert_eq!(
1435            render_table(&t, true),
1436            "| Year | Month |\n| - | - |\n| 2025 | January |\n| 2025 | February |"
1437        );
1438    }
1439
1440    /// No `column_header` anywhere (first-class cells without flags) → row 0
1441    /// stays the header, as before.
1442    #[test]
1443    fn unflagged_cells_keep_row_zero_as_header() {
1444        let mut t = Table {
1445            rows: vec![vec!["h".into()], vec!["d".into()]],
1446            ..Default::default()
1447        };
1448        t.cells = Some(
1449            [(0usize, "h"), (1, "d")]
1450                .into_iter()
1451                .map(|(r, text)| TableCell {
1452                    text: text.into(),
1453                    bbox: None,
1454                    start_row: r,
1455                    start_col: 0,
1456                    row_span: 1,
1457                    col_span: 1,
1458                    column_header: false,
1459                    row_header: false,
1460                    row_section: false,
1461                })
1462                .collect(),
1463        );
1464        assert_eq!(t.header_row_count(), 1);
1465        assert_eq!(render_table(&t, true), "| h |\n| - |\n| d |");
1466    }
1467
1468    #[test]
1469    fn renders_compact_table() {
1470        let mut doc = DoclingDocument::new("t");
1471        // The compact form is opt-in (the PDF backend sets it); default output uses
1472        // the padded GitHub serializer (covered by the regression fixtures).
1473        doc.compact_tables = true;
1474        doc.push(Node::Table(Table {
1475            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
1476            location: None,
1477            structure: None,
1478            cell_blocks: None,
1479            cells: None,
1480            caption: None,
1481            caption_parent: Default::default(),
1482        }));
1483        let md = doc.export_to_markdown();
1484        assert_eq!(md, "| a | b |\n| - | - |\n| 1 | 2 |\n");
1485    }
1486
1487    #[test]
1488    fn renders_padded_github_table_by_default() {
1489        let mut doc = DoclingDocument::new("t");
1490        doc.push(Node::Table(Table {
1491            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
1492            location: None,
1493            structure: None,
1494            cell_blocks: None,
1495            cells: None,
1496            caption: None,
1497            caption_parent: Default::default(),
1498        }));
1499        let md = doc.export_to_markdown();
1500        // Numeric data columns are right-aligned; columns padded to header+2.
1501        assert_eq!(md, "|   a |   b |\n|-----|-----|\n|   1 |   2 |\n");
1502    }
1503
1504    #[test]
1505    fn strict_unescapes_inline_underscores_legacy_keeps_them() {
1506        let mut doc = DoclingDocument::new("t");
1507        doc.add_heading(1, "a\\_b");
1508        doc.add_paragraph("x\\_y");
1509        doc.push(Node::ListItem {
1510            ordered: false,
1511            number: 1,
1512            first_in_list: true,
1513            text: "i\\_j".into(),
1514            level: 0,
1515            marker: None,
1516            location: None,
1517            dclx: None,
1518            href: None,
1519            layer: None,
1520        });
1521        // Legacy reproduces docling's `\_` escaping byte-for-byte.
1522        assert_eq!(doc.export_to_markdown(), "# a\\_b\n\nx\\_y\n\n- i\\_j\n");
1523        // Strict prefers literal underscores (Rust-only readability mode).
1524        assert_eq!(doc.export_to_markdown_with(true), "# a_b\n\nx_y\n\n- i_j\n");
1525    }
1526
1527    /// Drive a document's nodes through [`MarkdownStreamer`] in the given page
1528    /// splits and assert the concatenated chunks equal the buffered serializer.
1529    fn assert_stream_matches(
1530        doc: &DoclingDocument,
1531        strict: bool,
1532        images: ImageMode,
1533        splits: &[usize],
1534    ) {
1535        let (want, want_artifacts) = to_markdown_images(doc, strict, images, "artifacts");
1536        let mut streamer =
1537            MarkdownStreamer::with_artifacts(strict, images, doc.compact_tables, "artifacts")
1538                .with_page_break_placeholder(doc.page_break_placeholder.clone());
1539        let mut got = String::new();
1540        let mut got_artifacts = Vec::new();
1541        let mut start = 0;
1542        for &end in splits {
1543            // Links only matter in strict mode; feed them all with the first batch
1544            // that has content (document order is preserved by the queue).
1545            let links = if start == 0 {
1546                doc.links.as_slice()
1547            } else {
1548                &[]
1549            };
1550            got.push_str(&streamer.push(&doc.nodes[start..end], links));
1551            // Referenced mode: drain per push, as a real caller writing files
1552            // page by page would — numbering must continue across drains.
1553            got_artifacts.extend(streamer.take_artifacts());
1554            start = end;
1555        }
1556        got.push_str(&streamer.push(
1557            &doc.nodes[start..],
1558            if start == 0 {
1559                doc.links.as_slice()
1560            } else {
1561                &[]
1562            },
1563        ));
1564        got_artifacts.extend(streamer.take_artifacts());
1565        got.push_str(&streamer.finish());
1566        assert_eq!(
1567            got, want,
1568            "streamed output diverged (splits={splits:?}, strict={strict})"
1569        );
1570        assert_eq!(
1571            got_artifacts, want_artifacts,
1572            "streamed artifacts diverged (splits={splits:?}, strict={strict})"
1573        );
1574    }
1575
1576    #[test]
1577    fn streaming_is_byte_identical_to_buffered() {
1578        let mut doc = DoclingDocument::new("d");
1579        doc.add_heading(1, "Title");
1580        doc.add_paragraph("First paragraph.");
1581        doc.push(Node::ListItem {
1582            ordered: false,
1583            number: 1,
1584            first_in_list: true,
1585            text: "a".into(),
1586            level: 0,
1587            marker: None,
1588            location: None,
1589            dclx: None,
1590            href: None,
1591            layer: None,
1592        });
1593        doc.push(Node::ListItem {
1594            ordered: false,
1595            number: 2,
1596            first_in_list: false,
1597            text: "b".into(),
1598            level: 0,
1599            marker: None,
1600            location: None,
1601            dclx: None,
1602            href: None,
1603            layer: None,
1604        });
1605        doc.push(Node::Code {
1606            language: Some("rust".into()),
1607            text: "let x = 1;".into(),
1608            orig: None,
1609            pretty: None,
1610        });
1611        doc.push(Node::Table(Table {
1612            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
1613            location: None,
1614            structure: None,
1615            cell_blocks: None,
1616            cells: None,
1617            caption: None,
1618            caption_parent: Default::default(),
1619        }));
1620        doc.push(Node::Picture {
1621            caption: Some("Fig 1".into()),
1622            caption_href: None,
1623            image: Some(PictureImage {
1624                mimetype: "image/png".into(),
1625                width: 2,
1626                height: 2,
1627                data: b"png-one".to_vec(),
1628            }),
1629            classification: None,
1630            caption_parent: Default::default(),
1631        });
1632        doc.add_paragraph("Last paragraph.");
1633        // A second embedded picture, so referenced mode must keep numbering
1634        // (`image_000001`) across chunk boundaries.
1635        doc.push(Node::Picture {
1636            caption: None,
1637            caption_href: None,
1638            image: Some(PictureImage {
1639                mimetype: "image/png".into(),
1640                width: 2,
1641                height: 2,
1642                data: b"png-two".to_vec(),
1643            }),
1644            classification: None,
1645            caption_parent: Default::default(),
1646        });
1647
1648        // A run of list items must never straddle a split, so try splits that fall
1649        // on safe block boundaries (the streaming PDF assembler guarantees this).
1650        for &strict in &[false, true] {
1651            for &images in &[
1652                ImageMode::Placeholder,
1653                ImageMode::Embedded,
1654                ImageMode::Referenced,
1655            ] {
1656                for splits in [&[][..], &[1][..], &[2][..], &[4][..], &[1, 4, 6, 7][..]] {
1657                    assert_stream_matches(&doc, strict, images, splits);
1658                }
1659            }
1660        }
1661    }
1662
1663    #[test]
1664    fn streaming_applies_recovered_links_in_strict_mode() {
1665        let mut doc = DoclingDocument::new("d");
1666        doc.add_paragraph("See LinkedIn for details.");
1667        doc.add_paragraph("And GitHub too.");
1668        doc.links = vec![
1669            ("LinkedIn".into(), "https://lnkd/".into()),
1670            ("GitHub".into(), "https://gh/".into()),
1671        ];
1672        // The second anchor lives in the second block, so it must be carried across
1673        // the page boundary and placed when that block streams out.
1674        assert_stream_matches(&doc, true, ImageMode::Placeholder, &[1]);
1675    }
1676
1677    /// A three-page document with one empty page in the middle and page
1678    /// markers of both kinds, as the backends emit them.
1679    fn paged_doc() -> DoclingDocument {
1680        let mut doc = DoclingDocument::new("p");
1681        doc.push(Node::PageInfo {
1682            page_no: 1,
1683            width: 100.0,
1684            height: 100.0,
1685        });
1686        doc.add_heading(1, "Title");
1687        doc.add_paragraph("Page one.");
1688        // Page two: a marker plus furniture only — renders nothing.
1689        doc.push(Node::PageBreak);
1690        doc.push(Node::PageInfo {
1691            page_no: 2,
1692            width: 100.0,
1693            height: 100.0,
1694        });
1695        doc.push(Node::PageFurniture {
1696            footer: true,
1697            location: [0, 500, 511, 511],
1698            text: "2".into(),
1699        });
1700        doc.push(Node::PageBreak);
1701        doc.push(Node::PageInfo {
1702            page_no: 3,
1703            width: 100.0,
1704            height: 100.0,
1705        });
1706        doc.add_paragraph("Page three.");
1707        // A trailing boundary with nothing after it.
1708        doc.push(Node::PageBreak);
1709        doc
1710    }
1711
1712    #[test]
1713    fn page_break_placeholder_lands_between_pages_only() {
1714        let mut doc = paged_doc();
1715        // Off by default: docling's Markdown carries no page breaks.
1716        assert_eq!(
1717            doc.export_to_markdown(),
1718            "# Title\n\nPage one.\n\nPage three.\n"
1719        );
1720        doc.page_break_placeholder = Some("<!-- page break -->".into());
1721        // One break for the 1→3 transition (the empty page 2 and the doubled
1722        // PageBreak+PageInfo markers collapse), none before the first block,
1723        // none for the trailing boundary.
1724        assert_eq!(
1725            doc.export_to_markdown(),
1726            "# Title\n\nPage one.\n\n<!-- page break -->\n\nPage three.\n"
1727        );
1728        // An empty placeholder is still a (blank) part, as upstream's
1729        // `str.replace(marker, "")` leaves the delimiters around it.
1730        doc.page_break_placeholder = Some(String::new());
1731        assert_eq!(
1732            doc.export_to_markdown(),
1733            "# Title\n\nPage one.\n\n\n\nPage three.\n"
1734        );
1735    }
1736
1737    #[test]
1738    fn page_break_placeholder_never_leads_a_single_page() {
1739        let mut doc = DoclingDocument::new("one");
1740        doc.page_break_placeholder = Some("---".into());
1741        doc.push(Node::PageBreak);
1742        doc.push(Node::PageInfo {
1743            page_no: 1,
1744            width: 10.0,
1745            height: 10.0,
1746        });
1747        doc.add_paragraph("Only page.");
1748        assert_eq!(doc.export_to_markdown(), "Only page.\n");
1749        // Two boundaries with no content between them: still one break.
1750        doc.push(Node::PageBreak);
1751        doc.push(Node::PageBreak);
1752        doc.add_paragraph("Next.");
1753        assert_eq!(doc.export_to_markdown(), "Only page.\n\n---\n\nNext.\n");
1754    }
1755
1756    #[test]
1757    fn page_break_placeholder_streams_byte_identical() {
1758        let mut doc = paged_doc();
1759        doc.page_break_placeholder = Some("<!-- page break -->".into());
1760        // Split at every page marker (how the PDF pipeline pushes page batches)
1761        // and at odd places inside a page: the pending break must survive a
1762        // push that renders nothing (page two) and land on page three's block.
1763        for splits in [
1764            &[3usize][..],
1765            &[3, 6],
1766            &[3, 6, 8],
1767            &[1, 2, 3, 4, 5, 6, 7, 8, 9],
1768            &[8],
1769        ] {
1770            assert_stream_matches(&doc, false, ImageMode::Placeholder, splits);
1771            assert_stream_matches(&doc, true, ImageMode::Placeholder, splits);
1772        }
1773    }
1774
1775    #[test]
1776    fn strict_tightens_punctuation_spacing_legacy_keeps_it() {
1777        let mut doc = DoclingDocument::new("t");
1778        doc.add_paragraph("see [ 37 , 36 ] and ( x ) .");
1779        // Legacy keeps docling's spacing byte-for-byte.
1780        assert_eq!(doc.export_to_markdown(), "see [ 37 , 36 ] and ( x ) .\n");
1781        // Strict tightens punctuation for readable Markdown.
1782        assert_eq!(doc.export_to_markdown_with(true), "see [37, 36] and (x).\n");
1783    }
1784}