Skip to main content

docling_core/
markdown.rs

1//! Markdown serializer for [`DoclingDocument`].
2
3use crate::document::{DoclingDocument, Node, Table};
4
5/// How pictures are rendered (mirrors docling-core's `ImageRefMode`).
6#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
7pub enum ImageMode {
8    /// `<!-- image -->` (docling's default, and the only mode without image data).
9    #[default]
10    Placeholder,
11    /// `![Image](data:<mime>;base64,…)` — self-contained.
12    Embedded,
13    /// `![Image](<artifacts>/image_NNNNNN.<ext>)`; the bytes are returned for the
14    /// caller to write.
15    Referenced,
16}
17
18/// Serializer state threaded through the render walk.
19struct Ctx {
20    strict: bool,
21    /// Emit compact `| a | b |` tables instead of the padded GitHub serializer.
22    compact_tables: bool,
23    images: ImageMode,
24    artifacts_dir: String,
25    /// (relative path, bytes) for each referenced image — written by the caller.
26    artifacts: Vec<(String, Vec<u8>)>,
27    pic_index: usize,
28    /// Rendering the block content of a rich table cell (docling-core 2.94's
29    /// `in_table_cell`, docling-core#540): a heading has no valid Markdown
30    /// form inside a table, so it renders as plain text without `#` markers.
31    in_table_cell: bool,
32}
33
34/// Render a document to a Markdown string (pictures as placeholders).
35///
36/// `strict` selects the serializer-level behaviours that differ between
37/// docling-legacy output and cleaner Markdown — currently the code-fence
38/// language (legacy drops it, strict keeps it).
39pub fn to_markdown(doc: &DoclingDocument, strict: bool) -> String {
40    to_markdown_images(doc, strict, ImageMode::Placeholder, "artifacts").0
41}
42
43/// Render to Markdown with an explicit picture [`ImageMode`]. Returns the
44/// Markdown and, for [`ImageMode::Referenced`], the `(path, bytes)` of each image
45/// the caller should write (relative to the Markdown file).
46pub fn to_markdown_images(
47    doc: &DoclingDocument,
48    strict: bool,
49    images: ImageMode,
50    artifacts_dir: &str,
51) -> (String, Vec<(String, Vec<u8>)>) {
52    let mut ctx = Ctx {
53        strict,
54        compact_tables: doc.compact_tables,
55        images,
56        artifacts_dir: artifacts_dir.to_string(),
57        artifacts: Vec::new(),
58        pic_index: 0,
59        in_table_cell: false,
60    };
61    let mut blocks: Vec<String> = Vec::new();
62    render(&doc.nodes, &mut blocks, &mut ctx);
63    let mut body = blocks.join("\n\n");
64    // Strict mode only: turn recovered source hyperlinks into Markdown links.
65    // docling's standard pipeline drops them, so doing this in legacy mode would
66    // diverge from docling — hence strict-only, leaving conformance output intact.
67    if strict && !doc.links.is_empty() {
68        body = apply_links(&body, &doc.links);
69    }
70    let md = if body.is_empty() {
71        String::new()
72    } else {
73        format!("{body}\n")
74    };
75    (md, ctx.artifacts)
76}
77
78/// Render the block content of a *rich table cell* to Markdown — what
79/// docling-core's table serializer does for a `RichTableCell`
80/// (`doc_serializer.serialize(item, in_table_cell=True)`): the cell's
81/// paragraphs, lists and flattened nested tables render as in a document, but a
82/// heading loses its `#` markers (docling-core#540 — the Markdown spec has no
83/// headings inside tables). Pictures stay placeholders. The caller flattens the
84/// result into its cell text; the table serializer later turns the newlines
85/// into spaces.
86pub fn to_markdown_table_cell(doc: &DoclingDocument, strict: bool) -> String {
87    let mut ctx = Ctx {
88        strict,
89        compact_tables: doc.compact_tables,
90        images: ImageMode::Placeholder,
91        artifacts_dir: String::new(),
92        artifacts: Vec::new(),
93        pic_index: 0,
94        in_table_cell: true,
95    };
96    let mut blocks: Vec<String> = Vec::new();
97    render(&doc.nodes, &mut blocks, &mut ctx);
98    blocks.join("\n\n")
99}
100
101/// Wrap each recovered link's anchor text in Markdown `[anchor](href)`. Anchors
102/// arrive cleaned (curly quotes/dashes already normalized) but un-escaped, so we
103/// match against the body's HTML-escaped (`&`/`<`/`>`) form, the way prose nodes
104/// were serialized. Links are consumed in document order from a moving cursor, so
105/// a repeated anchor (e.g. two "issues") links its successive occurrences rather
106/// than all pointing at the first. An anchor that can't be located is skipped
107/// (its text may have been split across a line wrap or table cell).
108fn apply_links(body: &str, links: &[(String, String)]) -> String {
109    let mut out = body.to_string();
110    let mut cursor = 0usize;
111    for (anchor, href) in links {
112        let anchor = anchor
113            .replace('&', "&amp;")
114            .replace('<', "&lt;")
115            .replace('>', "&gt;");
116        if anchor.is_empty() {
117            continue;
118        }
119        if let Some(rel) = out[cursor..].find(&anchor) {
120            let at = cursor + rel;
121            // Don't relink inside an already-emitted `](` Markdown link target.
122            let replacement = format!("[{anchor}]({href})");
123            out.replace_range(at..at + anchor.len(), &replacement);
124            cursor = at + replacement.len();
125        }
126    }
127    out
128}
129
130/// Like [`apply_links`] but over a single chunk, consuming from a shared queue so
131/// the same `[anchor](href)` rewriting can be applied incrementally as Markdown is
132/// streamed out. Each queued link is matched (in document order) against `chunk`
133/// and rewritten in place; a link whose anchor is not in this chunk is carried
134/// forward in the queue for a later chunk. Anchors are recovered in document
135/// order and a chunk is always a contiguous run of whole blocks, so this
136/// reproduces [`apply_links`]' single moving cursor: the link lands in whichever
137/// chunk contains its anchor, identically to the buffered path. (A link whose
138/// anchor never appears is carried to the end and dropped — the same no-op
139/// `apply_links` performs for an unlocatable anchor.)
140fn apply_links_chunk(chunk: &str, queue: &mut Vec<(String, String)>) -> String {
141    let mut out = chunk.to_string();
142    let mut cursor = 0usize;
143    let mut carried: Vec<(String, String)> = Vec::new();
144    for (anchor_raw, href) in std::mem::take(queue) {
145        let anchor = anchor_raw
146            .replace('&', "&amp;")
147            .replace('<', "&lt;")
148            .replace('>', "&gt;");
149        if anchor.is_empty() {
150            continue;
151        }
152        if let Some(rel) = out[cursor..].find(&anchor) {
153            let at = cursor + rel;
154            let replacement = format!("[{anchor}]({href})");
155            out.replace_range(at..at + anchor.len(), &replacement);
156            cursor = at + replacement.len();
157        } else {
158            // Not in this chunk; try again when its block is flushed.
159            carried.push((anchor_raw, href));
160        }
161    }
162    *queue = carried;
163    out
164}
165
166/// Incremental Markdown serializer: feed finalized, in-document-order batches of
167/// [`Node`]s and receive Markdown chunks whose concatenation is **byte-identical**
168/// to [`to_markdown_images`] over the same nodes. This is the streaming
169/// counterpart of the buffered serializer — used to emit a document's Markdown in
170/// chunks (e.g. page by page, as the parallel PDF pipeline finishes pages) instead
171/// of building the whole string up front.
172///
173/// [`ImageMode::Placeholder`] and [`ImageMode::Embedded`] render inline.
174/// [`ImageMode::Referenced`] additionally hands each picture's bytes out through
175/// [`take_artifacts`](Self::take_artifacts) — construct with
176/// [`with_artifacts`](Self::with_artifacts) and drain after every push so the
177/// bytes can be written to disk as pages finish instead of accumulating for the
178/// whole document (issue #80's memory-bounded image handling).
179///
180/// Each [`push`](Self::push) must contain whole blocks in reading order: a caller
181/// must not split a run of list items across two pushes (the run would render as
182/// two separate lists). Finalized PDF page batches already satisfy this.
183pub struct MarkdownStreamer {
184    strict: bool,
185    images: ImageMode,
186    compact_tables: bool,
187    /// Whether any non-empty chunk has been emitted yet (drives `\n\n` joins and
188    /// the trailing newline).
189    emitted_any: bool,
190    /// Recovered links not yet placed (strict mode), consumed in document order.
191    links: Vec<(String, String)>,
192    /// Referenced mode: the link prefix, the not-yet-drained `(path, bytes)`
193    /// artifacts, and the running image number (continues across pushes so the
194    /// stream matches the buffered serializer's `image_000000…` numbering).
195    artifacts_dir: String,
196    artifacts: Vec<(String, Vec<u8>)>,
197    pic_index: usize,
198}
199
200impl MarkdownStreamer {
201    /// Create a streamer. `compact_tables` mirrors [`DoclingDocument::compact_tables`].
202    /// For [`ImageMode::Referenced`] use [`with_artifacts`](Self::with_artifacts).
203    pub fn new(strict: bool, images: ImageMode, compact_tables: bool) -> Self {
204        debug_assert!(
205            images != ImageMode::Referenced,
206            "referenced image mode needs an artifacts dir; use with_artifacts"
207        );
208        Self::with_artifacts(strict, images, compact_tables, "artifacts")
209    }
210
211    /// Like [`new`](Self::new) but with the artifacts link prefix, allowing
212    /// [`ImageMode::Referenced`]: pictures render as
213    /// `![Image](<artifacts_dir>/image_NNNNNN.<ext>)` and each push's image
214    /// bytes wait in [`take_artifacts`](Self::take_artifacts) for the caller to
215    /// write. The concatenated chunks and the artifact list match the buffered
216    /// [`to_markdown_images`] byte-for-byte.
217    pub fn with_artifacts(
218        strict: bool,
219        images: ImageMode,
220        compact_tables: bool,
221        artifacts_dir: &str,
222    ) -> Self {
223        Self {
224            strict,
225            images,
226            compact_tables,
227            emitted_any: false,
228            links: Vec::new(),
229            artifacts_dir: artifacts_dir.to_string(),
230            artifacts: Vec::new(),
231            pic_index: 0,
232        }
233    }
234
235    /// The `(relative path, bytes)` of images rendered by pushes since the last
236    /// drain ([`ImageMode::Referenced`] only — empty otherwise). Paths are
237    /// relative to the Markdown file, i.e. they start with the configured
238    /// artifacts dir.
239    pub fn take_artifacts(&mut self) -> Vec<(String, Vec<u8>)> {
240        std::mem::take(&mut self.artifacts)
241    }
242
243    /// Render one finalized batch of nodes (plus any links recovered from the same
244    /// span, in document order) into the next Markdown chunk. Returns an empty
245    /// string when the batch produces no output (e.g. empty tables/pictures), in
246    /// which case nothing should be written.
247    pub fn push(&mut self, nodes: &[Node], links: &[(String, String)]) -> String {
248        self.links.extend(links.iter().cloned());
249        let mut ctx = Ctx {
250            strict: self.strict,
251            compact_tables: self.compact_tables,
252            images: self.images,
253            artifacts_dir: std::mem::take(&mut self.artifacts_dir),
254            artifacts: std::mem::take(&mut self.artifacts),
255            pic_index: self.pic_index,
256            in_table_cell: false,
257        };
258        let mut blocks: Vec<String> = Vec::new();
259        render(nodes, &mut blocks, &mut ctx);
260        self.artifacts_dir = std::mem::take(&mut ctx.artifacts_dir);
261        self.artifacts = std::mem::take(&mut ctx.artifacts);
262        self.pic_index = ctx.pic_index;
263        if blocks.is_empty() {
264            return String::new();
265        }
266        let mut body = blocks.join("\n\n");
267        if self.strict && !self.links.is_empty() {
268            body = apply_links_chunk(&body, &mut self.links);
269        }
270        let chunk = if self.emitted_any {
271            format!("\n\n{body}")
272        } else {
273            body
274        };
275        self.emitted_any = true;
276        chunk
277    }
278
279    /// Emit the trailing newline that finishes the document (empty if no content
280    /// was produced). Call exactly once, after the final [`push`](Self::push).
281    pub fn finish(self) -> String {
282        if self.emitted_any {
283            "\n".to_string()
284        } else {
285            String::new()
286        }
287    }
288}
289
290/// In `strict` mode, rewrite inline text for readability rather than byte-for-byte
291/// docling fidelity: undo the legacy `\_` underscore escaping, and tighten stray
292/// spaces around punctuation (`[ 37 , 36 ]` → `[37, 36]`, `( x )` → `(x)`). This
293/// cleans up both the PDF backend's glyph-split spacing and the space the legacy
294/// emphasis serialization leaves before punctuation (`*a* ,` → `*a*,`).
295/// Legacy/default output keeps docling's spacing untouched. Only inline text
296/// nodes pass through here — code blocks and table cells are left alone.
297fn strict_text(text: &str, strict: bool) -> String {
298    if !strict {
299        return text.to_string();
300    }
301    text.replace("\\_", "_")
302        .replace(" ,", ",")
303        .replace(" .", ".")
304        .replace(" ;", ";")
305        .replace(" )", ")")
306        .replace("( ", "(")
307        .replace(" ]", "]")
308        .replace("[ ", "[")
309}
310
311/// docling-core 2.92's `_md_line_breaks` (docling-core#721): a single `\n`
312/// inside an item's text becomes a GFM hard line break (`"  \n"`, two trailing
313/// spaces) so renderers honour it; a blank line (`\n\n`) is a paragraph break
314/// and stays as is — the document scope already joins blocks with `\n\n`.
315/// Applied to body text, list items and captions, never to code/formulas.
316fn md_line_breaks(text: &str) -> String {
317    if !text.contains('\n') {
318        return text.to_string();
319    }
320    text.split("\n\n")
321        .map(|para| para.replace('\n', "  \n"))
322        .collect::<Vec<_>>()
323        .join("\n\n")
324}
325
326/// Undo [`md_line_breaks`] on a rich table cell's flattened Markdown so the
327/// non-Markdown exports (JSON `text`, LaTeX cells) see the cell's raw line
328/// breaks, as docling's do — a rich cell's text is its Markdown serialization
329/// in our model, and the two trailing spaces are a Markdown-only marker.
330pub(crate) fn strip_hard_breaks(text: &str) -> String {
331    if text.contains("  \n") {
332        text.replace("  \n", "\n")
333    } else {
334        text.to_string()
335    }
336}
337
338/// docling-core's `_heading_line_breaks`: a GFM heading cannot span lines, so a
339/// newline inside heading text collapses to a space (`# Hello World`, not a
340/// broken `# Hello\nWorld`).
341fn heading_line_breaks(text: &str) -> String {
342    text.replace('\n', " ")
343}
344
345fn render(nodes: &[Node], blocks: &mut Vec<String>, ctx: &mut Ctx) {
346    let mut i = 0;
347    while i < nodes.len() {
348        match &nodes[i] {
349            Node::ListItem { .. } => {
350                let start = i;
351                i += 1;
352                loop {
353                    match nodes.get(i) {
354                        Some(Node::ListItem { .. }) => i += 1,
355                        // An empty paragraph between two list items is absorbed
356                        // into the run — docling keeps such a ListGroup
357                        // contiguous rather than splitting it.
358                        Some(Node::Paragraph { text })
359                            if text.is_empty()
360                                && matches!(nodes.get(i + 1), Some(Node::ListItem { .. })) =>
361                        {
362                            i += 1
363                        }
364                        _ => break,
365                    }
366                }
367                render_list_run(&nodes[start..i], blocks, ctx.strict);
368            }
369            other => {
370                render_one(other, blocks, ctx);
371                i += 1;
372            }
373        }
374    }
375}
376
377/// Render a contiguous run of list items.
378///
379/// Ordered items use their explicit `number`. A new sibling list (marked by
380/// `first_in_list`) at the same depth is separated by a blank line, matching
381/// docling-core's serializer.
382fn render_list_run(items: &[Node], blocks: &mut Vec<String>, strict: bool) {
383    let mut lines: Vec<String> = Vec::new();
384    // Per level, the previous item's (ordered, number) so we can detect a new
385    // sibling list.
386    let mut prev: Vec<Option<(bool, u64)>> = Vec::new();
387    // Whether the previous top-level item was a multilevel projection — an
388    // ordered `1.2.`-style item rendered as a Markdown bullet (docx's DocLang
389    // overlay says ordered, the flat field says bullet). Word numbers such an
390    // item and its parent-level successor within one list (same `numId`), and
391    // docling keeps them in one group — so the kind-flip / number-continuity
392    // breaks below must not fire across it (docling#3902's
393    // docx_list_blank_spacer: `- 1.2. Sub two` directly followed by
394    // `2. Second section`, no blank line).
395    let mut prev_projected = false;
396
397    for item in items {
398        let Node::ListItem {
399            ordered,
400            number,
401            first_in_list,
402            text,
403            level,
404            marker: _,
405            location: _,
406            dclx,
407            href: _,
408            layer,
409        } = item
410        else {
411            continue;
412        };
413        // A non-body (furniture) list item is omitted from Markdown, matching
414        // docling's content-layer filtering.
415        if layer.is_some() {
416            continue;
417        }
418        let level = *level as usize;
419
420        // Returning to a shallower level ends the deeper sibling lists.
421        prev.truncate(level + 1);
422        while prev.len() <= level {
423            prev.push(None);
424        }
425
426        // A new sibling list at the same depth gets a blank line: the kind flips
427        // (`<ul>`↔`<ol>`), an ordered run breaks (`1, 2` then `42`), or the
428        // backend flagged a fresh list (e.g. Markdown's bullet changing `-`→`*`).
429        // Only at the top level: nested sibling groups are children of a list
430        // item, and docling joins an item's children without blank lines.
431        let eff_ordered = dclx.as_ref().map_or(*ordered, |d| d.ordered);
432        if level == 0 {
433            if let Some((prev_ordered, prev_number)) = prev[level] {
434                // A projected predecessor suppresses both heuristics for an
435                // ordered successor: the flat kind flip is an artifact of the
436                // bullet projection, and the numbering continues the deeper
437                // sequence (`1.2.` → `2.`), not this level's.
438                let same_word_list = prev_projected && eff_ordered;
439                let new_list = *first_in_list
440                    || (!same_word_list
441                        && (prev_ordered != *ordered || (*ordered && *number != prev_number + 1)));
442                if new_list {
443                    lines.push(String::new());
444                }
445            }
446            prev_projected = eff_ordered && !*ordered;
447        }
448
449        let indent = "    ".repeat(level);
450        let marker = if *ordered {
451            format!("{number}.")
452        } else {
453            "-".to_string()
454        };
455        lines.push(format!(
456            "{indent}{marker} {}",
457            md_line_breaks(&strict_text(text, strict))
458        ));
459        prev[level] = Some((*ordered, *number));
460    }
461
462    // A run consisting only of furniture (content-layer-filtered) items yields no
463    // lines; pushing an empty block here would surface as a stray blank line.
464    if !lines.is_empty() {
465        blocks.push(lines.join("\n"));
466    }
467}
468
469fn render_one(node: &Node, blocks: &mut Vec<String>, ctx: &mut Ctx) {
470    match node {
471        Node::Heading { level, text } => {
472            let text = heading_line_breaks(&strict_text(text, ctx.strict));
473            if ctx.in_table_cell {
474                // docling-core#540: no `#` markers inside a table cell.
475                blocks.push(text);
476            } else {
477                let hashes = "#".repeat((*level).clamp(1, 6) as usize);
478                blocks.push(format!("{hashes} {text}"));
479            }
480        }
481        // An empty body paragraph (docling's blank-line text item) contributes
482        // nothing to Markdown — only DocLang/JSON keep it.
483        Node::Paragraph { text } if text.is_empty() => {}
484        Node::Paragraph { text } => blocks.push(md_line_breaks(&strict_text(text, ctx.strict))),
485        Node::CheckboxItem { checked, text } => {
486            let mark = if *checked { "- [x] " } else { "- [ ] " };
487            blocks.push(md_line_breaks(&strict_text(
488                &format!("{mark}{text}"),
489                ctx.strict,
490            )));
491        }
492        Node::Code {
493            language,
494            text,
495            pretty,
496            ..
497        } => {
498            // Legacy docling never emits a language on the fence; strict keeps it.
499            let lang = match language {
500                Some(l) if ctx.strict => l.as_str(),
501                _ => "",
502            };
503            // Strict prefers the line-preserving rendering when the backend
504            // supplied one (PDF); legacy stays on docling's flat `text`.
505            let body = match pretty {
506                Some(p) if ctx.strict => p.as_str(),
507                _ => text.as_str(),
508            };
509            blocks.push(format!("```{lang}\n{body}\n```"));
510        }
511        // A CodeFormula-decoded display formula renders as docling's `$$…$$`
512        // (the un-enriched pipeline emits a placeholder paragraph instead).
513        Node::Formula { latex, .. } => blocks.push(format!("$${latex}$$")),
514        Node::Table(table) => {
515            // docling renders a table's caption as a text line before the grid.
516            // `caption` is already escaped (backend convention), like a paragraph.
517            if let Some(cap) = &table.caption {
518                if !cap.is_empty() {
519                    blocks.push(md_line_breaks(&strict_text(cap, ctx.strict)));
520                }
521            }
522            let rendered = render_table(table, ctx.compact_tables);
523            if !rendered.is_empty() {
524                blocks.push(rendered);
525            }
526        }
527        // Classification predictions don't affect docling's Markdown output.
528        Node::Picture { caption, image, .. } => {
529            if let Some(cap) = caption {
530                if !cap.is_empty() {
531                    blocks.push(md_line_breaks(cap));
532                }
533            }
534            blocks.push(picture_marker(image.as_ref(), ctx));
535        }
536        // A chart renders as docling's picture-with-meta markdown: the caption,
537        // the placeholder, the humanized classification ("line_chart" ->
538        // "Line chart"), then the chart's data grid as a regular table.
539        Node::Chart {
540            kind,
541            table,
542            caption,
543            ..
544        } => {
545            if let Some(cap) = caption {
546                if !cap.is_empty() {
547                    blocks.push(md_line_breaks(cap));
548                }
549            }
550            blocks.push(picture_marker(None, ctx));
551            blocks.push(humanize_label(kind));
552            let rendered = render_table(table, false);
553            if !rendered.is_empty() {
554                blocks.push(rendered);
555            }
556        }
557        // A DocLang-only node is omitted from Markdown.
558        Node::DoclangOnly(_) => {}
559        Node::Group { children, .. } => render(children, blocks, ctx),
560        Node::FieldRegion { items } => {
561            // The region container and each field item carry no text of their
562            // own; docling-core 2.93 (#724) serializes them to nothing (older
563            // releases emitted a `<!-- missing-text -->` marker for each), so
564            // only an item's marker/key/value appear, as separate paragraphs.
565            for item in items {
566                for part in [&item.marker, &item.key, &item.value].into_iter().flatten() {
567                    blocks.push(md_line_breaks(&strict_text(part, ctx.strict)));
568                }
569            }
570        }
571        // A rich inline group renders exactly like a paragraph of its Markdown
572        // text — the structured runs are DocLang-only.
573        Node::InlineGroup { md_text, .. } => {
574            blocks.push(md_line_breaks(&strict_text(md_text, ctx.strict)))
575        }
576        // A plain-text backend dump renders verbatim as a single block.
577        Node::TextDump(text) => {
578            if !text.is_empty() {
579                blocks.push(text.clone());
580            }
581        }
582        // Furniture (page headers/footers, HTML `<title>`) is excluded from
583        // Markdown by default, mirroring docling.
584        Node::Furniture { .. } => {}
585        Node::PageFurniture { .. } => {}
586        // A comment lives in the notes layer — omitted like other furniture;
587        // the annotation on a body item is JSON-only, so render the item.
588        Node::CommentSection { .. } => {}
589        Node::Commented { inner, .. } => render_one(inner, blocks, ctx),
590        // Layout provenance is DocLang-only; render the wrapped node.
591        Node::Located { inner, .. } => render_one(inner, blocks, ctx),
592        // Page breaks are DocLang-only; docling omits them from Markdown.
593        Node::PageBreak => {}
594        // Page markers feed the JSON export only.
595        Node::PageInfo { .. } => {}
596        // Runs of adjacent list items are merged by `render`; a stray single
597        // item (a hand-built document, or a `Located` wrapper around one)
598        // still renders as its own one-item list instead of panicking —
599        // `nodes` is public API, so every representable tree must serialize.
600        Node::ListItem { .. } => render_list_run(std::slice::from_ref(node), blocks, ctx.strict),
601    }
602}
603
604/// The Markdown for a picture under the active [`ImageMode`]; Referenced mode also
605/// records the bytes in `ctx.artifacts` for the caller to write.
606/// docling-core's `_humanize_text`: underscores to spaces, first letter
607/// capitalized ("line_chart" -> "Line chart").
608fn humanize_label(label: &str) -> String {
609    let text = label.replace('_', " ");
610    let mut chars = text.chars();
611    match chars.next() {
612        Some(f) => f.to_uppercase().collect::<String>() + chars.as_str(),
613        None => text,
614    }
615}
616
617fn picture_marker(image: Option<&crate::PictureImage>, ctx: &mut Ctx) -> String {
618    match (ctx.images, image) {
619        (ImageMode::Embedded, Some(img)) => format!("![Image]({})", img.data_uri()),
620        (ImageMode::Referenced, Some(img)) => {
621            let path = format!(
622                "{}/image_{:06}.{}",
623                ctx.artifacts_dir,
624                ctx.pic_index,
625                ext_for(&img.mimetype)
626            );
627            ctx.pic_index += 1;
628            ctx.artifacts.push((path.clone(), img.data.clone()));
629            format!("![Image]({path})")
630        }
631        // Placeholder, or any mode with no extracted image.
632        _ => "<!-- image -->".to_string(),
633    }
634}
635
636fn ext_for(mimetype: &str) -> &str {
637    match mimetype {
638        "image/jpeg" => "jpg",
639        "image/gif" => "gif",
640        "image/webp" => "webp",
641        "image/bmp" => "bmp",
642        "image/tiff" => "tif",
643        _ => "png",
644    }
645}
646
647/// Render a table. `compact` selects between two serializers:
648///
649/// - **padded** (default) — docling-core's `tabulate(tablefmt="github")`: columns
650///   are padded to a fixed width (header width + a minimum padding of 2, or the
651///   widest data cell); numeric columns (every data cell parses as a number) are
652///   right-aligned, others left-aligned; separators are plain dashes of
653///   `width + 2`. Matches current published docling (DOCX/HTML conformance).
654/// - **compact** — `| a | b |` cells with single-dash `| - | - |` separators, no
655///   width padding. Matches the committed PDF groundtruth corpus, which predates
656///   the padded serializer.
657///
658/// Each cell is first escaped (`\n` → space, `|` → `&#124;`) so it can't break the
659/// table. Row 0 is the header.
660/// Whether a table cell counts as a number for column alignment, matching
661/// `tabulate`'s detection: an ordinary float/int (`f64`-parseable, covering
662/// `1e2`/`inf`/`+1.5`) **or** a thousands-separated number like `7,015`.
663fn is_number_cell(t: &str) -> bool {
664    t.parse::<f64>().is_ok() || is_thousands_number(t)
665}
666
667/// A number with comma thousands-separators, per `tabulate`'s
668/// `_float_with_thousands_separators` regex
669/// (`^(([+-]?[0-9]{1,3})(?:,([0-9]{3}))*)?(?(1)\.[0-9]*|\.[0-9]+)?$`): the
670/// integer part is 1–3 digits then any number of `,ddd` groups; the fraction is
671/// optional (and, without an integer part, must have at least one digit).
672fn is_thousands_number(t: &str) -> bool {
673    let b = t.as_bytes();
674    let mut i = 0;
675    let start = i;
676    if i < b.len() && (b[i] == b'+' || b[i] == b'-') {
677        i += 1;
678    }
679    // First digit chunk: 1–3 digits.
680    let d0 = i;
681    while i < b.len() && b[i].is_ascii_digit() && i - d0 < 3 {
682        i += 1;
683    }
684    let has_int = i > d0;
685    if has_int {
686        // Subsequent `,ddd` groups (exactly three digits each).
687        while i + 3 < b.len() + 1
688            && b.get(i) == Some(&b',')
689            && b.get(i + 1).is_some_and(u8::is_ascii_digit)
690            && b.get(i + 2).is_some_and(u8::is_ascii_digit)
691            && b.get(i + 3).is_some_and(u8::is_ascii_digit)
692        {
693            i += 4;
694        }
695    } else {
696        // A sign only counts with an integer part.
697        i = start;
698    }
699    // Optional fraction.
700    if i < b.len() && b[i] == b'.' {
701        i += 1;
702        let f0 = i;
703        while i < b.len() && b[i].is_ascii_digit() {
704            i += 1;
705        }
706        if !has_int && i == f0 {
707            return false; // `.` with no digits and no integer part
708        }
709    } else if !has_int {
710        return false; // neither integer nor fractional part
711    }
712    i == b.len()
713}
714
715pub(crate) fn render_table(table: &Table, compact: bool) -> String {
716    if table.rows.is_empty() {
717        return String::new();
718    }
719    let num_cols = table.rows.iter().map(Vec::len).max().unwrap_or(0);
720    if num_cols == 0 {
721        return String::new();
722    }
723
724    // Escaped, rectangular grid (ragged rows padded with empty cells). `tabulate`
725    // strips data cells of surrounding whitespace but leaves the header row as-is.
726    let grid: Vec<Vec<String>> = table
727        .rows
728        .iter()
729        .enumerate()
730        .map(|(r, row)| {
731            (0..num_cols)
732                .map(|c| {
733                    let cell = escape_cell(row.get(c).map(String::as_str).unwrap_or(""));
734                    if r == 0 {
735                        cell
736                    } else {
737                        cell.trim().to_string()
738                    }
739                })
740                .collect()
741        })
742        .collect();
743
744    if compact {
745        // Compact: cells joined by " | ", no padding, single-dash separators.
746        let render_row = |r: usize| -> String { format!("| {} |", grid[r].join(" | ")) };
747        let mut lines = Vec::with_capacity(grid.len() + 1);
748        lines.push(render_row(0));
749        let sep: Vec<&str> = (0..num_cols).map(|_| "-").collect();
750        lines.push(format!("| {} |", sep.join(" | ")));
751        for r in 1..grid.len() {
752            lines.push(render_row(r));
753        }
754        return lines.join("\n");
755    }
756
757    // Display width (Unicode scalar count — good enough for now).
758    let dw = |s: &str| s.chars().count();
759    let data_rows = 1..grid.len();
760
761    // A column is right-aligned when at least one data cell is numeric and every
762    // non-empty data cell is numeric — matching `tabulate`'s column typing, where
763    // empty cells are "missing" (ignored) and a number may carry thousands
764    // separators (`7,015`), which a plain `f64` parse rejects.
765    let right: Vec<bool> = (0..num_cols)
766        .map(|c| {
767            let mut any = false;
768            for r in data_rows.clone() {
769                let t = grid[r][c].trim();
770                if t.is_empty() {
771                    continue;
772                }
773                if !is_number_cell(t) {
774                    return false;
775                }
776                any = true;
777            }
778            any
779        })
780        .collect();
781
782    // Column width = max(header_width + MIN_PADDING(2), max data-cell width).
783    let width: Vec<usize> = (0..num_cols)
784        .map(|c| {
785            let mut w = dw(&grid[0][c]) + 2;
786            for r in data_rows.clone() {
787                w = w.max(dw(&grid[r][c]));
788            }
789            w
790        })
791        .collect();
792
793    let fmt_cell = |s: &str, c: usize| -> String {
794        let pad = " ".repeat(width[c].saturating_sub(dw(s)));
795        let body = if right[c] {
796            format!("{pad}{s}")
797        } else {
798            format!("{s}{pad}")
799        };
800        format!(" {body} ")
801    };
802    let render_row = |r: usize| -> String {
803        let cells: Vec<String> = (0..num_cols).map(|c| fmt_cell(&grid[r][c], c)).collect();
804        format!("|{}|", cells.join("|"))
805    };
806
807    let mut lines = Vec::with_capacity(grid.len() + 1);
808    lines.push(render_row(0));
809    let sep: Vec<String> = (0..num_cols).map(|c| "-".repeat(width[c] + 2)).collect();
810    lines.push(format!("|{}|", sep.join("|")));
811    for r in data_rows {
812        lines.push(render_row(r));
813    }
814    lines.join("\n")
815}
816
817/// Escape a table cell so it can't break the markdown table: newlines become
818/// spaces and pipes become the `&#124;` HTML entity (matches docling-core).
819fn escape_cell(s: &str) -> String {
820    s.replace('\n', " ").replace('|', "&#124;")
821}
822
823#[cfg(test)]
824mod tests {
825    use super::*;
826    use crate::PictureImage;
827
828    #[test]
829    fn renders_headings_paragraphs_and_lists() {
830        let mut doc = DoclingDocument::new("demo");
831        doc.add_heading(1, "Title");
832        doc.add_paragraph("Hello world.");
833        doc.push(Node::ListItem {
834            ordered: false,
835            number: 1,
836            first_in_list: true,
837            text: "first".into(),
838            level: 0,
839            marker: None,
840            location: None,
841            dclx: None,
842            href: None,
843            layer: None,
844        });
845        doc.push(Node::ListItem {
846            ordered: false,
847            number: 2,
848            first_in_list: false,
849            text: "second".into(),
850            level: 0,
851            marker: None,
852            location: None,
853            dclx: None,
854            href: None,
855            layer: None,
856        });
857        let md = doc.export_to_markdown();
858        assert_eq!(md, "# Title\n\nHello world.\n\n- first\n- second\n");
859    }
860
861    /// docling-core 2.92 (#721): a single newline inside an item's text is a
862    /// GFM hard line break, a blank line stays a paragraph break, and a heading
863    /// collapses its newline to a space. Nested-table dumps stay verbatim.
864    #[test]
865    fn single_newlines_become_gfm_hard_line_breaks() {
866        let mut doc = DoclingDocument::new("t");
867        doc.push(Node::Heading {
868            level: 1,
869            text: "Hello\nWorld".into(),
870        });
871        doc.push(Node::Paragraph {
872            text: "line one\nline two\n\npara two".into(),
873        });
874        doc.push(Node::ListItem {
875            ordered: false,
876            number: 1,
877            first_in_list: true,
878            text: "item\ncontinued".into(),
879            level: 0,
880            marker: None,
881            location: None,
882            dclx: None,
883            href: None,
884            layer: None,
885        });
886        doc.push(Node::TextDump("A1 B1 \n\n\nC1".into()));
887        assert_eq!(
888            doc.export_to_markdown(),
889            "# Hello World\n\nline one  \nline two\n\npara two\n\n- item  \ncontinued\n\nA1 B1 \n\n\nC1\n"
890        );
891    }
892
893    /// docling-core#540: inside a rich table cell a heading is plain text;
894    /// docling-core#724: a field region renders only its items' key/value text.
895    #[test]
896    fn table_cell_mode_and_field_regions() {
897        let mut doc = DoclingDocument::new("t");
898        doc.push(Node::Heading {
899            level: 2,
900            text: "A  text".into(),
901        });
902        doc.push(Node::Paragraph {
903            text: "body".into(),
904        });
905        assert_eq!(to_markdown_table_cell(&doc, false), "A  text\n\nbody");
906        assert_eq!(doc.export_to_markdown(), "## A  text\n\nbody\n");
907
908        let mut doc = DoclingDocument::new("f");
909        doc.push(Node::FieldRegion {
910            items: vec![crate::FieldItem {
911                marker: None,
912                key: Some("Name:".into()),
913                value: Some("John Doe".into()),
914            }],
915        });
916        assert_eq!(doc.export_to_markdown(), "Name:\n\nJohn Doe\n");
917    }
918
919    #[test]
920    fn strict_renders_recovered_links_legacy_does_not() {
921        let mut doc = DoclingDocument::new("cv");
922        doc.add_paragraph("Find me on LinkedIn or GitHub.");
923        doc.links = vec![
924            ("LinkedIn".into(), "https://www.linkedin.com/in/x/".into()),
925            ("GitHub".into(), "https://github.com/x/".into()),
926        ];
927        // Legacy/docling mode: links are left untouched (conformance preserved).
928        assert_eq!(doc.export_to_markdown(), "Find me on LinkedIn or GitHub.\n");
929        // Strict mode: anchors become Markdown links.
930        assert_eq!(
931            doc.export_to_markdown_with(true),
932            "Find me on [LinkedIn](https://www.linkedin.com/in/x/) or [GitHub](https://github.com/x/).\n"
933        );
934    }
935
936    #[test]
937    fn strict_links_match_escaped_anchor_and_consume_in_order() {
938        let mut doc = DoclingDocument::new("d");
939        // The PDF assembler HTML-escapes prose, so by serialization time the body
940        // already carries `&amp;`; the anchor is stored un-escaped. The matcher must
941        // escape the anchor to find it. Two identical anchors link in document order.
942        doc.add_paragraph("AI &amp; ML here, and issues here, then issues there.");
943        doc.links = vec![
944            ("AI & ML".into(), "https://a/".into()),
945            ("issues".into(), "https://first/".into()),
946            ("issues".into(), "https://second/".into()),
947        ];
948        assert_eq!(
949            doc.export_to_markdown_with(true),
950            "[AI &amp; ML](https://a/) here, and [issues](https://first/) here, then [issues](https://second/) there.\n"
951        );
952    }
953
954    #[test]
955    fn renders_compact_table() {
956        let mut doc = DoclingDocument::new("t");
957        // The compact form is opt-in (the PDF backend sets it); default output uses
958        // the padded GitHub serializer (covered by the regression fixtures).
959        doc.compact_tables = true;
960        doc.push(Node::Table(Table {
961            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
962            location: None,
963            structure: None,
964            cell_blocks: None,
965            cells: None,
966            caption: None,
967        }));
968        let md = doc.export_to_markdown();
969        assert_eq!(md, "| a | b |\n| - | - |\n| 1 | 2 |\n");
970    }
971
972    #[test]
973    fn renders_padded_github_table_by_default() {
974        let mut doc = DoclingDocument::new("t");
975        doc.push(Node::Table(Table {
976            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
977            location: None,
978            structure: None,
979            cell_blocks: None,
980            cells: None,
981            caption: None,
982        }));
983        let md = doc.export_to_markdown();
984        // Numeric data columns are right-aligned; columns padded to header+2.
985        assert_eq!(md, "|   a |   b |\n|-----|-----|\n|   1 |   2 |\n");
986    }
987
988    #[test]
989    fn strict_unescapes_inline_underscores_legacy_keeps_them() {
990        let mut doc = DoclingDocument::new("t");
991        doc.add_heading(1, "a\\_b");
992        doc.add_paragraph("x\\_y");
993        doc.push(Node::ListItem {
994            ordered: false,
995            number: 1,
996            first_in_list: true,
997            text: "i\\_j".into(),
998            level: 0,
999            marker: None,
1000            location: None,
1001            dclx: None,
1002            href: None,
1003            layer: None,
1004        });
1005        // Legacy reproduces docling's `\_` escaping byte-for-byte.
1006        assert_eq!(doc.export_to_markdown(), "# a\\_b\n\nx\\_y\n\n- i\\_j\n");
1007        // Strict prefers literal underscores (Rust-only readability mode).
1008        assert_eq!(doc.export_to_markdown_with(true), "# a_b\n\nx_y\n\n- i_j\n");
1009    }
1010
1011    /// Drive a document's nodes through [`MarkdownStreamer`] in the given page
1012    /// splits and assert the concatenated chunks equal the buffered serializer.
1013    fn assert_stream_matches(
1014        doc: &DoclingDocument,
1015        strict: bool,
1016        images: ImageMode,
1017        splits: &[usize],
1018    ) {
1019        let (want, want_artifacts) = to_markdown_images(doc, strict, images, "artifacts");
1020        let mut streamer =
1021            MarkdownStreamer::with_artifacts(strict, images, doc.compact_tables, "artifacts");
1022        let mut got = String::new();
1023        let mut got_artifacts = Vec::new();
1024        let mut start = 0;
1025        for &end in splits {
1026            // Links only matter in strict mode; feed them all with the first batch
1027            // that has content (document order is preserved by the queue).
1028            let links = if start == 0 {
1029                doc.links.as_slice()
1030            } else {
1031                &[]
1032            };
1033            got.push_str(&streamer.push(&doc.nodes[start..end], links));
1034            // Referenced mode: drain per push, as a real caller writing files
1035            // page by page would — numbering must continue across drains.
1036            got_artifacts.extend(streamer.take_artifacts());
1037            start = end;
1038        }
1039        got.push_str(&streamer.push(
1040            &doc.nodes[start..],
1041            if start == 0 {
1042                doc.links.as_slice()
1043            } else {
1044                &[]
1045            },
1046        ));
1047        got_artifacts.extend(streamer.take_artifacts());
1048        got.push_str(&streamer.finish());
1049        assert_eq!(
1050            got, want,
1051            "streamed output diverged (splits={splits:?}, strict={strict})"
1052        );
1053        assert_eq!(
1054            got_artifacts, want_artifacts,
1055            "streamed artifacts diverged (splits={splits:?}, strict={strict})"
1056        );
1057    }
1058
1059    #[test]
1060    fn streaming_is_byte_identical_to_buffered() {
1061        let mut doc = DoclingDocument::new("d");
1062        doc.add_heading(1, "Title");
1063        doc.add_paragraph("First paragraph.");
1064        doc.push(Node::ListItem {
1065            ordered: false,
1066            number: 1,
1067            first_in_list: true,
1068            text: "a".into(),
1069            level: 0,
1070            marker: None,
1071            location: None,
1072            dclx: None,
1073            href: None,
1074            layer: None,
1075        });
1076        doc.push(Node::ListItem {
1077            ordered: false,
1078            number: 2,
1079            first_in_list: false,
1080            text: "b".into(),
1081            level: 0,
1082            marker: None,
1083            location: None,
1084            dclx: None,
1085            href: None,
1086            layer: None,
1087        });
1088        doc.push(Node::Code {
1089            language: Some("rust".into()),
1090            text: "let x = 1;".into(),
1091            orig: None,
1092            pretty: None,
1093        });
1094        doc.push(Node::Table(Table {
1095            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
1096            location: None,
1097            structure: None,
1098            cell_blocks: None,
1099            cells: None,
1100            caption: None,
1101        }));
1102        doc.push(Node::Picture {
1103            caption: Some("Fig 1".into()),
1104            caption_href: None,
1105            image: Some(PictureImage {
1106                mimetype: "image/png".into(),
1107                width: 2,
1108                height: 2,
1109                data: b"png-one".to_vec(),
1110            }),
1111            classification: None,
1112        });
1113        doc.add_paragraph("Last paragraph.");
1114        // A second embedded picture, so referenced mode must keep numbering
1115        // (`image_000001`) across chunk boundaries.
1116        doc.push(Node::Picture {
1117            caption: None,
1118            caption_href: None,
1119            image: Some(PictureImage {
1120                mimetype: "image/png".into(),
1121                width: 2,
1122                height: 2,
1123                data: b"png-two".to_vec(),
1124            }),
1125            classification: None,
1126        });
1127
1128        // A run of list items must never straddle a split, so try splits that fall
1129        // on safe block boundaries (the streaming PDF assembler guarantees this).
1130        for &strict in &[false, true] {
1131            for &images in &[
1132                ImageMode::Placeholder,
1133                ImageMode::Embedded,
1134                ImageMode::Referenced,
1135            ] {
1136                for splits in [&[][..], &[1][..], &[2][..], &[4][..], &[1, 4, 6, 7][..]] {
1137                    assert_stream_matches(&doc, strict, images, splits);
1138                }
1139            }
1140        }
1141    }
1142
1143    #[test]
1144    fn streaming_applies_recovered_links_in_strict_mode() {
1145        let mut doc = DoclingDocument::new("d");
1146        doc.add_paragraph("See LinkedIn for details.");
1147        doc.add_paragraph("And GitHub too.");
1148        doc.links = vec![
1149            ("LinkedIn".into(), "https://lnkd/".into()),
1150            ("GitHub".into(), "https://gh/".into()),
1151        ];
1152        // The second anchor lives in the second block, so it must be carried across
1153        // the page boundary and placed when that block streams out.
1154        assert_stream_matches(&doc, true, ImageMode::Placeholder, &[1]);
1155    }
1156
1157    #[test]
1158    fn strict_tightens_punctuation_spacing_legacy_keeps_it() {
1159        let mut doc = DoclingDocument::new("t");
1160        doc.add_paragraph("see [ 37 , 36 ] and ( x ) .");
1161        // Legacy keeps docling's spacing byte-for-byte.
1162        assert_eq!(doc.export_to_markdown(), "see [ 37 , 36 ] and ( x ) .\n");
1163        // Strict tightens punctuation for readable Markdown.
1164        assert_eq!(doc.export_to_markdown_with(true), "see [37, 36] and (x).\n");
1165    }
1166}