Skip to main content

docling_core/
markdown.rs

1//! Markdown serializer for [`DoclingDocument`].
2
3use crate::document::{DoclingDocument, Node, Table};
4
5/// How pictures are rendered (mirrors docling-core's `ImageRefMode`).
6#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
7pub enum ImageMode {
8    /// `<!-- image -->` (docling's default, and the only mode without image data).
9    #[default]
10    Placeholder,
11    /// `![Image](data:<mime>;base64,…)` — self-contained.
12    Embedded,
13    /// `![Image](<artifacts>/image_NNNNNN.<ext>)`; the bytes are returned for the
14    /// caller to write.
15    Referenced,
16}
17
18/// Serializer state threaded through the render walk.
19struct Ctx {
20    strict: bool,
21    /// Emit compact `| a | b |` tables instead of the padded GitHub serializer.
22    compact_tables: bool,
23    images: ImageMode,
24    artifacts_dir: String,
25    /// (relative path, bytes) for each referenced image — written by the caller.
26    artifacts: Vec<(String, Vec<u8>)>,
27    pic_index: usize,
28}
29
30/// Render a document to a Markdown string (pictures as placeholders).
31///
32/// `strict` selects the serializer-level behaviours that differ between
33/// docling-legacy output and cleaner Markdown — currently the code-fence
34/// language (legacy drops it, strict keeps it).
35pub fn to_markdown(doc: &DoclingDocument, strict: bool) -> String {
36    to_markdown_images(doc, strict, ImageMode::Placeholder, "artifacts").0
37}
38
39/// Render to Markdown with an explicit picture [`ImageMode`]. Returns the
40/// Markdown and, for [`ImageMode::Referenced`], the `(path, bytes)` of each image
41/// the caller should write (relative to the Markdown file).
42pub fn to_markdown_images(
43    doc: &DoclingDocument,
44    strict: bool,
45    images: ImageMode,
46    artifacts_dir: &str,
47) -> (String, Vec<(String, Vec<u8>)>) {
48    let mut ctx = Ctx {
49        strict,
50        compact_tables: doc.compact_tables,
51        images,
52        artifacts_dir: artifacts_dir.to_string(),
53        artifacts: Vec::new(),
54        pic_index: 0,
55    };
56    let mut blocks: Vec<String> = Vec::new();
57    render(&doc.nodes, &mut blocks, &mut ctx);
58    let mut body = blocks.join("\n\n");
59    // Strict mode only: turn recovered source hyperlinks into Markdown links.
60    // docling's standard pipeline drops them, so doing this in legacy mode would
61    // diverge from docling — hence strict-only, leaving conformance output intact.
62    if strict && !doc.links.is_empty() {
63        body = apply_links(&body, &doc.links);
64    }
65    let md = if body.is_empty() {
66        String::new()
67    } else {
68        format!("{body}\n")
69    };
70    (md, ctx.artifacts)
71}
72
73/// Wrap each recovered link's anchor text in Markdown `[anchor](href)`. Anchors
74/// arrive cleaned (curly quotes/dashes already normalized) but un-escaped, so we
75/// match against the body's HTML-escaped (`&`/`<`/`>`) form, the way prose nodes
76/// were serialized. Links are consumed in document order from a moving cursor, so
77/// a repeated anchor (e.g. two "issues") links its successive occurrences rather
78/// than all pointing at the first. An anchor that can't be located is skipped
79/// (its text may have been split across a line wrap or table cell).
80fn apply_links(body: &str, links: &[(String, String)]) -> String {
81    let mut out = body.to_string();
82    let mut cursor = 0usize;
83    for (anchor, href) in links {
84        let anchor = anchor
85            .replace('&', "&amp;")
86            .replace('<', "&lt;")
87            .replace('>', "&gt;");
88        if anchor.is_empty() {
89            continue;
90        }
91        if let Some(rel) = out[cursor..].find(&anchor) {
92            let at = cursor + rel;
93            // Don't relink inside an already-emitted `](` Markdown link target.
94            let replacement = format!("[{anchor}]({href})");
95            out.replace_range(at..at + anchor.len(), &replacement);
96            cursor = at + replacement.len();
97        }
98    }
99    out
100}
101
102/// Like [`apply_links`] but over a single chunk, consuming from a shared queue so
103/// the same `[anchor](href)` rewriting can be applied incrementally as Markdown is
104/// streamed out. Each queued link is matched (in document order) against `chunk`
105/// and rewritten in place; a link whose anchor is not in this chunk is carried
106/// forward in the queue for a later chunk. Anchors are recovered in document
107/// order and a chunk is always a contiguous run of whole blocks, so this
108/// reproduces [`apply_links`]' single moving cursor: the link lands in whichever
109/// chunk contains its anchor, identically to the buffered path. (A link whose
110/// anchor never appears is carried to the end and dropped — the same no-op
111/// `apply_links` performs for an unlocatable anchor.)
112fn apply_links_chunk(chunk: &str, queue: &mut Vec<(String, String)>) -> String {
113    let mut out = chunk.to_string();
114    let mut cursor = 0usize;
115    let mut carried: Vec<(String, String)> = Vec::new();
116    for (anchor_raw, href) in std::mem::take(queue) {
117        let anchor = anchor_raw
118            .replace('&', "&amp;")
119            .replace('<', "&lt;")
120            .replace('>', "&gt;");
121        if anchor.is_empty() {
122            continue;
123        }
124        if let Some(rel) = out[cursor..].find(&anchor) {
125            let at = cursor + rel;
126            let replacement = format!("[{anchor}]({href})");
127            out.replace_range(at..at + anchor.len(), &replacement);
128            cursor = at + replacement.len();
129        } else {
130            // Not in this chunk; try again when its block is flushed.
131            carried.push((anchor_raw, href));
132        }
133    }
134    *queue = carried;
135    out
136}
137
138/// Incremental Markdown serializer: feed finalized, in-document-order batches of
139/// [`Node`]s and receive Markdown chunks whose concatenation is **byte-identical**
140/// to [`to_markdown_images`] over the same nodes. This is the streaming
141/// counterpart of the buffered serializer — used to emit a document's Markdown in
142/// chunks (e.g. page by page, as the parallel PDF pipeline finishes pages) instead
143/// of building the whole string up front.
144///
145/// [`ImageMode::Placeholder`] and [`ImageMode::Embedded`] render inline.
146/// [`ImageMode::Referenced`] additionally hands each picture's bytes out through
147/// [`take_artifacts`](Self::take_artifacts) — construct with
148/// [`with_artifacts`](Self::with_artifacts) and drain after every push so the
149/// bytes can be written to disk as pages finish instead of accumulating for the
150/// whole document (issue #80's memory-bounded image handling).
151///
152/// Each [`push`](Self::push) must contain whole blocks in reading order: a caller
153/// must not split a run of list items across two pushes (the run would render as
154/// two separate lists). Finalized PDF page batches already satisfy this.
155pub struct MarkdownStreamer {
156    strict: bool,
157    images: ImageMode,
158    compact_tables: bool,
159    /// Whether any non-empty chunk has been emitted yet (drives `\n\n` joins and
160    /// the trailing newline).
161    emitted_any: bool,
162    /// Recovered links not yet placed (strict mode), consumed in document order.
163    links: Vec<(String, String)>,
164    /// Referenced mode: the link prefix, the not-yet-drained `(path, bytes)`
165    /// artifacts, and the running image number (continues across pushes so the
166    /// stream matches the buffered serializer's `image_000000…` numbering).
167    artifacts_dir: String,
168    artifacts: Vec<(String, Vec<u8>)>,
169    pic_index: usize,
170}
171
172impl MarkdownStreamer {
173    /// Create a streamer. `compact_tables` mirrors [`DoclingDocument::compact_tables`].
174    /// For [`ImageMode::Referenced`] use [`with_artifacts`](Self::with_artifacts).
175    pub fn new(strict: bool, images: ImageMode, compact_tables: bool) -> Self {
176        debug_assert!(
177            images != ImageMode::Referenced,
178            "referenced image mode needs an artifacts dir; use with_artifacts"
179        );
180        Self::with_artifacts(strict, images, compact_tables, "artifacts")
181    }
182
183    /// Like [`new`](Self::new) but with the artifacts link prefix, allowing
184    /// [`ImageMode::Referenced`]: pictures render as
185    /// `![Image](<artifacts_dir>/image_NNNNNN.<ext>)` and each push's image
186    /// bytes wait in [`take_artifacts`](Self::take_artifacts) for the caller to
187    /// write. The concatenated chunks and the artifact list match the buffered
188    /// [`to_markdown_images`] byte-for-byte.
189    pub fn with_artifacts(
190        strict: bool,
191        images: ImageMode,
192        compact_tables: bool,
193        artifacts_dir: &str,
194    ) -> Self {
195        Self {
196            strict,
197            images,
198            compact_tables,
199            emitted_any: false,
200            links: Vec::new(),
201            artifacts_dir: artifacts_dir.to_string(),
202            artifacts: Vec::new(),
203            pic_index: 0,
204        }
205    }
206
207    /// The `(relative path, bytes)` of images rendered by pushes since the last
208    /// drain ([`ImageMode::Referenced`] only — empty otherwise). Paths are
209    /// relative to the Markdown file, i.e. they start with the configured
210    /// artifacts dir.
211    pub fn take_artifacts(&mut self) -> Vec<(String, Vec<u8>)> {
212        std::mem::take(&mut self.artifacts)
213    }
214
215    /// Render one finalized batch of nodes (plus any links recovered from the same
216    /// span, in document order) into the next Markdown chunk. Returns an empty
217    /// string when the batch produces no output (e.g. empty tables/pictures), in
218    /// which case nothing should be written.
219    pub fn push(&mut self, nodes: &[Node], links: &[(String, String)]) -> String {
220        self.links.extend(links.iter().cloned());
221        let mut ctx = Ctx {
222            strict: self.strict,
223            compact_tables: self.compact_tables,
224            images: self.images,
225            artifacts_dir: std::mem::take(&mut self.artifacts_dir),
226            artifacts: std::mem::take(&mut self.artifacts),
227            pic_index: self.pic_index,
228        };
229        let mut blocks: Vec<String> = Vec::new();
230        render(nodes, &mut blocks, &mut ctx);
231        self.artifacts_dir = std::mem::take(&mut ctx.artifacts_dir);
232        self.artifacts = std::mem::take(&mut ctx.artifacts);
233        self.pic_index = ctx.pic_index;
234        if blocks.is_empty() {
235            return String::new();
236        }
237        let mut body = blocks.join("\n\n");
238        if self.strict && !self.links.is_empty() {
239            body = apply_links_chunk(&body, &mut self.links);
240        }
241        let chunk = if self.emitted_any {
242            format!("\n\n{body}")
243        } else {
244            body
245        };
246        self.emitted_any = true;
247        chunk
248    }
249
250    /// Emit the trailing newline that finishes the document (empty if no content
251    /// was produced). Call exactly once, after the final [`push`](Self::push).
252    pub fn finish(self) -> String {
253        if self.emitted_any {
254            "\n".to_string()
255        } else {
256            String::new()
257        }
258    }
259}
260
261/// In `strict` mode, rewrite inline text for readability rather than byte-for-byte
262/// docling fidelity: undo the legacy `\_` underscore escaping, and tighten stray
263/// spaces around punctuation (`[ 37 , 36 ]` → `[37, 36]`, `( x )` → `(x)`). This
264/// cleans up both the PDF backend's glyph-split spacing and the space the legacy
265/// emphasis serialization leaves before punctuation (`*a* ,` → `*a*,`).
266/// Legacy/default output keeps docling's spacing untouched. Only inline text
267/// nodes pass through here — code blocks and table cells are left alone.
268fn strict_text(text: &str, strict: bool) -> String {
269    if !strict {
270        return text.to_string();
271    }
272    text.replace("\\_", "_")
273        .replace(" ,", ",")
274        .replace(" .", ".")
275        .replace(" ;", ";")
276        .replace(" )", ")")
277        .replace("( ", "(")
278        .replace(" ]", "]")
279        .replace("[ ", "[")
280}
281
282fn render(nodes: &[Node], blocks: &mut Vec<String>, ctx: &mut Ctx) {
283    let mut i = 0;
284    while i < nodes.len() {
285        match &nodes[i] {
286            Node::ListItem { .. } => {
287                let start = i;
288                i += 1;
289                loop {
290                    match nodes.get(i) {
291                        Some(Node::ListItem { .. }) => i += 1,
292                        // An empty paragraph between two list items is absorbed
293                        // into the run — docling keeps such a ListGroup
294                        // contiguous rather than splitting it.
295                        Some(Node::Paragraph { text })
296                            if text.is_empty()
297                                && matches!(nodes.get(i + 1), Some(Node::ListItem { .. })) =>
298                        {
299                            i += 1
300                        }
301                        _ => break,
302                    }
303                }
304                render_list_run(&nodes[start..i], blocks, ctx.strict);
305            }
306            other => {
307                render_one(other, blocks, ctx);
308                i += 1;
309            }
310        }
311    }
312}
313
314/// Render a contiguous run of list items.
315///
316/// Ordered items use their explicit `number`. A new sibling list (marked by
317/// `first_in_list`) at the same depth is separated by a blank line, matching
318/// docling-core's serializer.
319fn render_list_run(items: &[Node], blocks: &mut Vec<String>, strict: bool) {
320    let mut lines: Vec<String> = Vec::new();
321    // Per level, the previous item's (ordered, number) so we can detect a new
322    // sibling list.
323    let mut prev: Vec<Option<(bool, u64)>> = Vec::new();
324
325    for item in items {
326        let Node::ListItem {
327            ordered,
328            number,
329            first_in_list,
330            text,
331            level,
332            marker: _,
333            location: _,
334            dclx: _,
335            href: _,
336            layer,
337        } = item
338        else {
339            continue;
340        };
341        // A non-body (furniture) list item is omitted from Markdown, matching
342        // docling's content-layer filtering.
343        if layer.is_some() {
344            continue;
345        }
346        let level = *level as usize;
347
348        // Returning to a shallower level ends the deeper sibling lists.
349        prev.truncate(level + 1);
350        while prev.len() <= level {
351            prev.push(None);
352        }
353
354        // A new sibling list at the same depth gets a blank line: the kind flips
355        // (`<ul>`↔`<ol>`), an ordered run breaks (`1, 2` then `42`), or the
356        // backend flagged a fresh list (e.g. Markdown's bullet changing `-`→`*`).
357        // Only at the top level: nested sibling groups are children of a list
358        // item, and docling joins an item's children without blank lines.
359        if level == 0 {
360            if let Some((prev_ordered, prev_number)) = prev[level] {
361                let new_list = *first_in_list
362                    || prev_ordered != *ordered
363                    || (*ordered && *number != prev_number + 1);
364                if new_list {
365                    lines.push(String::new());
366                }
367            }
368        }
369
370        let indent = "    ".repeat(level);
371        let marker = if *ordered {
372            format!("{number}.")
373        } else {
374            "-".to_string()
375        };
376        lines.push(format!("{indent}{marker} {}", strict_text(text, strict)));
377        prev[level] = Some((*ordered, *number));
378    }
379
380    // A run consisting only of furniture (content-layer-filtered) items yields no
381    // lines; pushing an empty block here would surface as a stray blank line.
382    if !lines.is_empty() {
383        blocks.push(lines.join("\n"));
384    }
385}
386
387fn render_one(node: &Node, blocks: &mut Vec<String>, ctx: &mut Ctx) {
388    match node {
389        Node::Heading { level, text } => {
390            let hashes = "#".repeat((*level).clamp(1, 6) as usize);
391            blocks.push(format!("{hashes} {}", strict_text(text, ctx.strict)));
392        }
393        // An empty body paragraph (docling's blank-line text item) contributes
394        // nothing to Markdown — only DocLang/JSON keep it.
395        Node::Paragraph { text } if text.is_empty() => {}
396        Node::Paragraph { text } => blocks.push(strict_text(text, ctx.strict)),
397        Node::CheckboxItem { checked, text } => {
398            let mark = if *checked { "- [x] " } else { "- [ ] " };
399            blocks.push(strict_text(&format!("{mark}{text}"), ctx.strict));
400        }
401        Node::Code {
402            language,
403            text,
404            pretty,
405            ..
406        } => {
407            // Legacy docling never emits a language on the fence; strict keeps it.
408            let lang = match language {
409                Some(l) if ctx.strict => l.as_str(),
410                _ => "",
411            };
412            // Strict prefers the line-preserving rendering when the backend
413            // supplied one (PDF); legacy stays on docling's flat `text`.
414            let body = match pretty {
415                Some(p) if ctx.strict => p.as_str(),
416                _ => text.as_str(),
417            };
418            blocks.push(format!("```{lang}\n{body}\n```"));
419        }
420        // A CodeFormula-decoded display formula renders as docling's `$$…$$`
421        // (the un-enriched pipeline emits a placeholder paragraph instead).
422        Node::Formula { latex, .. } => blocks.push(format!("$${latex}$$")),
423        Node::Table(table) => {
424            // docling renders a table's caption as a text line before the grid.
425            // `caption` is already escaped (backend convention), like a paragraph.
426            if let Some(cap) = &table.caption {
427                if !cap.is_empty() {
428                    blocks.push(strict_text(cap, ctx.strict));
429                }
430            }
431            let rendered = render_table(table, ctx.compact_tables);
432            if !rendered.is_empty() {
433                blocks.push(rendered);
434            }
435        }
436        // Classification predictions don't affect docling's Markdown output.
437        Node::Picture { caption, image, .. } => {
438            if let Some(cap) = caption {
439                if !cap.is_empty() {
440                    blocks.push(cap.clone());
441                }
442            }
443            blocks.push(picture_marker(image.as_ref(), ctx));
444        }
445        // A chart renders as docling's picture-with-meta markdown: the caption,
446        // the placeholder, the humanized classification ("line_chart" ->
447        // "Line chart"), then the chart's data grid as a regular table.
448        Node::Chart {
449            kind,
450            table,
451            caption,
452            ..
453        } => {
454            if let Some(cap) = caption {
455                if !cap.is_empty() {
456                    blocks.push(cap.clone());
457                }
458            }
459            blocks.push(picture_marker(None, ctx));
460            blocks.push(humanize_label(kind));
461            let rendered = render_table(table, false);
462            if !rendered.is_empty() {
463                blocks.push(rendered);
464            }
465        }
466        // A DocLang-only node is omitted from Markdown.
467        Node::DoclangOnly(_) => {}
468        Node::Group { children, .. } => render(children, blocks, ctx),
469        Node::FieldRegion { items } => {
470            // docling renders the region container (which carries no text of its
471            // own) as a `<!-- missing-text -->` marker, then each field item the
472            // same way, followed by that item's marker/key/value as separate
473            // paragraphs.
474            blocks.push(MISSING_TEXT.to_string());
475            for item in items {
476                blocks.push(MISSING_TEXT.to_string());
477                for part in [&item.marker, &item.key, &item.value].into_iter().flatten() {
478                    blocks.push(strict_text(part, ctx.strict));
479                }
480            }
481        }
482        // A rich inline group renders exactly like a paragraph of its Markdown
483        // text — the structured runs are DocLang-only.
484        Node::InlineGroup { md_text, .. } => blocks.push(strict_text(md_text, ctx.strict)),
485        // A plain-text backend dump renders verbatim as a single block.
486        Node::TextDump(text) => {
487            if !text.is_empty() {
488                blocks.push(text.clone());
489            }
490        }
491        // Furniture (page headers/footers, HTML `<title>`) is excluded from
492        // Markdown by default, mirroring docling.
493        Node::Furniture { .. } => {}
494        Node::PageFurniture { .. } => {}
495        // Layout provenance is DocLang-only; render the wrapped node.
496        Node::Located { inner, .. } => render_one(inner, blocks, ctx),
497        // Page breaks are DocLang-only; docling omits them from Markdown.
498        Node::PageBreak => {}
499        // Page markers feed the JSON export only.
500        Node::PageInfo { .. } => {}
501        // Runs of adjacent list items are merged by `render`; a stray single
502        // item (a hand-built document, or a `Located` wrapper around one)
503        // still renders as its own one-item list instead of panicking —
504        // `nodes` is public API, so every representable tree must serialize.
505        Node::ListItem { .. } => render_list_run(std::slice::from_ref(node), blocks, ctx.strict),
506    }
507}
508
509/// docling's placeholder for a structural node (a field region / item) that has
510/// no text of its own.
511const MISSING_TEXT: &str = "<!-- missing-text -->";
512
513/// The Markdown for a picture under the active [`ImageMode`]; Referenced mode also
514/// records the bytes in `ctx.artifacts` for the caller to write.
515/// docling-core's `_humanize_text`: underscores to spaces, first letter
516/// capitalized ("line_chart" -> "Line chart").
517fn humanize_label(label: &str) -> String {
518    let text = label.replace('_', " ");
519    let mut chars = text.chars();
520    match chars.next() {
521        Some(f) => f.to_uppercase().collect::<String>() + chars.as_str(),
522        None => text,
523    }
524}
525
526fn picture_marker(image: Option<&crate::PictureImage>, ctx: &mut Ctx) -> String {
527    match (ctx.images, image) {
528        (ImageMode::Embedded, Some(img)) => format!("![Image]({})", img.data_uri()),
529        (ImageMode::Referenced, Some(img)) => {
530            let path = format!(
531                "{}/image_{:06}.{}",
532                ctx.artifacts_dir,
533                ctx.pic_index,
534                ext_for(&img.mimetype)
535            );
536            ctx.pic_index += 1;
537            ctx.artifacts.push((path.clone(), img.data.clone()));
538            format!("![Image]({path})")
539        }
540        // Placeholder, or any mode with no extracted image.
541        _ => "<!-- image -->".to_string(),
542    }
543}
544
545fn ext_for(mimetype: &str) -> &str {
546    match mimetype {
547        "image/jpeg" => "jpg",
548        "image/gif" => "gif",
549        "image/webp" => "webp",
550        "image/bmp" => "bmp",
551        "image/tiff" => "tif",
552        _ => "png",
553    }
554}
555
556/// Render a table. `compact` selects between two serializers:
557///
558/// - **padded** (default) — docling-core's `tabulate(tablefmt="github")`: columns
559///   are padded to a fixed width (header width + a minimum padding of 2, or the
560///   widest data cell); numeric columns (every data cell parses as a number) are
561///   right-aligned, others left-aligned; separators are plain dashes of
562///   `width + 2`. Matches current published docling (DOCX/HTML conformance).
563/// - **compact** — `| a | b |` cells with single-dash `| - | - |` separators, no
564///   width padding. Matches the committed PDF groundtruth corpus, which predates
565///   the padded serializer.
566///
567/// Each cell is first escaped (`\n` → space, `|` → `&#124;`) so it can't break the
568/// table. Row 0 is the header.
569/// Whether a table cell counts as a number for column alignment, matching
570/// `tabulate`'s detection: an ordinary float/int (`f64`-parseable, covering
571/// `1e2`/`inf`/`+1.5`) **or** a thousands-separated number like `7,015`.
572fn is_number_cell(t: &str) -> bool {
573    t.parse::<f64>().is_ok() || is_thousands_number(t)
574}
575
576/// A number with comma thousands-separators, per `tabulate`'s
577/// `_float_with_thousands_separators` regex
578/// (`^(([+-]?[0-9]{1,3})(?:,([0-9]{3}))*)?(?(1)\.[0-9]*|\.[0-9]+)?$`): the
579/// integer part is 1–3 digits then any number of `,ddd` groups; the fraction is
580/// optional (and, without an integer part, must have at least one digit).
581fn is_thousands_number(t: &str) -> bool {
582    let b = t.as_bytes();
583    let mut i = 0;
584    let start = i;
585    if i < b.len() && (b[i] == b'+' || b[i] == b'-') {
586        i += 1;
587    }
588    // First digit chunk: 1–3 digits.
589    let d0 = i;
590    while i < b.len() && b[i].is_ascii_digit() && i - d0 < 3 {
591        i += 1;
592    }
593    let has_int = i > d0;
594    if has_int {
595        // Subsequent `,ddd` groups (exactly three digits each).
596        while i + 3 < b.len() + 1
597            && b.get(i) == Some(&b',')
598            && b.get(i + 1).is_some_and(u8::is_ascii_digit)
599            && b.get(i + 2).is_some_and(u8::is_ascii_digit)
600            && b.get(i + 3).is_some_and(u8::is_ascii_digit)
601        {
602            i += 4;
603        }
604    } else {
605        // A sign only counts with an integer part.
606        i = start;
607    }
608    // Optional fraction.
609    if i < b.len() && b[i] == b'.' {
610        i += 1;
611        let f0 = i;
612        while i < b.len() && b[i].is_ascii_digit() {
613            i += 1;
614        }
615        if !has_int && i == f0 {
616            return false; // `.` with no digits and no integer part
617        }
618    } else if !has_int {
619        return false; // neither integer nor fractional part
620    }
621    i == b.len()
622}
623
624pub(crate) fn render_table(table: &Table, compact: bool) -> String {
625    if table.rows.is_empty() {
626        return String::new();
627    }
628    let num_cols = table.rows.iter().map(Vec::len).max().unwrap_or(0);
629    if num_cols == 0 {
630        return String::new();
631    }
632
633    // Escaped, rectangular grid (ragged rows padded with empty cells). `tabulate`
634    // strips data cells of surrounding whitespace but leaves the header row as-is.
635    let grid: Vec<Vec<String>> = table
636        .rows
637        .iter()
638        .enumerate()
639        .map(|(r, row)| {
640            (0..num_cols)
641                .map(|c| {
642                    let cell = escape_cell(row.get(c).map(String::as_str).unwrap_or(""));
643                    if r == 0 {
644                        cell
645                    } else {
646                        cell.trim().to_string()
647                    }
648                })
649                .collect()
650        })
651        .collect();
652
653    if compact {
654        // Compact: cells joined by " | ", no padding, single-dash separators.
655        let render_row = |r: usize| -> String { format!("| {} |", grid[r].join(" | ")) };
656        let mut lines = Vec::with_capacity(grid.len() + 1);
657        lines.push(render_row(0));
658        let sep: Vec<&str> = (0..num_cols).map(|_| "-").collect();
659        lines.push(format!("| {} |", sep.join(" | ")));
660        for r in 1..grid.len() {
661            lines.push(render_row(r));
662        }
663        return lines.join("\n");
664    }
665
666    // Display width (Unicode scalar count — good enough for now).
667    let dw = |s: &str| s.chars().count();
668    let data_rows = 1..grid.len();
669
670    // A column is right-aligned when at least one data cell is numeric and every
671    // non-empty data cell is numeric — matching `tabulate`'s column typing, where
672    // empty cells are "missing" (ignored) and a number may carry thousands
673    // separators (`7,015`), which a plain `f64` parse rejects.
674    let right: Vec<bool> = (0..num_cols)
675        .map(|c| {
676            let mut any = false;
677            for r in data_rows.clone() {
678                let t = grid[r][c].trim();
679                if t.is_empty() {
680                    continue;
681                }
682                if !is_number_cell(t) {
683                    return false;
684                }
685                any = true;
686            }
687            any
688        })
689        .collect();
690
691    // Column width = max(header_width + MIN_PADDING(2), max data-cell width).
692    let width: Vec<usize> = (0..num_cols)
693        .map(|c| {
694            let mut w = dw(&grid[0][c]) + 2;
695            for r in data_rows.clone() {
696                w = w.max(dw(&grid[r][c]));
697            }
698            w
699        })
700        .collect();
701
702    let fmt_cell = |s: &str, c: usize| -> String {
703        let pad = " ".repeat(width[c].saturating_sub(dw(s)));
704        let body = if right[c] {
705            format!("{pad}{s}")
706        } else {
707            format!("{s}{pad}")
708        };
709        format!(" {body} ")
710    };
711    let render_row = |r: usize| -> String {
712        let cells: Vec<String> = (0..num_cols).map(|c| fmt_cell(&grid[r][c], c)).collect();
713        format!("|{}|", cells.join("|"))
714    };
715
716    let mut lines = Vec::with_capacity(grid.len() + 1);
717    lines.push(render_row(0));
718    let sep: Vec<String> = (0..num_cols).map(|c| "-".repeat(width[c] + 2)).collect();
719    lines.push(format!("|{}|", sep.join("|")));
720    for r in data_rows {
721        lines.push(render_row(r));
722    }
723    lines.join("\n")
724}
725
726/// Escape a table cell so it can't break the markdown table: newlines become
727/// spaces and pipes become the `&#124;` HTML entity (matches docling-core).
728fn escape_cell(s: &str) -> String {
729    s.replace('\n', " ").replace('|', "&#124;")
730}
731
732#[cfg(test)]
733mod tests {
734    use super::*;
735    use crate::PictureImage;
736
737    #[test]
738    fn renders_headings_paragraphs_and_lists() {
739        let mut doc = DoclingDocument::new("demo");
740        doc.add_heading(1, "Title");
741        doc.add_paragraph("Hello world.");
742        doc.push(Node::ListItem {
743            ordered: false,
744            number: 1,
745            first_in_list: true,
746            text: "first".into(),
747            level: 0,
748            marker: None,
749            location: None,
750            dclx: None,
751            href: None,
752            layer: None,
753        });
754        doc.push(Node::ListItem {
755            ordered: false,
756            number: 2,
757            first_in_list: false,
758            text: "second".into(),
759            level: 0,
760            marker: None,
761            location: None,
762            dclx: None,
763            href: None,
764            layer: None,
765        });
766        let md = doc.export_to_markdown();
767        assert_eq!(md, "# Title\n\nHello world.\n\n- first\n- second\n");
768    }
769
770    #[test]
771    fn strict_renders_recovered_links_legacy_does_not() {
772        let mut doc = DoclingDocument::new("cv");
773        doc.add_paragraph("Find me on LinkedIn or GitHub.");
774        doc.links = vec![
775            ("LinkedIn".into(), "https://www.linkedin.com/in/x/".into()),
776            ("GitHub".into(), "https://github.com/x/".into()),
777        ];
778        // Legacy/docling mode: links are left untouched (conformance preserved).
779        assert_eq!(doc.export_to_markdown(), "Find me on LinkedIn or GitHub.\n");
780        // Strict mode: anchors become Markdown links.
781        assert_eq!(
782            doc.export_to_markdown_with(true),
783            "Find me on [LinkedIn](https://www.linkedin.com/in/x/) or [GitHub](https://github.com/x/).\n"
784        );
785    }
786
787    #[test]
788    fn strict_links_match_escaped_anchor_and_consume_in_order() {
789        let mut doc = DoclingDocument::new("d");
790        // The PDF assembler HTML-escapes prose, so by serialization time the body
791        // already carries `&amp;`; the anchor is stored un-escaped. The matcher must
792        // escape the anchor to find it. Two identical anchors link in document order.
793        doc.add_paragraph("AI &amp; ML here, and issues here, then issues there.");
794        doc.links = vec![
795            ("AI & ML".into(), "https://a/".into()),
796            ("issues".into(), "https://first/".into()),
797            ("issues".into(), "https://second/".into()),
798        ];
799        assert_eq!(
800            doc.export_to_markdown_with(true),
801            "[AI &amp; ML](https://a/) here, and [issues](https://first/) here, then [issues](https://second/) there.\n"
802        );
803    }
804
805    #[test]
806    fn renders_compact_table() {
807        let mut doc = DoclingDocument::new("t");
808        // The compact form is opt-in (the PDF backend sets it); default output uses
809        // the padded GitHub serializer (covered by the regression fixtures).
810        doc.compact_tables = true;
811        doc.push(Node::Table(Table {
812            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
813            location: None,
814            structure: None,
815            cell_blocks: None,
816            caption: None,
817        }));
818        let md = doc.export_to_markdown();
819        assert_eq!(md, "| a | b |\n| - | - |\n| 1 | 2 |\n");
820    }
821
822    #[test]
823    fn renders_padded_github_table_by_default() {
824        let mut doc = DoclingDocument::new("t");
825        doc.push(Node::Table(Table {
826            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
827            location: None,
828            structure: None,
829            cell_blocks: None,
830            caption: None,
831        }));
832        let md = doc.export_to_markdown();
833        // Numeric data columns are right-aligned; columns padded to header+2.
834        assert_eq!(md, "|   a |   b |\n|-----|-----|\n|   1 |   2 |\n");
835    }
836
837    #[test]
838    fn strict_unescapes_inline_underscores_legacy_keeps_them() {
839        let mut doc = DoclingDocument::new("t");
840        doc.add_heading(1, "a\\_b");
841        doc.add_paragraph("x\\_y");
842        doc.push(Node::ListItem {
843            ordered: false,
844            number: 1,
845            first_in_list: true,
846            text: "i\\_j".into(),
847            level: 0,
848            marker: None,
849            location: None,
850            dclx: None,
851            href: None,
852            layer: None,
853        });
854        // Legacy reproduces docling's `\_` escaping byte-for-byte.
855        assert_eq!(doc.export_to_markdown(), "# a\\_b\n\nx\\_y\n\n- i\\_j\n");
856        // Strict prefers literal underscores (Rust-only readability mode).
857        assert_eq!(doc.export_to_markdown_with(true), "# a_b\n\nx_y\n\n- i_j\n");
858    }
859
860    /// Drive a document's nodes through [`MarkdownStreamer`] in the given page
861    /// splits and assert the concatenated chunks equal the buffered serializer.
862    fn assert_stream_matches(
863        doc: &DoclingDocument,
864        strict: bool,
865        images: ImageMode,
866        splits: &[usize],
867    ) {
868        let (want, want_artifacts) = to_markdown_images(doc, strict, images, "artifacts");
869        let mut streamer =
870            MarkdownStreamer::with_artifacts(strict, images, doc.compact_tables, "artifacts");
871        let mut got = String::new();
872        let mut got_artifacts = Vec::new();
873        let mut start = 0;
874        for &end in splits {
875            // Links only matter in strict mode; feed them all with the first batch
876            // that has content (document order is preserved by the queue).
877            let links = if start == 0 {
878                doc.links.as_slice()
879            } else {
880                &[]
881            };
882            got.push_str(&streamer.push(&doc.nodes[start..end], links));
883            // Referenced mode: drain per push, as a real caller writing files
884            // page by page would — numbering must continue across drains.
885            got_artifacts.extend(streamer.take_artifacts());
886            start = end;
887        }
888        got.push_str(&streamer.push(
889            &doc.nodes[start..],
890            if start == 0 {
891                doc.links.as_slice()
892            } else {
893                &[]
894            },
895        ));
896        got_artifacts.extend(streamer.take_artifacts());
897        got.push_str(&streamer.finish());
898        assert_eq!(
899            got, want,
900            "streamed output diverged (splits={splits:?}, strict={strict})"
901        );
902        assert_eq!(
903            got_artifacts, want_artifacts,
904            "streamed artifacts diverged (splits={splits:?}, strict={strict})"
905        );
906    }
907
908    #[test]
909    fn streaming_is_byte_identical_to_buffered() {
910        let mut doc = DoclingDocument::new("d");
911        doc.add_heading(1, "Title");
912        doc.add_paragraph("First paragraph.");
913        doc.push(Node::ListItem {
914            ordered: false,
915            number: 1,
916            first_in_list: true,
917            text: "a".into(),
918            level: 0,
919            marker: None,
920            location: None,
921            dclx: None,
922            href: None,
923            layer: None,
924        });
925        doc.push(Node::ListItem {
926            ordered: false,
927            number: 2,
928            first_in_list: false,
929            text: "b".into(),
930            level: 0,
931            marker: None,
932            location: None,
933            dclx: None,
934            href: None,
935            layer: None,
936        });
937        doc.push(Node::Code {
938            language: Some("rust".into()),
939            text: "let x = 1;".into(),
940            orig: None,
941            pretty: None,
942        });
943        doc.push(Node::Table(Table {
944            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
945            location: None,
946            structure: None,
947            cell_blocks: None,
948            caption: None,
949        }));
950        doc.push(Node::Picture {
951            caption: Some("Fig 1".into()),
952            image: Some(PictureImage {
953                mimetype: "image/png".into(),
954                width: 2,
955                height: 2,
956                data: b"png-one".to_vec(),
957            }),
958            classification: None,
959        });
960        doc.add_paragraph("Last paragraph.");
961        // A second embedded picture, so referenced mode must keep numbering
962        // (`image_000001`) across chunk boundaries.
963        doc.push(Node::Picture {
964            caption: None,
965            image: Some(PictureImage {
966                mimetype: "image/png".into(),
967                width: 2,
968                height: 2,
969                data: b"png-two".to_vec(),
970            }),
971            classification: None,
972        });
973
974        // A run of list items must never straddle a split, so try splits that fall
975        // on safe block boundaries (the streaming PDF assembler guarantees this).
976        for &strict in &[false, true] {
977            for &images in &[
978                ImageMode::Placeholder,
979                ImageMode::Embedded,
980                ImageMode::Referenced,
981            ] {
982                for splits in [&[][..], &[1][..], &[2][..], &[4][..], &[1, 4, 6, 7][..]] {
983                    assert_stream_matches(&doc, strict, images, splits);
984                }
985            }
986        }
987    }
988
989    #[test]
990    fn streaming_applies_recovered_links_in_strict_mode() {
991        let mut doc = DoclingDocument::new("d");
992        doc.add_paragraph("See LinkedIn for details.");
993        doc.add_paragraph("And GitHub too.");
994        doc.links = vec![
995            ("LinkedIn".into(), "https://lnkd/".into()),
996            ("GitHub".into(), "https://gh/".into()),
997        ];
998        // The second anchor lives in the second block, so it must be carried across
999        // the page boundary and placed when that block streams out.
1000        assert_stream_matches(&doc, true, ImageMode::Placeholder, &[1]);
1001    }
1002
1003    #[test]
1004    fn strict_tightens_punctuation_spacing_legacy_keeps_it() {
1005        let mut doc = DoclingDocument::new("t");
1006        doc.add_paragraph("see [ 37 , 36 ] and ( x ) .");
1007        // Legacy keeps docling's spacing byte-for-byte.
1008        assert_eq!(doc.export_to_markdown(), "see [ 37 , 36 ] and ( x ) .\n");
1009        // Strict tightens punctuation for readable Markdown.
1010        assert_eq!(doc.export_to_markdown_with(true), "see [37, 36] and (x).\n");
1011    }
1012}