Skip to main content

docling_core/
markdown.rs

1//! Markdown serializer for [`DoclingDocument`].
2
3use crate::document::{DoclingDocument, Node, Table};
4
5/// What docling's Markdown serializer writes for an item it has no component
6/// for — a `KeyValueItem` (the XBRL fact graph) is the one such item a backend
7/// produces.
8const MISSING_KEY_VALUE_ITEM: &str = "<!-- missing-key-value-item -->";
9
10/// How pictures are rendered (mirrors docling-core's `ImageRefMode`).
11#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
12pub enum ImageMode {
13    /// `<!-- image -->` (docling's default, and the only mode without image data).
14    #[default]
15    Placeholder,
16    /// `![Image](data:<mime>;base64,…)` — self-contained.
17    Embedded,
18    /// `![Image](<artifacts>/image_NNNNNN.<ext>)`; the bytes are returned for the
19    /// caller to write.
20    Referenced,
21}
22
23/// Serializer state threaded through the render walk.
24struct Ctx {
25    strict: bool,
26    /// Emit compact `| a | b |` tables instead of the padded GitHub serializer.
27    compact_tables: bool,
28    images: ImageMode,
29    artifacts_dir: String,
30    /// (relative path, bytes) for each referenced image — written by the caller.
31    artifacts: Vec<(String, Vec<u8>)>,
32    pic_index: usize,
33    /// Rendering the block content of a rich table cell (docling-core 2.94's
34    /// `in_table_cell`, docling-core#540): a heading has no valid Markdown
35    /// form inside a table, so it renders as plain text without `#` markers.
36    in_table_cell: bool,
37    /// docling-core's `MarkdownParams.page_break_placeholder`: the text that
38    /// separates two pages ([`DoclingDocument::page_break_placeholder`]).
39    /// `None` omits page breaks, docling's default.
40    page_break: Option<String>,
41    /// A page boundary has been crossed since the last rendered block, so the
42    /// next block is preceded by the placeholder. docling yields its
43    /// `_PageBreakNode` between two *items* whose `prov.page_no` differ, so a
44    /// boundary before the first block or after the last one emits nothing,
45    /// and a run of empty pages collapses into a single break.
46    pending_page_break: bool,
47    /// Whether any block has been rendered yet — for a streamer, across every
48    /// earlier push too — which is what makes a boundary a *pending* break.
49    emitted_any: bool,
50}
51
52/// Render a document to a Markdown string (pictures as placeholders).
53///
54/// `strict` selects the serializer-level behaviours that differ between
55/// docling-legacy output and cleaner Markdown — currently the code-fence
56/// language (legacy drops it, strict keeps it).
57pub fn to_markdown(doc: &DoclingDocument, strict: bool) -> String {
58    to_markdown_images(doc, strict, ImageMode::Placeholder, "artifacts").0
59}
60
61/// Render to Markdown with an explicit picture [`ImageMode`]. Returns the
62/// Markdown and, for [`ImageMode::Referenced`], the `(path, bytes)` of each image
63/// the caller should write (relative to the Markdown file).
64pub fn to_markdown_images(
65    doc: &DoclingDocument,
66    strict: bool,
67    images: ImageMode,
68    artifacts_dir: &str,
69) -> (String, Vec<(String, Vec<u8>)>) {
70    let mut ctx = Ctx {
71        strict,
72        compact_tables: doc.compact_tables,
73        images,
74        artifacts_dir: artifacts_dir.to_string(),
75        artifacts: Vec::new(),
76        pic_index: 0,
77        in_table_cell: false,
78        page_break: doc.page_break_placeholder.clone(),
79        pending_page_break: false,
80        emitted_any: false,
81    };
82    let mut blocks: Vec<String> = Vec::new();
83    render(&doc.nodes, &mut blocks, &mut ctx);
84    let mut body = blocks.join("\n\n");
85    // Strict mode only: turn recovered source hyperlinks into Markdown links.
86    // docling's standard pipeline drops them, so doing this in legacy mode would
87    // diverge from docling — hence strict-only, leaving conformance output intact.
88    if strict && !doc.links.is_empty() {
89        body = apply_links(&body, &doc.links);
90    }
91    let md = if body.is_empty() {
92        String::new()
93    } else {
94        format!("{body}\n")
95    };
96    (md, ctx.artifacts)
97}
98
99/// Render the block content of a *rich table cell* to Markdown — what
100/// docling-core's table serializer does for a `RichTableCell`
101/// (`doc_serializer.serialize(item, in_table_cell=True)`): the cell's
102/// paragraphs, lists and flattened nested tables render as in a document, but a
103/// heading loses its `#` markers (docling-core#540 — the Markdown spec has no
104/// headings inside tables). Pictures stay placeholders. The caller flattens the
105/// result into its cell text; the table serializer later turns the newlines
106/// into spaces.
107pub fn to_markdown_table_cell(doc: &DoclingDocument, strict: bool) -> String {
108    let mut ctx = Ctx {
109        strict,
110        compact_tables: doc.compact_tables,
111        images: ImageMode::Placeholder,
112        artifacts_dir: String::new(),
113        artifacts: Vec::new(),
114        pic_index: 0,
115        in_table_cell: true,
116        // A rich cell is one page's content; its sub-document carries no
117        // page boundaries and docling's `_iterate_items` runs the page-break
118        // scan over the document root only.
119        page_break: None,
120        pending_page_break: false,
121        emitted_any: false,
122    };
123    let mut blocks: Vec<String> = Vec::new();
124    render(&doc.nodes, &mut blocks, &mut ctx);
125    blocks.join("\n\n")
126}
127
128/// Wrap each recovered link's anchor text in Markdown `[anchor](href)`. Anchors
129/// arrive cleaned (curly quotes/dashes already normalized) but un-escaped, so we
130/// match against the body's HTML-escaped (`&`/`<`/`>`) form, the way prose nodes
131/// were serialized. Links are consumed in document order from a moving cursor, so
132/// a repeated anchor (e.g. two "issues") links its successive occurrences rather
133/// than all pointing at the first. An anchor that can't be located is skipped
134/// (its text may have been split across a line wrap or table cell).
135fn apply_links(body: &str, links: &[(String, String)]) -> String {
136    let mut out = body.to_string();
137    let mut cursor = 0usize;
138    for (anchor, href) in links {
139        let anchor = anchor
140            .replace('&', "&amp;")
141            .replace('<', "&lt;")
142            .replace('>', "&gt;");
143        if anchor.is_empty() {
144            continue;
145        }
146        if let Some(rel) = out[cursor..].find(&anchor) {
147            let at = cursor + rel;
148            // Don't relink inside an already-emitted `](` Markdown link target.
149            let replacement = format!("[{anchor}]({href})");
150            out.replace_range(at..at + anchor.len(), &replacement);
151            cursor = at + replacement.len();
152        }
153    }
154    out
155}
156
157/// Like [`apply_links`] but over a single chunk, consuming from a shared queue so
158/// the same `[anchor](href)` rewriting can be applied incrementally as Markdown is
159/// streamed out. Each queued link is matched (in document order) against `chunk`
160/// and rewritten in place; a link whose anchor is not in this chunk is carried
161/// forward in the queue for a later chunk. Anchors are recovered in document
162/// order and a chunk is always a contiguous run of whole blocks, so this
163/// reproduces [`apply_links`]' single moving cursor: the link lands in whichever
164/// chunk contains its anchor, identically to the buffered path. (A link whose
165/// anchor never appears is carried to the end and dropped — the same no-op
166/// `apply_links` performs for an unlocatable anchor.)
167fn apply_links_chunk(chunk: &str, queue: &mut Vec<(String, String)>) -> String {
168    let mut out = chunk.to_string();
169    let mut cursor = 0usize;
170    let mut carried: Vec<(String, String)> = Vec::new();
171    for (anchor_raw, href) in std::mem::take(queue) {
172        let anchor = anchor_raw
173            .replace('&', "&amp;")
174            .replace('<', "&lt;")
175            .replace('>', "&gt;");
176        if anchor.is_empty() {
177            continue;
178        }
179        if let Some(rel) = out[cursor..].find(&anchor) {
180            let at = cursor + rel;
181            let replacement = format!("[{anchor}]({href})");
182            out.replace_range(at..at + anchor.len(), &replacement);
183            cursor = at + replacement.len();
184        } else {
185            // Not in this chunk; try again when its block is flushed.
186            carried.push((anchor_raw, href));
187        }
188    }
189    *queue = carried;
190    out
191}
192
193/// Incremental Markdown serializer: feed finalized, in-document-order batches of
194/// [`Node`]s and receive Markdown chunks whose concatenation is **byte-identical**
195/// to [`to_markdown_images`] over the same nodes. This is the streaming
196/// counterpart of the buffered serializer — used to emit a document's Markdown in
197/// chunks (e.g. page by page, as the parallel PDF pipeline finishes pages) instead
198/// of building the whole string up front.
199///
200/// [`ImageMode::Placeholder`] and [`ImageMode::Embedded`] render inline.
201/// [`ImageMode::Referenced`] additionally hands each picture's bytes out through
202/// [`take_artifacts`](Self::take_artifacts) — construct with
203/// [`with_artifacts`](Self::with_artifacts) and drain after every push so the
204/// bytes can be written to disk as pages finish instead of accumulating for the
205/// whole document (issue #80's memory-bounded image handling).
206///
207/// Each [`push`](Self::push) must contain whole blocks in reading order: a caller
208/// must not split a run of list items across two pushes (the run would render as
209/// two separate lists). Finalized PDF page batches already satisfy this.
210pub struct MarkdownStreamer {
211    strict: bool,
212    images: ImageMode,
213    compact_tables: bool,
214    /// Whether any non-empty chunk has been emitted yet (drives `\n\n` joins and
215    /// the trailing newline).
216    emitted_any: bool,
217    /// Recovered links not yet placed (strict mode), consumed in document order.
218    links: Vec<(String, String)>,
219    /// Referenced mode: the link prefix, the not-yet-drained `(path, bytes)`
220    /// artifacts, and the running image number (continues across pushes so the
221    /// stream matches the buffered serializer's `image_000000…` numbering).
222    artifacts_dir: String,
223    artifacts: Vec<(String, Vec<u8>)>,
224    pic_index: usize,
225    /// [`DoclingDocument::page_break_placeholder`] and the boundary carried
226    /// over from the previous push (a page batch opens with its page marker,
227    /// so the break it implies is paid by that batch's first block).
228    page_break: Option<String>,
229    pending_page_break: bool,
230}
231
232impl MarkdownStreamer {
233    /// Create a streamer. `compact_tables` mirrors [`DoclingDocument::compact_tables`].
234    /// For [`ImageMode::Referenced`] use [`with_artifacts`](Self::with_artifacts).
235    pub fn new(strict: bool, images: ImageMode, compact_tables: bool) -> Self {
236        debug_assert!(
237            images != ImageMode::Referenced,
238            "referenced image mode needs an artifacts dir; use with_artifacts"
239        );
240        Self::with_artifacts(strict, images, compact_tables, "artifacts")
241    }
242
243    /// Like [`new`](Self::new) but with the artifacts link prefix, allowing
244    /// [`ImageMode::Referenced`]: pictures render as
245    /// `![Image](<artifacts_dir>/image_NNNNNN.<ext>)` and each push's image
246    /// bytes wait in [`take_artifacts`](Self::take_artifacts) for the caller to
247    /// write. The concatenated chunks and the artifact list match the buffered
248    /// [`to_markdown_images`] byte-for-byte.
249    pub fn with_artifacts(
250        strict: bool,
251        images: ImageMode,
252        compact_tables: bool,
253        artifacts_dir: &str,
254    ) -> Self {
255        Self {
256            strict,
257            images,
258            compact_tables,
259            emitted_any: false,
260            links: Vec::new(),
261            artifacts_dir: artifacts_dir.to_string(),
262            artifacts: Vec::new(),
263            pic_index: 0,
264            page_break: None,
265            pending_page_break: false,
266        }
267    }
268
269    /// Insert `placeholder` between pages, mirroring
270    /// [`DoclingDocument::page_break_placeholder`] for the buffered path (the
271    /// concatenated chunks stay byte-identical to it). `None` — the default —
272    /// omits page breaks. Set before the first [`push`](Self::push).
273    pub fn with_page_break_placeholder(mut self, placeholder: Option<String>) -> Self {
274        self.page_break = placeholder;
275        self
276    }
277
278    /// The `(relative path, bytes)` of images rendered by pushes since the last
279    /// drain ([`ImageMode::Referenced`] only — empty otherwise). Paths are
280    /// relative to the Markdown file, i.e. they start with the configured
281    /// artifacts dir.
282    pub fn take_artifacts(&mut self) -> Vec<(String, Vec<u8>)> {
283        std::mem::take(&mut self.artifacts)
284    }
285
286    /// Render one finalized batch of nodes (plus any links recovered from the same
287    /// span, in document order) into the next Markdown chunk. Returns an empty
288    /// string when the batch produces no output (e.g. empty tables/pictures), in
289    /// which case nothing should be written.
290    pub fn push(&mut self, nodes: &[Node], links: &[(String, String)]) -> String {
291        self.links.extend(links.iter().cloned());
292        let mut ctx = Ctx {
293            strict: self.strict,
294            compact_tables: self.compact_tables,
295            images: self.images,
296            artifacts_dir: std::mem::take(&mut self.artifacts_dir),
297            artifacts: std::mem::take(&mut self.artifacts),
298            pic_index: self.pic_index,
299            in_table_cell: false,
300            page_break: std::mem::take(&mut self.page_break),
301            pending_page_break: self.pending_page_break,
302            emitted_any: self.emitted_any,
303        };
304        let mut blocks: Vec<String> = Vec::new();
305        render(nodes, &mut blocks, &mut ctx);
306        self.artifacts_dir = std::mem::take(&mut ctx.artifacts_dir);
307        self.artifacts = std::mem::take(&mut ctx.artifacts);
308        self.pic_index = ctx.pic_index;
309        self.page_break = std::mem::take(&mut ctx.page_break);
310        self.pending_page_break = ctx.pending_page_break;
311        if blocks.is_empty() {
312            return String::new();
313        }
314        let mut body = blocks.join("\n\n");
315        if self.strict && !self.links.is_empty() {
316            body = apply_links_chunk(&body, &mut self.links);
317        }
318        let chunk = if self.emitted_any {
319            format!("\n\n{body}")
320        } else {
321            body
322        };
323        self.emitted_any = true;
324        chunk
325    }
326
327    /// Emit the trailing newline that finishes the document (empty if no content
328    /// was produced). Call exactly once, after the final [`push`](Self::push).
329    pub fn finish(self) -> String {
330        if self.emitted_any {
331            "\n".to_string()
332        } else {
333            String::new()
334        }
335    }
336}
337
338/// In `strict` mode, rewrite inline text for readability rather than byte-for-byte
339/// docling fidelity: undo the legacy `\_` underscore escaping, and tighten stray
340/// spaces around punctuation (`[ 37 , 36 ]` → `[37, 36]`, `( x )` → `(x)`). This
341/// cleans up both the PDF backend's glyph-split spacing and the space the legacy
342/// emphasis serialization leaves before punctuation (`*a* ,` → `*a*,`).
343/// Legacy/default output keeps docling's spacing untouched. Only inline text
344/// nodes pass through here — code blocks and table cells are left alone.
345fn strict_text(text: &str, strict: bool) -> String {
346    if !strict {
347        return text.to_string();
348    }
349    text.replace("\\_", "_")
350        .replace(" ,", ",")
351        .replace(" .", ".")
352        .replace(" ;", ";")
353        .replace(" )", ")")
354        .replace("( ", "(")
355        .replace(" ]", "]")
356        .replace("[ ", "[")
357}
358
359/// docling-core 2.92's `_md_line_breaks` (docling-core#721): a single `\n`
360/// inside an item's text becomes a GFM hard line break (`"  \n"`, two trailing
361/// spaces) so renderers honour it; a blank line (`\n\n`) is a paragraph break
362/// and stays as is — the document scope already joins blocks with `\n\n`.
363/// Applied to body text, list items and captions, never to code/formulas.
364fn md_line_breaks(text: &str) -> String {
365    if !text.contains('\n') {
366        return text.to_string();
367    }
368    text.split("\n\n")
369        .map(|para| para.replace('\n', "  \n"))
370        .collect::<Vec<_>>()
371        .join("\n\n")
372}
373
374/// Undo [`md_line_breaks`] on a rich table cell's flattened Markdown so the
375/// non-Markdown exports (JSON `text`, LaTeX cells) see the cell's raw line
376/// breaks, as docling's do — a rich cell's text is its Markdown serialization
377/// in our model, and the two trailing spaces are a Markdown-only marker.
378pub(crate) fn strip_hard_breaks(text: &str) -> String {
379    if text.contains("  \n") {
380        text.replace("  \n", "\n")
381    } else {
382        text.to_string()
383    }
384}
385
386/// docling-core's `_heading_line_breaks`: a GFM heading cannot span lines, so a
387/// newline inside heading text collapses to a space (`# Hello World`, not a
388/// broken `# Hello\nWorld`).
389fn heading_line_breaks(text: &str) -> String {
390    text.replace('\n', " ")
391}
392
393fn render(nodes: &[Node], blocks: &mut Vec<String>, ctx: &mut Ctx) {
394    let mut i = 0;
395    while i < nodes.len() {
396        let before = blocks.len();
397        match &nodes[i] {
398            // A page boundary: the explicit `PageBreak` (slides, DjVu / DocTags
399            // pages) or the `PageInfo` marker that opens every PDF page and
400            // spreadsheet sheet — a sheet boundary carries both, and the flag
401            // absorbs the pair into one break. docling's `_iterate_items`
402            // yields a `_PageBreakNode` only between two items on different
403            // pages (a group's leading item counts for the group), which is
404            // exactly "a boundary between two rendered blocks": nothing before
405            // the first block, nothing after the last, consecutive boundaries
406            // — empty or furniture-only pages — collapsed into one.
407            Node::PageBreak | Node::PageInfo { .. } => {
408                if ctx.page_break.is_some() && ctx.emitted_any {
409                    ctx.pending_page_break = true;
410                }
411                i += 1;
412            }
413            Node::ListItem { .. } => {
414                let start = i;
415                i += 1;
416                loop {
417                    match nodes.get(i) {
418                        Some(Node::ListItem { .. }) => i += 1,
419                        // An empty paragraph between two list items is absorbed
420                        // into the run — docling keeps such a ListGroup
421                        // contiguous rather than splitting it.
422                        Some(Node::Paragraph { text })
423                            if text.is_empty()
424                                && matches!(nodes.get(i + 1), Some(Node::ListItem { .. })) =>
425                        {
426                            i += 1
427                        }
428                        _ => break,
429                    }
430                }
431                render_list_run(&nodes[start..i], blocks, ctx.strict);
432            }
433            other => {
434                render_one(other, blocks, ctx);
435                i += 1;
436            }
437        }
438        if blocks.len() > before {
439            if ctx.pending_page_break {
440                // The placeholder is a block of its own, joined by the document
441                // delimiter like docling's `_PageBreakSerResult` part — an
442                // empty placeholder therefore leaves the doubled `\n\n`
443                // upstream leaves too.
444                if let Some(placeholder) = &ctx.page_break {
445                    blocks.insert(before, placeholder.clone());
446                }
447                ctx.pending_page_break = false;
448            }
449            ctx.emitted_any = true;
450        }
451    }
452}
453
454/// Render a contiguous run of list items.
455///
456/// Ordered items use their explicit `number`. A new sibling list (marked by
457/// `first_in_list`) at the same depth is separated by a blank line, matching
458/// docling-core's serializer.
459fn render_list_run(items: &[Node], blocks: &mut Vec<String>, strict: bool) {
460    let mut lines: Vec<String> = Vec::new();
461    // Whether a top-level item has been rendered yet — a fresh-list flag on
462    // the very first item opens nothing.
463    let mut any_top = false;
464
465    for item in items {
466        let Node::ListItem {
467            ordered,
468            number,
469            first_in_list,
470            text,
471            level,
472            marker: orig_marker,
473            location: _,
474            dclx: _,
475            href: _,
476            layer,
477        } = item
478        else {
479            continue;
480        };
481        // A non-body (furniture) list item is omitted from Markdown, matching
482        // docling's content-layer filtering.
483        if layer.is_some() {
484            continue;
485        }
486        let level = *level as usize;
487
488        // A new sibling list at the top level gets a blank line — and only the
489        // backend knows where one starts (`first_in_list`: Word's `numId`
490        // changing, an HTML `<ul>` closing, a Markdown bullet switching
491        // `-`→`*`). The serializer used to guess it from a kind flip or a
492        // number gap as well, which split lists docling keeps whole (an
493        // AsciiDoc `1.` … `5.`, mixed `*`/`1.` markers) — #385. Only at the
494        // top level: nested sibling groups are children of a list item, and
495        // docling joins an item's children without blank lines.
496        if level == 0 {
497            if any_top && *first_in_list {
498                lines.push(String::new());
499            }
500            any_top = true;
501        }
502
503        let indent = "    ".repeat(level);
504        // docling-core's `case_already_valid`: a marker of digits and a dot
505        // prints verbatim — Python's `\d+\.` admits every Unicode decimal
506        // digit, so a DOCX `decimalFullWidth` marker (`1.`, docling#4336)
507        // is kept as it is rather than renumbered in ASCII.
508        let verbatim = orig_marker.as_deref().filter(|m| {
509            m.strip_suffix('.')
510                .is_some_and(|d| !d.is_empty() && d.chars().all(char::is_numeric))
511        });
512        let marker = match verbatim {
513            Some(m) if *ordered => m.to_string(),
514            _ if *ordered => format!("{number}."),
515            _ => "-".to_string(),
516        };
517        lines.push(format!("{indent}{marker} {}", list_item_text(text, strict)));
518    }
519
520    // A run consisting only of furniture (content-layer-filtered) items yields no
521    // lines; pushing an empty block here would surface as a stray blank line.
522    if !lines.is_empty() {
523        blocks.push(lines.join("\n"));
524    }
525}
526
527/// A list item's Markdown body. The GFM hard-line-break rule (docling-core#721)
528/// applies to the item's own text; pictures the HTML backend folded into the
529/// item (`"\n[alt\n]<!-- image -->"` per `<img>` inside the `<li>`) are
530/// docling's picture *children* of the item, which its serializer prints after
531/// the item line with plain newlines — so a folded tail keeps its newlines
532/// unmarked. The tail is recognised structurally: every line after the first is
533/// an image marker or an alt caption directly followed by one.
534fn list_item_text(text: &str, strict: bool) -> String {
535    let escaped = strict_text(text, strict);
536    if let Some((own, tail)) = escaped.split_once('\n') {
537        if is_folded_child_tail(tail) {
538            return format!("{}\n{tail}", md_line_breaks(own));
539        }
540    }
541    md_line_breaks(&escaped)
542}
543
544/// Whether everything after a list item's own first line is a folded *child*
545/// block rather than a continuation of the item's text: an image marker
546/// (optionally preceded by its caption/alt line) or a fenced code block. The
547/// AsciiDoc backend indents such a block to the item's own depth (as
548/// docling-core's list serializer does for each part it emits), so a leading
549/// indent is ignored here.
550fn is_folded_child_tail(tail: &str) -> bool {
551    const MARKER: &str = "<!-- image -->";
552    const FENCE: &str = "```";
553    let mut lines = tail.split('\n').peekable();
554    let mut any = false;
555    while let Some(line) = lines.next() {
556        let line = line.trim_start();
557        if line == MARKER {
558            any = true;
559        } else if line == FENCE {
560            // Skip the block's body; an unclosed fence is not a folded child.
561            loop {
562                match lines.next() {
563                    Some(l) if l.trim_start() == FENCE => break,
564                    Some(_) => {}
565                    None => return false,
566                }
567            }
568            any = true;
569        } else if lines.next().map(str::trim_start) == Some(MARKER) {
570            any = true; // an alt caption line, then its marker
571        } else {
572            return false;
573        }
574    }
575    any
576}
577
578fn render_one(node: &Node, blocks: &mut Vec<String>, ctx: &mut Ctx) {
579    match node {
580        Node::Heading { level, text } => {
581            let text = heading_line_breaks(&strict_text(text, ctx.strict));
582            if ctx.in_table_cell {
583                // docling-core#540: no `#` markers inside a table cell.
584                blocks.push(text);
585            } else {
586                let hashes = "#".repeat((*level).clamp(1, 6) as usize);
587                blocks.push(format!("{hashes} {text}"));
588            }
589        }
590        // An empty body paragraph (docling's blank-line text item) contributes
591        // nothing to Markdown — only DocLang/JSON keep it.
592        Node::Paragraph { text } if text.is_empty() => {}
593        Node::Paragraph { text } => blocks.push(md_line_breaks(&strict_text(text, ctx.strict))),
594        // A standalone caption item renders like a text item; its hyperlink
595        // annotation becomes a Markdown link around the whole caption.
596        Node::Caption { text, .. } if text.is_empty() => {}
597        Node::Caption { text, href } => {
598            let body = md_line_breaks(&strict_text(text, ctx.strict));
599            blocks.push(match href {
600                Some(url) => format!("[{body}]({url})"),
601                None => body,
602            });
603        }
604        Node::CheckboxItem { checked, text } => {
605            let mark = if *checked { "- [x] " } else { "- [ ] " };
606            blocks.push(md_line_breaks(&strict_text(
607                &format!("{mark}{text}"),
608                ctx.strict,
609            )));
610        }
611        Node::Code {
612            language,
613            text,
614            pretty,
615            ..
616        } => {
617            // Legacy docling never emits a language on the fence; strict keeps it.
618            let lang = match language {
619                Some(l) if ctx.strict => l.as_str(),
620                _ => "",
621            };
622            // Strict prefers the line-preserving rendering when the backend
623            // supplied one (PDF); legacy stays on docling's flat `text`.
624            let body = match pretty {
625                Some(p) if ctx.strict => p.as_str(),
626                _ => text.as_str(),
627            };
628            blocks.push(format!("```{lang}\n{body}\n```"));
629        }
630        // A CodeFormula-decoded display formula renders as docling's `$$…$$`
631        // (the un-enriched pipeline emits a placeholder paragraph instead).
632        Node::Formula { latex, .. } => blocks.push(format!("$${latex}$$")),
633        Node::Table(table) => {
634            // docling renders a table's caption as a text line before the grid.
635            // `caption` is already escaped (backend convention), like a paragraph.
636            if let Some(cap) = &table.caption {
637                if !cap.is_empty() {
638                    blocks.push(md_line_breaks(&strict_text(cap, ctx.strict)));
639                }
640            }
641            let rendered = render_table(table, ctx.compact_tables);
642            if !rendered.is_empty() {
643                blocks.push(rendered);
644            }
645        }
646        // Classification predictions don't affect docling's Markdown output.
647        Node::Picture { caption, image, .. } => {
648            if let Some(cap) = caption {
649                if !cap.is_empty() {
650                    blocks.push(md_line_breaks(cap));
651                }
652            }
653            blocks.push(picture_marker(image.as_ref(), ctx));
654        }
655        // A chart renders as docling's picture-with-meta markdown: the caption,
656        // the placeholder, the humanized classification ("line_chart" ->
657        // "Line chart"), then the chart's data grid as a regular table.
658        Node::Chart {
659            kind,
660            table,
661            caption,
662            ..
663        } => {
664            if let Some(cap) = caption {
665                if !cap.is_empty() {
666                    blocks.push(md_line_breaks(cap));
667                }
668            }
669            blocks.push(picture_marker(None, ctx));
670            blocks.push(humanize_label(kind));
671            let rendered = render_table(table, false);
672            if !rendered.is_empty() {
673                blocks.push(rendered);
674            }
675        }
676        // A DocLang-only node is omitted from Markdown.
677        Node::DoclangOnly(_) => {}
678        // A group on a non-body layer (a hidden spreadsheet sheet) renders
679        // nothing, like every other non-body item.
680        Node::Group { layer: Some(_), .. } => {}
681        Node::Group { children, .. } => render(children, blocks, ctx),
682        Node::FieldRegion { items } => {
683            // The region container and each field item carry no text of their
684            // own; docling-core 2.93 (#724) serializes them to nothing (older
685            // releases emitted a `<!-- missing-text -->` marker for each), so
686            // only an item's marker/key/value appear, as separate paragraphs.
687            for item in items {
688                for part in [&item.marker, &item.key, &item.value].into_iter().flatten() {
689                    blocks.push(md_line_breaks(&strict_text(part, ctx.strict)));
690                }
691            }
692        }
693        // docling's Markdown serializer has no component for a `KeyValueItem`
694        // and writes its fallback placeholder in the item's place.
695        Node::KeyValueGraph { .. } => blocks.push(MISSING_KEY_VALUE_ITEM.to_string()),
696        // A rich inline group renders exactly like a paragraph of its Markdown
697        // text — the structured runs are DocLang-only.
698        Node::InlineGroup { md_text, .. } => {
699            blocks.push(md_line_breaks(&strict_text(md_text, ctx.strict)))
700        }
701        // A plain-text backend dump renders verbatim as a single block.
702        Node::TextDump(text) => {
703            if !text.is_empty() {
704                blocks.push(text.clone());
705            }
706        }
707        // Furniture (page headers/footers, HTML `<title>`) is excluded from
708        // Markdown by default, mirroring docling.
709        Node::Furniture { .. } => {}
710        Node::PageFurniture { .. } | Node::FurnitureText { .. } => {}
711        // A picture's contained text is JSON-only: docling's Markdown picture
712        // serializer prints the caption and the image, never the children.
713        Node::PictureChildren(_) => {}
714        // A comment lives in the notes layer — omitted like other furniture;
715        // the annotation on a body item is JSON-only, so render the item.
716        Node::CommentSection { .. } => {}
717        Node::Commented { inner, .. } => render_one(inner, blocks, ctx),
718        // Layout provenance is DocLang-only; render the wrapped node.
719        Node::Located { inner, .. } | Node::Prov { inner, .. } => render_one(inner, blocks, ctx),
720        // Page breaks are DocLang-only; docling omits them from Markdown.
721        Node::PageBreak => {}
722        // Page markers feed the JSON export only.
723        Node::PageInfo { .. } => {}
724        // Runs of adjacent list items are merged by `render`; a stray single
725        // item (a hand-built document, or a `Located` wrapper around one)
726        // still renders as its own one-item list instead of panicking —
727        // `nodes` is public API, so every representable tree must serialize.
728        Node::ListItem { .. } => render_list_run(std::slice::from_ref(node), blocks, ctx.strict),
729    }
730}
731
732/// The Markdown for a picture under the active [`ImageMode`]; Referenced mode also
733/// records the bytes in `ctx.artifacts` for the caller to write.
734/// docling-core's `_humanize_text`: underscores to spaces, first letter
735/// capitalized ("line_chart" -> "Line chart").
736fn humanize_label(label: &str) -> String {
737    let text = label.replace('_', " ");
738    let mut chars = text.chars();
739    match chars.next() {
740        Some(f) => f.to_uppercase().collect::<String>() + chars.as_str(),
741        None => text,
742    }
743}
744
745fn picture_marker(image: Option<&crate::PictureImage>, ctx: &mut Ctx) -> String {
746    match (ctx.images, image) {
747        // docling embeds the `ImageRef`'s PNG (a JPEG re-encoded).
748        (ImageMode::Embedded, Some(img)) => {
749            format!("![Image]({})", crate::pixel_digest::docling_data_uri(img).1)
750        }
751        (ImageMode::Referenced, Some(img)) => {
752            let path = format!(
753                "{}/image_{:06}.{}",
754                ctx.artifacts_dir,
755                ctx.pic_index,
756                ext_for(&img.mimetype)
757            );
758            ctx.pic_index += 1;
759            ctx.artifacts.push((path.clone(), img.data.clone()));
760            format!("![Image]({})", escape_uri_path(&path))
761        }
762        // Placeholder, or any mode with no extracted image.
763        _ => "<!-- image -->".to_string(),
764    }
765}
766
767/// Encode a URL or filesystem path as a Markdown link destination —
768/// docling-core's `MarkdownPictureSerializer._escape_uri_path`
769/// (docling-core#698, 2.94). Handles URLs of any scheme as well as POSIX and
770/// Windows paths, keeps relative paths relative and never double-encodes:
771/// backslashes become `/` (a backslash is both the Windows separator and a
772/// Markdown escape), a UNC share `//host/…` and an absolute Windows path
773/// `C:/…` become RFC 8089 `file://` URLs (the one spelling a renderer cannot
774/// misread as a scheme-relative URL or a `C:` scheme), a URL keeps its
775/// scheme / authority / delimiters with only the components encoded, and
776/// everything else is percent-encoded as a path. `%` is kept so an
777/// already-encoded destination stays as it is; spaces and parentheses are
778/// encoded because they would end (or unbalance) a Markdown inline link.
779pub(crate) fn escape_uri_path(value: &str) -> String {
780    const KEEP: &str = "/%:@+,;=~$!&'*";
781    let s = value.replace('\\', "/");
782    if let Some(rest) = s.strip_prefix("//") {
783        // A fileshare: `file://<host>/<path>`, the host possibly empty.
784        let rest = rest.trim_start_matches('/');
785        let (host, tail) = rest.split_once('/').unwrap_or((rest, ""));
786        return format!("file://{host}{}", percent_quote(&format!("/{tail}"), KEEP));
787    }
788    let bytes = s.as_bytes();
789    if bytes.len() >= 3 && bytes[0].is_ascii_alphabetic() && bytes[1] == b':' && bytes[2] == b'/' {
790        // A Windows path with a drive letter: `file:///C:/…`.
791        return format!("file:///{}", percent_quote(&s, KEEP));
792    }
793    // A URL keeps its scheme, authority and delimiters; only its components are
794    // encoded. A single-character scheme cannot be real (it is a drive letter,
795    // handled above), so it is read as a path — like `urlsplit`.
796    if let Some((scheme, rest)) = s.split_once(':') {
797        let valid_scheme = scheme.len() > 1
798            && scheme.as_bytes()[0].is_ascii_alphabetic()
799            && scheme
800                .bytes()
801                .all(|b| b.is_ascii_alphanumeric() || matches!(b, b'+' | b'-' | b'.'));
802        if valid_scheme {
803            let (authority, rest) = match rest.strip_prefix("//") {
804                Some(r) => {
805                    let end = r.find(['/', '?', '#']).unwrap_or(r.len());
806                    (Some(&r[..end]), &r[end..])
807                }
808                None => (None, rest),
809            };
810            let (before_frag, fragment) = rest.split_once('#').unwrap_or((rest, ""));
811            let (path, query) = before_frag.split_once('?').unwrap_or((before_frag, ""));
812            let mut out = format!("{scheme}:");
813            if let Some(a) = authority {
814                out.push_str("//");
815                out.push_str(a);
816            }
817            out.push_str(&percent_quote(path, KEEP));
818            if !query.is_empty() {
819                out.push('?');
820                out.push_str(&percent_quote(query, KEEP));
821            }
822            if !fragment.is_empty() {
823                out.push('#');
824                out.push_str(&percent_quote(fragment, KEEP));
825            }
826            return out;
827        }
828    }
829    // A relative or root-relative local path.
830    percent_quote(&s, KEEP)
831}
832
833/// `urllib.parse.quote(s, safe)`: unreserved ASCII (`A–Z a–z 0–9 _ . - ~`) and
834/// the `safe` set stay, every other byte of the UTF-8 encoding becomes `%XX`.
835fn percent_quote(s: &str, safe: &str) -> String {
836    let mut out = String::with_capacity(s.len());
837    for &b in s.as_bytes() {
838        let keep = b.is_ascii_alphanumeric()
839            || matches!(b, b'_' | b'.' | b'-' | b'~')
840            || (b.is_ascii() && safe.contains(b as char));
841        if keep {
842            out.push(b as char);
843        } else {
844            out.push_str(&format!("%{b:02X}"));
845        }
846    }
847    out
848}
849
850pub(crate) fn ext_for(mimetype: &str) -> &str {
851    match mimetype {
852        "image/jpeg" => "jpg",
853        "image/gif" => "gif",
854        "image/webp" => "webp",
855        "image/bmp" => "bmp",
856        "image/tiff" => "tif",
857        _ => "png",
858    }
859}
860
861/// Render a table. `compact` selects between two serializers:
862///
863/// - **padded** (default) — docling-core's `tabulate(tablefmt="github")`: columns
864///   are padded to a fixed width (header width + a minimum padding of 2, or the
865///   widest data cell); numeric columns (every data cell parses as a number) are
866///   right-aligned, others left-aligned; separators are plain dashes of
867///   `width + 2`. Matches current published docling (DOCX/HTML conformance).
868/// - **compact** — `| a | b |` cells with single-dash `| - | - |` separators, no
869///   width padding. Matches the committed PDF groundtruth corpus, which predates
870///   the padded serializer.
871///
872/// Each cell is first escaped (`\n` → space, `|` → `&#124;`) so it can't break the
873/// table. The header row is the table's leading `column_header` block flattened
874/// to one row ([`Table::header_row_count`] + [`flatten_header_rows`],
875/// docling-core#723); alignment and widths are computed over the body rows.
876/// Whether a table cell counts as a number for column alignment, matching
877/// `tabulate`'s detection: an ordinary float/int (`f64`-parseable, covering
878/// `1e2`/`inf`/`+1.5`) **or** a thousands-separated number like `7,015`.
879fn is_number_cell(t: &str) -> bool {
880    t.parse::<f64>().is_ok() || is_thousands_number(t)
881}
882
883/// A number with comma thousands-separators, per `tabulate`'s
884/// `_float_with_thousands_separators` regex
885/// (`^(([+-]?[0-9]{1,3})(?:,([0-9]{3}))*)?(?(1)\.[0-9]*|\.[0-9]+)?$`): the
886/// integer part is 1–3 digits then any number of `,ddd` groups; the fraction is
887/// optional (and, without an integer part, must have at least one digit).
888fn is_thousands_number(t: &str) -> bool {
889    let b = t.as_bytes();
890    let mut i = 0;
891    let start = i;
892    if i < b.len() && (b[i] == b'+' || b[i] == b'-') {
893        i += 1;
894    }
895    // First digit chunk: 1–3 digits.
896    let d0 = i;
897    while i < b.len() && b[i].is_ascii_digit() && i - d0 < 3 {
898        i += 1;
899    }
900    let has_int = i > d0;
901    if has_int {
902        // Subsequent `,ddd` groups (exactly three digits each).
903        while i + 3 < b.len() + 1
904            && b.get(i) == Some(&b',')
905            && b.get(i + 1).is_some_and(u8::is_ascii_digit)
906            && b.get(i + 2).is_some_and(u8::is_ascii_digit)
907            && b.get(i + 3).is_some_and(u8::is_ascii_digit)
908        {
909            i += 4;
910        }
911    } else {
912        // A sign only counts with an integer part.
913        i = start;
914    }
915    // Optional fraction.
916    if i < b.len() && b[i] == b'.' {
917        i += 1;
918        let f0 = i;
919        while i < b.len() && b[i].is_ascii_digit() {
920            i += 1;
921        }
922        if !has_int && i == f0 {
923            return false; // `.` with no digits and no integer part
924        }
925    } else if !has_int {
926        return false; // neither integer nor fractional part
927    }
928    i == b.len()
929}
930
931/// The single GFM header row for a table: the leading header rows (see
932/// [`Table::header_row_count`]) flattened per column, texts joined with
933/// `" - "` after dropping consecutive duplicates — docling-core's
934/// `_flatten_header_rows` (docling-core#723). The duplicate rule is what
935/// keeps a row-spanning header from being joined to itself (the grid repeats
936/// its text into every row it covers); it is position-based, so two stacked
937/// levels sharing a label collapse too — GFM has one header row, and upstream
938/// accepts that loss. No header rows → one empty header cell per column.
939fn flatten_header_rows(header_rows: &[Vec<String>], num_cols: usize) -> Vec<String> {
940    (0..num_cols)
941        .map(|c| {
942            let mut parts: Vec<&str> = Vec::new();
943            for row in header_rows {
944                let text = row.get(c).map(String::as_str).unwrap_or("");
945                if !text.is_empty() && parts.last() != Some(&text) {
946                    parts.push(text);
947                }
948            }
949            parts.join(" - ")
950        })
951        .collect()
952}
953
954pub(crate) fn render_table(table: &Table, compact: bool) -> String {
955    if table.rows.is_empty() {
956        return String::new();
957    }
958    let num_cols = table.rows.iter().map(Vec::len).max().unwrap_or(0);
959    if num_cols == 0 {
960        return String::new();
961    }
962
963    // Escaped, rectangular grid (ragged rows padded with empty cells). The
964    // header block is resolved to the one row GFM allows (docling-core#723);
965    // `tabulate` strips data cells of surrounding whitespace but leaves the
966    // header texts as-is.
967    let num_headers = table.header_row_count().min(table.rows.len());
968    let escaped = |r: usize| -> Vec<String> {
969        (0..num_cols)
970            .map(|c| escape_cell(table.rows[r].get(c).map(String::as_str).unwrap_or("")))
971            .collect()
972    };
973    let header_rows: Vec<Vec<String>> = (0..num_headers).map(escaped).collect();
974    let header = flatten_header_rows(&header_rows, num_cols);
975    let body: Vec<Vec<String>> = (num_headers..table.rows.len())
976        .map(|r| {
977            escaped(r)
978                .into_iter()
979                .map(|c| c.trim().to_string())
980                .collect()
981        })
982        .collect();
983
984    if compact {
985        // Compact: cells joined by " | ", no padding, single-dash separators.
986        let render_row = |row: &[String]| -> String { format!("| {} |", row.join(" | ")) };
987        let mut lines = Vec::with_capacity(body.len() + 2);
988        lines.push(render_row(&header));
989        let sep: Vec<&str> = (0..num_cols).map(|_| "-").collect();
990        lines.push(format!("| {} |", sep.join(" | ")));
991        for row in &body {
992            lines.push(render_row(row));
993        }
994        return lines.join("\n");
995    }
996
997    // Display width (Unicode scalar count — good enough for now).
998    let dw = |s: &str| s.chars().count();
999
1000    // A column is right-aligned when at least one body cell is numeric and every
1001    // non-empty body cell is numeric — matching `tabulate`'s column typing, where
1002    // empty cells are "missing" (ignored) and a number may carry thousands
1003    // separators (`7,015`), which a plain `f64` parse rejects.
1004    let right: Vec<bool> = (0..num_cols)
1005        .map(|c| {
1006            let mut any = false;
1007            for row in &body {
1008                let t = row[c].trim();
1009                if t.is_empty() {
1010                    continue;
1011                }
1012                if !is_number_cell(t) {
1013                    return false;
1014                }
1015                any = true;
1016            }
1017            any
1018        })
1019        .collect();
1020
1021    // Column width = max(header_width + MIN_PADDING(2), max body-cell width).
1022    let width: Vec<usize> = (0..num_cols)
1023        .map(|c| {
1024            let mut w = dw(&header[c]) + 2;
1025            for row in &body {
1026                w = w.max(dw(&row[c]));
1027            }
1028            w
1029        })
1030        .collect();
1031
1032    let fmt_cell = |s: &str, c: usize| -> String {
1033        let pad = " ".repeat(width[c].saturating_sub(dw(s)));
1034        let body = if right[c] {
1035            format!("{pad}{s}")
1036        } else {
1037            format!("{s}{pad}")
1038        };
1039        format!(" {body} ")
1040    };
1041    let render_row = |row: &[String]| -> String {
1042        let cells: Vec<String> = (0..num_cols).map(|c| fmt_cell(&row[c], c)).collect();
1043        format!("|{}|", cells.join("|"))
1044    };
1045
1046    let mut lines = Vec::with_capacity(body.len() + 2);
1047    lines.push(render_row(&header));
1048    let sep: Vec<String> = (0..num_cols).map(|c| "-".repeat(width[c] + 2)).collect();
1049    lines.push(format!("|{}|", sep.join("|")));
1050    for row in &body {
1051        lines.push(render_row(row));
1052    }
1053    lines.join("\n")
1054}
1055
1056/// Escape a table cell so it can't break the markdown table: newlines become
1057/// spaces and pipes become the `&#124;` HTML entity (matches docling-core).
1058fn escape_cell(s: &str) -> String {
1059    s.replace('\n', " ").replace('|', "&#124;")
1060}
1061
1062#[cfg(test)]
1063mod tests {
1064    use super::*;
1065    use crate::{PictureImage, TableCell, TableStructure};
1066
1067    /// #385: where one list ends and the next begins is the backend's call
1068    /// (`first_in_list`), never the serializer's. An ordered run `1.` → `5.`
1069    /// is one list (an AsciiDoc numbered list around a nested one), and so are
1070    /// mixed bullet/ordered items the backend did not separate; only a flagged
1071    /// item opens a new list and earns the blank line.
1072    #[test]
1073    fn list_boundaries_come_from_the_backend_not_the_numbering() {
1074        let item = |ordered: bool, number: u64, first_in_list: bool, text: &str| Node::ListItem {
1075            ordered,
1076            number,
1077            first_in_list,
1078            text: text.into(),
1079            level: 0,
1080            marker: None,
1081            location: None,
1082            dclx: None,
1083            href: None,
1084            layer: None,
1085        };
1086        let md = |items: Vec<Node>| {
1087            let mut doc = DoclingDocument::new("t");
1088            for n in items {
1089                doc.push(n);
1090            }
1091            doc.export_to_markdown()
1092        };
1093        // A number gap alone is not a boundary.
1094        assert_eq!(
1095            md(vec![
1096                item(true, 1, true, "one"),
1097                item(true, 5, false, "five")
1098            ]),
1099            "1. one\n5. five\n"
1100        );
1101        // Nor is a kind flip the backend did not flag …
1102        assert_eq!(
1103            md(vec![
1104                item(false, 0, true, "bullet"),
1105                item(true, 1, false, "one"),
1106                item(false, 0, false, "bullet two"),
1107            ]),
1108            "- bullet\n1. one\n- bullet two\n"
1109        );
1110        // … while a flagged item is one, whatever its number says.
1111        assert_eq!(
1112            md(vec![
1113                item(true, 1, true, "a"),
1114                item(true, 2, false, "b"),
1115                item(true, 3, true, "new list, continuing count"),
1116            ]),
1117            "1. a\n2. b\n\n3. new list, continuing count\n"
1118        );
1119    }
1120
1121    #[test]
1122    fn renders_headings_paragraphs_and_lists() {
1123        let mut doc = DoclingDocument::new("demo");
1124        doc.add_heading(1, "Title");
1125        doc.add_paragraph("Hello world.");
1126        doc.push(Node::ListItem {
1127            ordered: false,
1128            number: 1,
1129            first_in_list: true,
1130            text: "first".into(),
1131            level: 0,
1132            marker: None,
1133            location: None,
1134            dclx: None,
1135            href: None,
1136            layer: None,
1137        });
1138        doc.push(Node::ListItem {
1139            ordered: false,
1140            number: 2,
1141            first_in_list: false,
1142            text: "second".into(),
1143            level: 0,
1144            marker: None,
1145            location: None,
1146            dclx: None,
1147            href: None,
1148            layer: None,
1149        });
1150        let md = doc.export_to_markdown();
1151        assert_eq!(md, "# Title\n\nHello world.\n\n- first\n- second\n");
1152    }
1153
1154    /// docling-core 2.92 (#721): a single newline inside an item's text is a
1155    /// GFM hard line break, a blank line stays a paragraph break, and a heading
1156    /// collapses its newline to a space. Nested-table dumps stay verbatim.
1157    #[test]
1158    fn single_newlines_become_gfm_hard_line_breaks() {
1159        let mut doc = DoclingDocument::new("t");
1160        doc.push(Node::Heading {
1161            level: 1,
1162            text: "Hello\nWorld".into(),
1163        });
1164        doc.push(Node::Paragraph {
1165            text: "line one\nline two\n\npara two".into(),
1166        });
1167        doc.push(Node::ListItem {
1168            ordered: false,
1169            number: 1,
1170            first_in_list: true,
1171            text: "item\ncontinued".into(),
1172            level: 0,
1173            marker: None,
1174            location: None,
1175            dclx: None,
1176            href: None,
1177            layer: None,
1178        });
1179        doc.push(Node::TextDump("A1 B1 \n\n\nC1".into()));
1180        assert_eq!(
1181            doc.export_to_markdown(),
1182            "# Hello World\n\nline one  \nline two\n\npara two\n\n- item  \ncontinued\n\nA1 B1 \n\n\nC1\n"
1183        );
1184    }
1185
1186    /// docling-core#540: inside a rich table cell a heading is plain text;
1187    /// docling-core#724: a field region renders only its items' key/value text.
1188    #[test]
1189    fn table_cell_mode_and_field_regions() {
1190        let mut doc = DoclingDocument::new("t");
1191        doc.push(Node::Heading {
1192            level: 2,
1193            text: "A  text".into(),
1194        });
1195        doc.push(Node::Paragraph {
1196            text: "body".into(),
1197        });
1198        assert_eq!(to_markdown_table_cell(&doc, false), "A  text\n\nbody");
1199        assert_eq!(doc.export_to_markdown(), "## A  text\n\nbody\n");
1200
1201        let mut doc = DoclingDocument::new("f");
1202        doc.push(Node::FieldRegion {
1203            items: vec![crate::FieldItem {
1204                marker: None,
1205                key: Some("Name:".into()),
1206                value: Some("John Doe".into()),
1207                value_kind: None,
1208            }],
1209        });
1210        assert_eq!(doc.export_to_markdown(), "Name:\n\nJohn Doe\n");
1211    }
1212
1213    #[test]
1214    fn strict_renders_recovered_links_legacy_does_not() {
1215        let mut doc = DoclingDocument::new("cv");
1216        doc.add_paragraph("Find me on LinkedIn or GitHub.");
1217        doc.links = vec![
1218            ("LinkedIn".into(), "https://www.linkedin.com/in/x/".into()),
1219            ("GitHub".into(), "https://github.com/x/".into()),
1220        ];
1221        // Legacy/docling mode: links are left untouched (conformance preserved).
1222        assert_eq!(doc.export_to_markdown(), "Find me on LinkedIn or GitHub.\n");
1223        // Strict mode: anchors become Markdown links.
1224        assert_eq!(
1225            doc.export_to_markdown_with(true),
1226            "Find me on [LinkedIn](https://www.linkedin.com/in/x/) or [GitHub](https://github.com/x/).\n"
1227        );
1228    }
1229
1230    #[test]
1231    fn strict_links_match_escaped_anchor_and_consume_in_order() {
1232        let mut doc = DoclingDocument::new("d");
1233        // The PDF assembler HTML-escapes prose, so by serialization time the body
1234        // already carries `&amp;`; the anchor is stored un-escaped. The matcher must
1235        // escape the anchor to find it. Two identical anchors link in document order.
1236        doc.add_paragraph("AI &amp; ML here, and issues here, then issues there.");
1237        doc.links = vec![
1238            ("AI & ML".into(), "https://a/".into()),
1239            ("issues".into(), "https://first/".into()),
1240            ("issues".into(), "https://second/".into()),
1241        ];
1242        assert_eq!(
1243            doc.export_to_markdown_with(true),
1244            "[AI &amp; ML](https://a/) here, and [issues](https://first/) here, then [issues](https://second/) there.\n"
1245        );
1246    }
1247
1248    /// docling-core#698: the referenced-image destination is percent-encoded —
1249    /// upstream's own case table (paths, Windows flavours, UNC, URLs) plus
1250    /// idempotency on the encoded result.
1251    #[test]
1252    fn referenced_image_destinations_are_escaped() {
1253        let cases = [
1254            (
1255                "doc_artifacts/image_000001_ab12.png",
1256                "doc_artifacts/image_000001_ab12.png",
1257            ),
1258            (
1259                "My Report_artifacts/img.png",
1260                "My%20Report_artifacts/img.png",
1261            ),
1262            ("artifacts/img (1).png", "artifacts/img%20%281%29.png"),
1263            ("100%_scale/a#b?c.png", "100%_scale/a%23b%3Fc.png"),
1264            ("/home/a b/img.png", "/home/a%20b/img.png"),
1265            (
1266                "My Report_artifacts\\img.png",
1267                "My%20Report_artifacts/img.png",
1268            ),
1269            (
1270                "C:/Users/me/My Docs/img.png",
1271                "file:///C:/Users/me/My%20Docs/img.png",
1272            ),
1273            ("C:\\Users\\me\\img.png", "file:///C:/Users/me/img.png"),
1274            (
1275                "//server/share/My Docs/img.png",
1276                "file://server/share/My%20Docs/img.png",
1277            ),
1278            ("\\\\server\\share\\img.png", "file://server/share/img.png"),
1279            ("file:///home/a b/img.png", "file:///home/a%20b/img.png"),
1280            (
1281                "s3://bucket/My Report_artifacts/img.png",
1282                "s3://bucket/My%20Report_artifacts/img.png",
1283            ),
1284            (
1285                "https://example.com:8080/a b.png?w=1&h=2#frag",
1286                "https://example.com:8080/a%20b.png?w=1&h=2#frag",
1287            ),
1288            (
1289                "https://example.com/img (1).png",
1290                "https://example.com/img%20%281%29.png",
1291            ),
1292            ("caf\u{e9}/im\u{e4}ge.png", "caf%C3%A9/im%C3%A4ge.png"),
1293        ];
1294        for (input, expected) in cases {
1295            assert_eq!(escape_uri_path(input), expected, "input {input:?}");
1296            assert_eq!(
1297                escape_uri_path(expected),
1298                expected,
1299                "idempotent {expected:?}"
1300            );
1301        }
1302        // The whole marker, through the referenced-image export.
1303        let mut doc = DoclingDocument::new("t");
1304        doc.push(Node::Picture {
1305            caption: None,
1306            caption_href: None,
1307            image: Some(PictureImage {
1308                dpi: PictureImage::DEFAULT_DPI,
1309                mimetype: "image/png".into(),
1310                width: 1,
1311                height: 1,
1312                data: b"x".to_vec(),
1313            }),
1314            classification: None,
1315            caption_parent: Default::default(),
1316        });
1317        let (md, files) = doc
1318            .export_to_markdown_with_images(ImageMode::Referenced, "My Report (final)_artifacts");
1319        assert!(
1320            md.contains("![Image](My%20Report%20%28final%29_artifacts/image_000000.png)"),
1321            "got:\n{md}"
1322        );
1323        // The file path handed back for writing stays unescaped.
1324        assert_eq!(files[0].0, "My Report (final)_artifacts/image_000000.png");
1325    }
1326
1327    /// Pictures the HTML backend folds into a list item print after the item
1328    /// line with plain newlines; a `<br>` newline in the item's own text is
1329    /// still a GFM hard line break.
1330    #[test]
1331    fn folded_list_item_pictures_keep_plain_newlines() {
1332        assert_eq!(
1333            list_item_text("Step\n<!-- image -->", false),
1334            "Step\n<!-- image -->"
1335        );
1336        assert_eq!(
1337            list_item_text("Step\nAlt text\n<!-- image -->\n<!-- image -->", false),
1338            "Step\nAlt text\n<!-- image -->\n<!-- image -->"
1339        );
1340        assert_eq!(
1341            list_item_text("line one\nline two", false),
1342            "line one  \nline two"
1343        );
1344    }
1345
1346    /// docling-core#723: the header block is the leading run of rows on which a
1347    /// `column_header` cell starts, flattened per column with " - ".
1348    #[test]
1349    fn stacked_header_rows_flatten_into_one() {
1350        let mut t = Table {
1351            rows: vec![
1352                vec!["".into(), "% of Total".into(), "% of Total".into()],
1353                vec!["class".into(), "Train".into(), "Test".into()],
1354                vec!["Caption".into(), "2.04".into(), "1.77".into()],
1355            ],
1356            ..Default::default()
1357        };
1358        t.structure = Some(TableStructure {
1359            header_row: vec![true, true, false],
1360            col_continuation: vec![
1361                vec![false, false, true],
1362                vec![false, false, false],
1363                vec![false, false, false],
1364            ],
1365            ..Default::default()
1366        });
1367        assert_eq!(t.header_row_count(), 2);
1368        assert_eq!(
1369            render_table(&t, true),
1370            "| class | % of Total - Train | % of Total - Test |\n| - | - | - |\n| Caption | 2.04 | 1.77 |"
1371        );
1372        // padded: widths from the flattened header, alignment from body rows
1373        assert_eq!(
1374            render_table(&t, false),
1375            "| class   |   % of Total - Train |   % of Total - Test |\n\
1376             |---------|----------------------|---------------------|\n\
1377             | Caption |                 2.04 |                1.77 |"
1378        );
1379    }
1380
1381    /// A header spanning two rows is repeated into the second row by the grid;
1382    /// that row is not a header row unless another header cell starts there.
1383    #[test]
1384    fn vertically_spanning_header_does_not_extend_the_block() {
1385        let mut t = Table {
1386            rows: vec![
1387                vec!["Name".into(), "Value".into()],
1388                vec!["Name".into(), "1".into()],
1389                vec!["x".into(), "2".into()],
1390            ],
1391            ..Default::default()
1392        };
1393        t.structure = Some(TableStructure {
1394            col_header: vec![vec![true, true], vec![true, false], vec![false, false]],
1395            row_continuation: vec![vec![false, false], vec![true, false], vec![false, false]],
1396            ..Default::default()
1397        });
1398        assert_eq!(t.header_row_count(), 1);
1399        assert_eq!(
1400            render_table(&t, true),
1401            "| Name | Value |\n| - | - |\n| Name | 1 |\n| x | 2 |"
1402        );
1403    }
1404
1405    /// Flags that begin on a later row promote nothing: every row stays in the
1406    /// body under an empty header row (tabulate's `headers=["", ""]`).
1407    #[test]
1408    fn header_flags_not_on_row_zero_keep_all_rows_in_the_body() {
1409        let mut t = Table {
1410            rows: vec![
1411                vec!["1".into(), "2".into()],
1412                vec!["a".into(), "b".into()],
1413                vec!["333".into(), "4".into()],
1414            ],
1415            ..Default::default()
1416        };
1417        t.structure = Some(TableStructure {
1418            header_row: vec![false, true, false],
1419            ..Default::default()
1420        });
1421        assert_eq!(t.header_row_count(), 0);
1422        assert_eq!(
1423            render_table(&t, false),
1424            "|     |    |\n|-----|----|\n| 1   | 2  |\n| a   | b  |\n| 333 | 4  |"
1425        );
1426    }
1427
1428    /// A pivot table's row headers (`<th rowspan>`) carry `row_header`, not
1429    /// `column_header` (docling#4216), so the data row beside them is not
1430    /// pulled into the header block — what this port used to reach with a
1431    /// deviation now falls out of the flags themselves.
1432    #[test]
1433    fn pivot_row_headers_do_not_extend_the_header() {
1434        let mut t = Table {
1435            rows: vec![
1436                vec!["Year".into(), "Month".into()],
1437                vec!["2025".into(), "January".into()],
1438                vec!["2025".into(), "February".into()],
1439            ],
1440            ..Default::default()
1441        };
1442        t.structure = Some(TableStructure {
1443            col_header: vec![vec![true, true], vec![false, false], vec![false, false]],
1444            row_header: vec![vec![false, false], vec![true, false], vec![true, false]],
1445            row_continuation: vec![vec![false, false], vec![false, false], vec![true, false]],
1446            ..Default::default()
1447        });
1448        assert_eq!(t.header_row_count(), 1);
1449        assert_eq!(
1450            render_table(&t, true),
1451            "| Year | Month |\n| - | - |\n| 2025 | January |\n| 2025 | February |"
1452        );
1453    }
1454
1455    /// No `column_header` anywhere (first-class cells without flags) → row 0
1456    /// stays the header, as before.
1457    #[test]
1458    fn unflagged_cells_keep_row_zero_as_header() {
1459        let mut t = Table {
1460            rows: vec![vec!["h".into()], vec!["d".into()]],
1461            ..Default::default()
1462        };
1463        t.cells = Some(
1464            [(0usize, "h"), (1, "d")]
1465                .into_iter()
1466                .map(|(r, text)| TableCell {
1467                    text: text.into(),
1468                    bbox: None,
1469                    start_row: r,
1470                    start_col: 0,
1471                    row_span: 1,
1472                    col_span: 1,
1473                    column_header: false,
1474                    row_header: false,
1475                    row_section: false,
1476                })
1477                .collect(),
1478        );
1479        assert_eq!(t.header_row_count(), 1);
1480        assert_eq!(render_table(&t, true), "| h |\n| - |\n| d |");
1481    }
1482
1483    #[test]
1484    fn renders_compact_table() {
1485        let mut doc = DoclingDocument::new("t");
1486        // The compact form is opt-in (the PDF backend sets it); default output uses
1487        // the padded GitHub serializer (covered by the regression fixtures).
1488        doc.compact_tables = true;
1489        doc.push(Node::Table(Table {
1490            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
1491            location: None,
1492            structure: None,
1493            cell_blocks: None,
1494            cells: None,
1495            caption: None,
1496            caption_parent: Default::default(),
1497        }));
1498        let md = doc.export_to_markdown();
1499        assert_eq!(md, "| a | b |\n| - | - |\n| 1 | 2 |\n");
1500    }
1501
1502    #[test]
1503    fn renders_padded_github_table_by_default() {
1504        let mut doc = DoclingDocument::new("t");
1505        doc.push(Node::Table(Table {
1506            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
1507            location: None,
1508            structure: None,
1509            cell_blocks: None,
1510            cells: None,
1511            caption: None,
1512            caption_parent: Default::default(),
1513        }));
1514        let md = doc.export_to_markdown();
1515        // Numeric data columns are right-aligned; columns padded to header+2.
1516        assert_eq!(md, "|   a |   b |\n|-----|-----|\n|   1 |   2 |\n");
1517    }
1518
1519    #[test]
1520    fn strict_unescapes_inline_underscores_legacy_keeps_them() {
1521        let mut doc = DoclingDocument::new("t");
1522        doc.add_heading(1, "a\\_b");
1523        doc.add_paragraph("x\\_y");
1524        doc.push(Node::ListItem {
1525            ordered: false,
1526            number: 1,
1527            first_in_list: true,
1528            text: "i\\_j".into(),
1529            level: 0,
1530            marker: None,
1531            location: None,
1532            dclx: None,
1533            href: None,
1534            layer: None,
1535        });
1536        // Legacy reproduces docling's `\_` escaping byte-for-byte.
1537        assert_eq!(doc.export_to_markdown(), "# a\\_b\n\nx\\_y\n\n- i\\_j\n");
1538        // Strict prefers literal underscores (Rust-only readability mode).
1539        assert_eq!(doc.export_to_markdown_with(true), "# a_b\n\nx_y\n\n- i_j\n");
1540    }
1541
1542    /// Drive a document's nodes through [`MarkdownStreamer`] in the given page
1543    /// splits and assert the concatenated chunks equal the buffered serializer.
1544    fn assert_stream_matches(
1545        doc: &DoclingDocument,
1546        strict: bool,
1547        images: ImageMode,
1548        splits: &[usize],
1549    ) {
1550        let (want, want_artifacts) = to_markdown_images(doc, strict, images, "artifacts");
1551        let mut streamer =
1552            MarkdownStreamer::with_artifacts(strict, images, doc.compact_tables, "artifacts")
1553                .with_page_break_placeholder(doc.page_break_placeholder.clone());
1554        let mut got = String::new();
1555        let mut got_artifacts = Vec::new();
1556        let mut start = 0;
1557        for &end in splits {
1558            // Links only matter in strict mode; feed them all with the first batch
1559            // that has content (document order is preserved by the queue).
1560            let links = if start == 0 {
1561                doc.links.as_slice()
1562            } else {
1563                &[]
1564            };
1565            got.push_str(&streamer.push(&doc.nodes[start..end], links));
1566            // Referenced mode: drain per push, as a real caller writing files
1567            // page by page would — numbering must continue across drains.
1568            got_artifacts.extend(streamer.take_artifacts());
1569            start = end;
1570        }
1571        got.push_str(&streamer.push(
1572            &doc.nodes[start..],
1573            if start == 0 {
1574                doc.links.as_slice()
1575            } else {
1576                &[]
1577            },
1578        ));
1579        got_artifacts.extend(streamer.take_artifacts());
1580        got.push_str(&streamer.finish());
1581        assert_eq!(
1582            got, want,
1583            "streamed output diverged (splits={splits:?}, strict={strict})"
1584        );
1585        assert_eq!(
1586            got_artifacts, want_artifacts,
1587            "streamed artifacts diverged (splits={splits:?}, strict={strict})"
1588        );
1589    }
1590
1591    #[test]
1592    fn streaming_is_byte_identical_to_buffered() {
1593        let mut doc = DoclingDocument::new("d");
1594        doc.add_heading(1, "Title");
1595        doc.add_paragraph("First paragraph.");
1596        doc.push(Node::ListItem {
1597            ordered: false,
1598            number: 1,
1599            first_in_list: true,
1600            text: "a".into(),
1601            level: 0,
1602            marker: None,
1603            location: None,
1604            dclx: None,
1605            href: None,
1606            layer: None,
1607        });
1608        doc.push(Node::ListItem {
1609            ordered: false,
1610            number: 2,
1611            first_in_list: false,
1612            text: "b".into(),
1613            level: 0,
1614            marker: None,
1615            location: None,
1616            dclx: None,
1617            href: None,
1618            layer: None,
1619        });
1620        doc.push(Node::Code {
1621            language: Some("rust".into()),
1622            text: "let x = 1;".into(),
1623            orig: None,
1624            pretty: None,
1625        });
1626        doc.push(Node::Table(Table {
1627            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
1628            location: None,
1629            structure: None,
1630            cell_blocks: None,
1631            cells: None,
1632            caption: None,
1633            caption_parent: Default::default(),
1634        }));
1635        doc.push(Node::Picture {
1636            caption: Some("Fig 1".into()),
1637            caption_href: None,
1638            image: Some(PictureImage {
1639                dpi: PictureImage::DEFAULT_DPI,
1640                mimetype: "image/png".into(),
1641                width: 2,
1642                height: 2,
1643                data: b"png-one".to_vec(),
1644            }),
1645            classification: None,
1646            caption_parent: Default::default(),
1647        });
1648        doc.add_paragraph("Last paragraph.");
1649        // A second embedded picture, so referenced mode must keep numbering
1650        // (`image_000001`) across chunk boundaries.
1651        doc.push(Node::Picture {
1652            caption: None,
1653            caption_href: None,
1654            image: Some(PictureImage {
1655                dpi: PictureImage::DEFAULT_DPI,
1656                mimetype: "image/png".into(),
1657                width: 2,
1658                height: 2,
1659                data: b"png-two".to_vec(),
1660            }),
1661            classification: None,
1662            caption_parent: Default::default(),
1663        });
1664
1665        // A run of list items must never straddle a split, so try splits that fall
1666        // on safe block boundaries (the streaming PDF assembler guarantees this).
1667        for &strict in &[false, true] {
1668            for &images in &[
1669                ImageMode::Placeholder,
1670                ImageMode::Embedded,
1671                ImageMode::Referenced,
1672            ] {
1673                for splits in [&[][..], &[1][..], &[2][..], &[4][..], &[1, 4, 6, 7][..]] {
1674                    assert_stream_matches(&doc, strict, images, splits);
1675                }
1676            }
1677        }
1678    }
1679
1680    #[test]
1681    fn streaming_applies_recovered_links_in_strict_mode() {
1682        let mut doc = DoclingDocument::new("d");
1683        doc.add_paragraph("See LinkedIn for details.");
1684        doc.add_paragraph("And GitHub too.");
1685        doc.links = vec![
1686            ("LinkedIn".into(), "https://lnkd/".into()),
1687            ("GitHub".into(), "https://gh/".into()),
1688        ];
1689        // The second anchor lives in the second block, so it must be carried across
1690        // the page boundary and placed when that block streams out.
1691        assert_stream_matches(&doc, true, ImageMode::Placeholder, &[1]);
1692    }
1693
1694    /// A three-page document with one empty page in the middle and page
1695    /// markers of both kinds, as the backends emit them.
1696    fn paged_doc() -> DoclingDocument {
1697        let mut doc = DoclingDocument::new("p");
1698        doc.push(Node::PageInfo {
1699            page_no: 1,
1700            width: 100.0,
1701            height: 100.0,
1702        });
1703        doc.add_heading(1, "Title");
1704        doc.add_paragraph("Page one.");
1705        // Page two: a marker plus furniture only — renders nothing.
1706        doc.push(Node::PageBreak);
1707        doc.push(Node::PageInfo {
1708            page_no: 2,
1709            width: 100.0,
1710            height: 100.0,
1711        });
1712        doc.push(Node::PageFurniture {
1713            footer: true,
1714            location: [0, 500, 511, 511],
1715            text: "2".into(),
1716        });
1717        doc.push(Node::PageBreak);
1718        doc.push(Node::PageInfo {
1719            page_no: 3,
1720            width: 100.0,
1721            height: 100.0,
1722        });
1723        doc.add_paragraph("Page three.");
1724        // A trailing boundary with nothing after it.
1725        doc.push(Node::PageBreak);
1726        doc
1727    }
1728
1729    #[test]
1730    fn page_break_placeholder_lands_between_pages_only() {
1731        let mut doc = paged_doc();
1732        // Off by default: docling's Markdown carries no page breaks.
1733        assert_eq!(
1734            doc.export_to_markdown(),
1735            "# Title\n\nPage one.\n\nPage three.\n"
1736        );
1737        doc.page_break_placeholder = Some("<!-- page break -->".into());
1738        // One break for the 1→3 transition (the empty page 2 and the doubled
1739        // PageBreak+PageInfo markers collapse), none before the first block,
1740        // none for the trailing boundary.
1741        assert_eq!(
1742            doc.export_to_markdown(),
1743            "# Title\n\nPage one.\n\n<!-- page break -->\n\nPage three.\n"
1744        );
1745        // An empty placeholder is still a (blank) part, as upstream's
1746        // `str.replace(marker, "")` leaves the delimiters around it.
1747        doc.page_break_placeholder = Some(String::new());
1748        assert_eq!(
1749            doc.export_to_markdown(),
1750            "# Title\n\nPage one.\n\n\n\nPage three.\n"
1751        );
1752    }
1753
1754    #[test]
1755    fn page_break_placeholder_never_leads_a_single_page() {
1756        let mut doc = DoclingDocument::new("one");
1757        doc.page_break_placeholder = Some("---".into());
1758        doc.push(Node::PageBreak);
1759        doc.push(Node::PageInfo {
1760            page_no: 1,
1761            width: 10.0,
1762            height: 10.0,
1763        });
1764        doc.add_paragraph("Only page.");
1765        assert_eq!(doc.export_to_markdown(), "Only page.\n");
1766        // Two boundaries with no content between them: still one break.
1767        doc.push(Node::PageBreak);
1768        doc.push(Node::PageBreak);
1769        doc.add_paragraph("Next.");
1770        assert_eq!(doc.export_to_markdown(), "Only page.\n\n---\n\nNext.\n");
1771    }
1772
1773    #[test]
1774    fn page_break_placeholder_streams_byte_identical() {
1775        let mut doc = paged_doc();
1776        doc.page_break_placeholder = Some("<!-- page break -->".into());
1777        // Split at every page marker (how the PDF pipeline pushes page batches)
1778        // and at odd places inside a page: the pending break must survive a
1779        // push that renders nothing (page two) and land on page three's block.
1780        for splits in [
1781            &[3usize][..],
1782            &[3, 6],
1783            &[3, 6, 8],
1784            &[1, 2, 3, 4, 5, 6, 7, 8, 9],
1785            &[8],
1786        ] {
1787            assert_stream_matches(&doc, false, ImageMode::Placeholder, splits);
1788            assert_stream_matches(&doc, true, ImageMode::Placeholder, splits);
1789        }
1790    }
1791
1792    #[test]
1793    fn strict_tightens_punctuation_spacing_legacy_keeps_it() {
1794        let mut doc = DoclingDocument::new("t");
1795        doc.add_paragraph("see [ 37 , 36 ] and ( x ) .");
1796        // Legacy keeps docling's spacing byte-for-byte.
1797        assert_eq!(doc.export_to_markdown(), "see [ 37 , 36 ] and ( x ) .\n");
1798        // Strict tightens punctuation for readable Markdown.
1799        assert_eq!(doc.export_to_markdown_with(true), "see [37, 36] and (x).\n");
1800    }
1801}