Skip to main content

docling_core/
markdown.rs

1//! Markdown serializer for [`DoclingDocument`].
2
3use crate::document::{DoclingDocument, Node, Table};
4
5/// What docling's Markdown serializer writes for an item it has no component
6/// for — a `KeyValueItem` (the XBRL fact graph) is the one such item a backend
7/// produces.
8const MISSING_KEY_VALUE_ITEM: &str = "<!-- missing-key-value-item -->";
9
10/// How pictures are rendered (mirrors docling-core's `ImageRefMode`).
11#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
12pub enum ImageMode {
13    /// `<!-- image -->` (docling's default, and the only mode without image data).
14    #[default]
15    Placeholder,
16    /// `![Image](data:<mime>;base64,…)` — self-contained.
17    Embedded,
18    /// `![Image](<artifacts>/image_NNNNNN.<ext>)`; the bytes are returned for the
19    /// caller to write.
20    Referenced,
21}
22
23/// Serializer state threaded through the render walk.
24struct Ctx {
25    strict: bool,
26    /// Emit compact `| a | b |` tables instead of the padded GitHub serializer.
27    compact_tables: bool,
28    images: ImageMode,
29    artifacts_dir: String,
30    /// (relative path, bytes) for each referenced image — written by the caller.
31    artifacts: Vec<(String, Vec<u8>)>,
32    pic_index: usize,
33    /// Rendering the block content of a rich table cell (docling-core 2.94's
34    /// `in_table_cell`, docling-core#540): a heading has no valid Markdown
35    /// form inside a table, so it renders as plain text without `#` markers.
36    in_table_cell: bool,
37    /// docling-core's `MarkdownParams.page_break_placeholder`: the text that
38    /// separates two pages ([`DoclingDocument::page_break_placeholder`]).
39    /// `None` omits page breaks, docling's default.
40    page_break: Option<String>,
41    /// A page boundary has been crossed since the last rendered block, so the
42    /// next block is preceded by the placeholder. docling yields its
43    /// `_PageBreakNode` between two *items* whose `prov.page_no` differ, so a
44    /// boundary before the first block or after the last one emits nothing,
45    /// and a run of empty pages collapses into a single break.
46    pending_page_break: bool,
47    /// Whether any block has been rendered yet — for a streamer, across every
48    /// earlier push too — which is what makes a boundary a *pending* break.
49    emitted_any: bool,
50}
51
52/// Render a document to a Markdown string (pictures as placeholders).
53///
54/// `strict` selects the serializer-level behaviours that differ between
55/// docling-legacy output and cleaner Markdown — currently the code-fence
56/// language (legacy drops it, strict keeps it).
57pub fn to_markdown(doc: &DoclingDocument, strict: bool) -> String {
58    to_markdown_images(doc, strict, ImageMode::Placeholder, "artifacts").0
59}
60
61/// Render to Markdown with an explicit picture [`ImageMode`]. Returns the
62/// Markdown and, for [`ImageMode::Referenced`], the `(path, bytes)` of each image
63/// the caller should write (relative to the Markdown file).
64pub fn to_markdown_images(
65    doc: &DoclingDocument,
66    strict: bool,
67    images: ImageMode,
68    artifacts_dir: &str,
69) -> (String, Vec<(String, Vec<u8>)>) {
70    let mut ctx = Ctx {
71        strict,
72        compact_tables: doc.compact_tables,
73        images,
74        artifacts_dir: artifacts_dir.to_string(),
75        artifacts: Vec::new(),
76        pic_index: 0,
77        in_table_cell: false,
78        page_break: doc.page_break_placeholder.clone(),
79        pending_page_break: false,
80        emitted_any: false,
81    };
82    let mut blocks: Vec<String> = Vec::new();
83    render(&doc.nodes, &mut blocks, &mut ctx);
84    let mut body = blocks.join("\n\n");
85    // Strict mode only: turn recovered source hyperlinks into Markdown links.
86    // docling's standard pipeline drops them, so doing this in legacy mode would
87    // diverge from docling — hence strict-only, leaving conformance output intact.
88    if strict && !doc.links.is_empty() {
89        body = apply_links(&body, &doc.links);
90    }
91    let md = if body.is_empty() {
92        String::new()
93    } else {
94        format!("{body}\n")
95    };
96    (md, ctx.artifacts)
97}
98
99/// Render the block content of a *rich table cell* to Markdown — what
100/// docling-core's table serializer does for a `RichTableCell`
101/// (`doc_serializer.serialize(item, in_table_cell=True)`): the cell's
102/// paragraphs, lists and flattened nested tables render as in a document, but a
103/// heading loses its `#` markers (docling-core#540 — the Markdown spec has no
104/// headings inside tables). Pictures stay placeholders. The caller flattens the
105/// result into its cell text; the table serializer later turns the newlines
106/// into spaces.
107pub fn to_markdown_table_cell(doc: &DoclingDocument, strict: bool) -> String {
108    let mut ctx = Ctx {
109        strict,
110        compact_tables: doc.compact_tables,
111        images: ImageMode::Placeholder,
112        artifacts_dir: String::new(),
113        artifacts: Vec::new(),
114        pic_index: 0,
115        in_table_cell: true,
116        // A rich cell is one page's content; its sub-document carries no
117        // page boundaries and docling's `_iterate_items` runs the page-break
118        // scan over the document root only.
119        page_break: None,
120        pending_page_break: false,
121        emitted_any: false,
122    };
123    let mut blocks: Vec<String> = Vec::new();
124    render(&doc.nodes, &mut blocks, &mut ctx);
125    blocks.join("\n\n")
126}
127
128/// Wrap each recovered link's anchor text in Markdown `[anchor](href)`. Anchors
129/// arrive cleaned (curly quotes/dashes already normalized) but un-escaped, so we
130/// match against the body's HTML-escaped (`&`/`<`/`>`) form, the way prose nodes
131/// were serialized. Links are consumed in document order from a moving cursor, so
132/// a repeated anchor (e.g. two "issues") links its successive occurrences rather
133/// than all pointing at the first. An anchor that can't be located is skipped
134/// (its text may have been split across a line wrap or table cell).
135fn apply_links(body: &str, links: &[(String, String)]) -> String {
136    let mut out = body.to_string();
137    let mut cursor = 0usize;
138    for (anchor, href) in links {
139        let anchor = anchor
140            .replace('&', "&amp;")
141            .replace('<', "&lt;")
142            .replace('>', "&gt;");
143        if anchor.is_empty() {
144            continue;
145        }
146        if let Some(rel) = out[cursor..].find(&anchor) {
147            let at = cursor + rel;
148            // Don't relink inside an already-emitted `](` Markdown link target.
149            let replacement = format!("[{anchor}]({href})");
150            out.replace_range(at..at + anchor.len(), &replacement);
151            cursor = at + replacement.len();
152        }
153    }
154    out
155}
156
157/// Like [`apply_links`] but over a single chunk, consuming from a shared queue so
158/// the same `[anchor](href)` rewriting can be applied incrementally as Markdown is
159/// streamed out. Each queued link is matched (in document order) against `chunk`
160/// and rewritten in place; a link whose anchor is not in this chunk is carried
161/// forward in the queue for a later chunk. Anchors are recovered in document
162/// order and a chunk is always a contiguous run of whole blocks, so this
163/// reproduces [`apply_links`]' single moving cursor: the link lands in whichever
164/// chunk contains its anchor, identically to the buffered path. (A link whose
165/// anchor never appears is carried to the end and dropped — the same no-op
166/// `apply_links` performs for an unlocatable anchor.)
167fn apply_links_chunk(chunk: &str, queue: &mut Vec<(String, String)>) -> String {
168    let mut out = chunk.to_string();
169    let mut cursor = 0usize;
170    let mut carried: Vec<(String, String)> = Vec::new();
171    for (anchor_raw, href) in std::mem::take(queue) {
172        let anchor = anchor_raw
173            .replace('&', "&amp;")
174            .replace('<', "&lt;")
175            .replace('>', "&gt;");
176        if anchor.is_empty() {
177            continue;
178        }
179        if let Some(rel) = out[cursor..].find(&anchor) {
180            let at = cursor + rel;
181            let replacement = format!("[{anchor}]({href})");
182            out.replace_range(at..at + anchor.len(), &replacement);
183            cursor = at + replacement.len();
184        } else {
185            // Not in this chunk; try again when its block is flushed.
186            carried.push((anchor_raw, href));
187        }
188    }
189    *queue = carried;
190    out
191}
192
193/// Incremental Markdown serializer: feed finalized, in-document-order batches of
194/// [`Node`]s and receive Markdown chunks whose concatenation is **byte-identical**
195/// to [`to_markdown_images`] over the same nodes. This is the streaming
196/// counterpart of the buffered serializer — used to emit a document's Markdown in
197/// chunks (e.g. page by page, as the parallel PDF pipeline finishes pages) instead
198/// of building the whole string up front.
199///
200/// [`ImageMode::Placeholder`] and [`ImageMode::Embedded`] render inline.
201/// [`ImageMode::Referenced`] additionally hands each picture's bytes out through
202/// [`take_artifacts`](Self::take_artifacts) — construct with
203/// [`with_artifacts`](Self::with_artifacts) and drain after every push so the
204/// bytes can be written to disk as pages finish instead of accumulating for the
205/// whole document (issue #80's memory-bounded image handling).
206///
207/// Each [`push`](Self::push) must contain whole blocks in reading order: a caller
208/// must not split a run of list items across two pushes (the run would render as
209/// two separate lists). Finalized PDF page batches already satisfy this.
210pub struct MarkdownStreamer {
211    strict: bool,
212    images: ImageMode,
213    compact_tables: bool,
214    /// Whether any non-empty chunk has been emitted yet (drives `\n\n` joins and
215    /// the trailing newline).
216    emitted_any: bool,
217    /// Recovered links not yet placed (strict mode), consumed in document order.
218    links: Vec<(String, String)>,
219    /// Referenced mode: the link prefix, the not-yet-drained `(path, bytes)`
220    /// artifacts, and the running image number (continues across pushes so the
221    /// stream matches the buffered serializer's `image_000000…` numbering).
222    artifacts_dir: String,
223    artifacts: Vec<(String, Vec<u8>)>,
224    pic_index: usize,
225    /// [`DoclingDocument::page_break_placeholder`] and the boundary carried
226    /// over from the previous push (a page batch opens with its page marker,
227    /// so the break it implies is paid by that batch's first block).
228    page_break: Option<String>,
229    pending_page_break: bool,
230}
231
232impl MarkdownStreamer {
233    /// Create a streamer. `compact_tables` mirrors [`DoclingDocument::compact_tables`].
234    /// For [`ImageMode::Referenced`] use [`with_artifacts`](Self::with_artifacts).
235    pub fn new(strict: bool, images: ImageMode, compact_tables: bool) -> Self {
236        debug_assert!(
237            images != ImageMode::Referenced,
238            "referenced image mode needs an artifacts dir; use with_artifacts"
239        );
240        Self::with_artifacts(strict, images, compact_tables, "artifacts")
241    }
242
243    /// Like [`new`](Self::new) but with the artifacts link prefix, allowing
244    /// [`ImageMode::Referenced`]: pictures render as
245    /// `![Image](<artifacts_dir>/image_NNNNNN.<ext>)` and each push's image
246    /// bytes wait in [`take_artifacts`](Self::take_artifacts) for the caller to
247    /// write. The concatenated chunks and the artifact list match the buffered
248    /// [`to_markdown_images`] byte-for-byte.
249    pub fn with_artifacts(
250        strict: bool,
251        images: ImageMode,
252        compact_tables: bool,
253        artifacts_dir: &str,
254    ) -> Self {
255        Self {
256            strict,
257            images,
258            compact_tables,
259            emitted_any: false,
260            links: Vec::new(),
261            artifacts_dir: artifacts_dir.to_string(),
262            artifacts: Vec::new(),
263            pic_index: 0,
264            page_break: None,
265            pending_page_break: false,
266        }
267    }
268
269    /// Insert `placeholder` between pages, mirroring
270    /// [`DoclingDocument::page_break_placeholder`] for the buffered path (the
271    /// concatenated chunks stay byte-identical to it). `None` — the default —
272    /// omits page breaks. Set before the first [`push`](Self::push).
273    pub fn with_page_break_placeholder(mut self, placeholder: Option<String>) -> Self {
274        self.page_break = placeholder;
275        self
276    }
277
278    /// The `(relative path, bytes)` of images rendered by pushes since the last
279    /// drain ([`ImageMode::Referenced`] only — empty otherwise). Paths are
280    /// relative to the Markdown file, i.e. they start with the configured
281    /// artifacts dir.
282    pub fn take_artifacts(&mut self) -> Vec<(String, Vec<u8>)> {
283        std::mem::take(&mut self.artifacts)
284    }
285
286    /// Render one finalized batch of nodes (plus any links recovered from the same
287    /// span, in document order) into the next Markdown chunk. Returns an empty
288    /// string when the batch produces no output (e.g. empty tables/pictures), in
289    /// which case nothing should be written.
290    pub fn push(&mut self, nodes: &[Node], links: &[(String, String)]) -> String {
291        self.links.extend(links.iter().cloned());
292        let mut ctx = Ctx {
293            strict: self.strict,
294            compact_tables: self.compact_tables,
295            images: self.images,
296            artifacts_dir: std::mem::take(&mut self.artifacts_dir),
297            artifacts: std::mem::take(&mut self.artifacts),
298            pic_index: self.pic_index,
299            in_table_cell: false,
300            page_break: std::mem::take(&mut self.page_break),
301            pending_page_break: self.pending_page_break,
302            emitted_any: self.emitted_any,
303        };
304        let mut blocks: Vec<String> = Vec::new();
305        render(nodes, &mut blocks, &mut ctx);
306        self.artifacts_dir = std::mem::take(&mut ctx.artifacts_dir);
307        self.artifacts = std::mem::take(&mut ctx.artifacts);
308        self.pic_index = ctx.pic_index;
309        self.page_break = std::mem::take(&mut ctx.page_break);
310        self.pending_page_break = ctx.pending_page_break;
311        if blocks.is_empty() {
312            return String::new();
313        }
314        let mut body = blocks.join("\n\n");
315        if self.strict && !self.links.is_empty() {
316            body = apply_links_chunk(&body, &mut self.links);
317        }
318        let chunk = if self.emitted_any {
319            format!("\n\n{body}")
320        } else {
321            body
322        };
323        self.emitted_any = true;
324        chunk
325    }
326
327    /// Emit the trailing newline that finishes the document (empty if no content
328    /// was produced). Call exactly once, after the final [`push`](Self::push).
329    pub fn finish(self) -> String {
330        if self.emitted_any {
331            "\n".to_string()
332        } else {
333            String::new()
334        }
335    }
336}
337
338/// In `strict` mode, rewrite inline text for readability rather than byte-for-byte
339/// docling fidelity: undo the legacy `\_` underscore escaping, and tighten stray
340/// spaces around punctuation (`[ 37 , 36 ]` → `[37, 36]`, `( x )` → `(x)`). This
341/// cleans up both the PDF backend's glyph-split spacing and the space the legacy
342/// emphasis serialization leaves before punctuation (`*a* ,` → `*a*,`).
343/// Legacy/default output keeps docling's spacing untouched. Only inline text
344/// nodes pass through here — code blocks and table cells are left alone.
345fn strict_text(text: &str, strict: bool) -> String {
346    if !strict {
347        return text.to_string();
348    }
349    text.replace("\\_", "_")
350        .replace(" ,", ",")
351        .replace(" .", ".")
352        .replace(" ;", ";")
353        .replace(" )", ")")
354        .replace("( ", "(")
355        .replace(" ]", "]")
356        .replace("[ ", "[")
357}
358
359/// docling-core 2.92's `_md_line_breaks` (docling-core#721): a single `\n`
360/// inside an item's text becomes a GFM hard line break (`"  \n"`, two trailing
361/// spaces) so renderers honour it; a blank line (`\n\n`) is a paragraph break
362/// and stays as is — the document scope already joins blocks with `\n\n`.
363/// Applied to body text, list items and captions, never to code/formulas.
364fn md_line_breaks(text: &str) -> String {
365    if !text.contains('\n') {
366        return text.to_string();
367    }
368    text.split("\n\n")
369        .map(|para| para.replace('\n', "  \n"))
370        .collect::<Vec<_>>()
371        .join("\n\n")
372}
373
374/// Undo [`md_line_breaks`] on a rich table cell's flattened Markdown so the
375/// non-Markdown exports (JSON `text`, LaTeX cells) see the cell's raw line
376/// breaks, as docling's do — a rich cell's text is its Markdown serialization
377/// in our model, and the two trailing spaces are a Markdown-only marker.
378pub(crate) fn strip_hard_breaks(text: &str) -> String {
379    if text.contains("  \n") {
380        text.replace("  \n", "\n")
381    } else {
382        text.to_string()
383    }
384}
385
386/// docling-core's `_heading_line_breaks`: a GFM heading cannot span lines, so a
387/// newline inside heading text collapses to a space (`# Hello World`, not a
388/// broken `# Hello\nWorld`).
389fn heading_line_breaks(text: &str) -> String {
390    text.replace('\n', " ")
391}
392
393fn render(nodes: &[Node], blocks: &mut Vec<String>, ctx: &mut Ctx) {
394    let mut i = 0;
395    while i < nodes.len() {
396        let before = blocks.len();
397        match &nodes[i] {
398            // A page boundary: the explicit `PageBreak` (slides, DjVu / DocTags
399            // pages) or the `PageInfo` marker that opens every PDF page and
400            // spreadsheet sheet — a sheet boundary carries both, and the flag
401            // absorbs the pair into one break. docling's `_iterate_items`
402            // yields a `_PageBreakNode` only between two items on different
403            // pages (a group's leading item counts for the group), which is
404            // exactly "a boundary between two rendered blocks": nothing before
405            // the first block, nothing after the last, consecutive boundaries
406            // — empty or furniture-only pages — collapsed into one.
407            Node::PageBreak | Node::PageInfo { .. } => {
408                if ctx.page_break.is_some() && ctx.emitted_any {
409                    ctx.pending_page_break = true;
410                }
411                i += 1;
412            }
413            Node::ListItem { .. } => {
414                let start = i;
415                i += 1;
416                loop {
417                    match nodes.get(i) {
418                        Some(Node::ListItem { .. }) => i += 1,
419                        // An empty paragraph between two list items is absorbed
420                        // into the run — docling keeps such a ListGroup
421                        // contiguous rather than splitting it.
422                        Some(Node::Paragraph { text })
423                            if text.is_empty()
424                                && matches!(nodes.get(i + 1), Some(Node::ListItem { .. })) =>
425                        {
426                            i += 1
427                        }
428                        _ => break,
429                    }
430                }
431                render_list_run(&nodes[start..i], blocks, ctx.strict);
432            }
433            other => {
434                render_one(other, blocks, ctx);
435                i += 1;
436            }
437        }
438        if blocks.len() > before {
439            if ctx.pending_page_break {
440                // The placeholder is a block of its own, joined by the document
441                // delimiter like docling's `_PageBreakSerResult` part — an
442                // empty placeholder therefore leaves the doubled `\n\n`
443                // upstream leaves too.
444                if let Some(placeholder) = &ctx.page_break {
445                    blocks.insert(before, placeholder.clone());
446                }
447                ctx.pending_page_break = false;
448            }
449            ctx.emitted_any = true;
450        }
451    }
452}
453
454/// Render a contiguous run of list items.
455///
456/// Ordered items use their explicit `number`. A new sibling list (marked by
457/// `first_in_list`) at the same depth is separated by a blank line, matching
458/// docling-core's serializer.
459fn render_list_run(items: &[Node], blocks: &mut Vec<String>, strict: bool) {
460    let mut lines: Vec<String> = Vec::new();
461    // Whether a top-level item has been rendered yet — a fresh-list flag on
462    // the very first item opens nothing.
463    let mut any_top = false;
464
465    for item in items {
466        let Node::ListItem {
467            ordered,
468            number,
469            first_in_list,
470            text,
471            level,
472            marker: orig_marker,
473            location: _,
474            dclx: _,
475            href: _,
476            layer,
477        } = item
478        else {
479            continue;
480        };
481        // A non-body (furniture) list item is omitted from Markdown, matching
482        // docling's content-layer filtering.
483        if layer.is_some() {
484            continue;
485        }
486        let level = *level as usize;
487
488        // A new sibling list at the top level gets a blank line — and only the
489        // backend knows where one starts (`first_in_list`: Word's `numId`
490        // changing, an HTML `<ul>` closing, a Markdown bullet switching
491        // `-`→`*`). The serializer used to guess it from a kind flip or a
492        // number gap as well, which split lists docling keeps whole (an
493        // AsciiDoc `1.` … `5.`, mixed `*`/`1.` markers) — #385. Only at the
494        // top level: nested sibling groups are children of a list item, and
495        // docling joins an item's children without blank lines.
496        if level == 0 {
497            if any_top && *first_in_list {
498                lines.push(String::new());
499            }
500            any_top = true;
501        }
502
503        let indent = "    ".repeat(level);
504        // docling-core's `case_already_valid`: a marker of digits and a dot
505        // prints verbatim — Python's `\d+\.` admits every Unicode decimal
506        // digit, so a DOCX `decimalFullWidth` marker (`1.`, docling#4336)
507        // is kept as it is rather than renumbered in ASCII.
508        let verbatim = orig_marker.as_deref().filter(|m| {
509            m.strip_suffix('.')
510                .is_some_and(|d| !d.is_empty() && d.chars().all(char::is_numeric))
511        });
512        let marker = match verbatim {
513            Some(m) if *ordered => m.to_string(),
514            _ if *ordered => format!("{number}."),
515            _ => "-".to_string(),
516        };
517        lines.push(format!("{indent}{marker} {}", list_item_text(text, strict)));
518    }
519
520    // A run consisting only of furniture (content-layer-filtered) items yields no
521    // lines; pushing an empty block here would surface as a stray blank line.
522    if !lines.is_empty() {
523        blocks.push(lines.join("\n"));
524    }
525}
526
527/// A list item's Markdown body. The GFM hard-line-break rule (docling-core#721)
528/// applies to the item's own text; pictures the HTML backend folded into the
529/// item (`"\n[alt\n]<!-- image -->"` per `<img>` inside the `<li>`) are
530/// docling's picture *children* of the item, which its serializer prints after
531/// the item line with plain newlines — so a folded tail keeps its newlines
532/// unmarked. The tail is recognised structurally: every line after the first is
533/// an image marker or an alt caption directly followed by one.
534fn list_item_text(text: &str, strict: bool) -> String {
535    let escaped = strict_text(text, strict);
536    if let Some((own, tail)) = escaped.split_once('\n') {
537        if is_folded_child_tail(tail) {
538            return format!("{}\n{tail}", md_line_breaks(own));
539        }
540    }
541    md_line_breaks(&escaped)
542}
543
544/// Whether everything after a list item's own first line is a folded *child*
545/// block rather than a continuation of the item's text: an image marker
546/// (optionally preceded by its caption/alt line) or a fenced code block. The
547/// AsciiDoc backend indents such a block to the item's own depth (as
548/// docling-core's list serializer does for each part it emits), so a leading
549/// indent is ignored here.
550fn is_folded_child_tail(tail: &str) -> bool {
551    const MARKER: &str = "<!-- image -->";
552    const FENCE: &str = "```";
553    let mut lines = tail.split('\n').peekable();
554    let mut any = false;
555    while let Some(line) = lines.next() {
556        let line = line.trim_start();
557        if line == MARKER {
558            any = true;
559        } else if line == FENCE {
560            // Skip the block's body; an unclosed fence is not a folded child.
561            loop {
562                match lines.next() {
563                    Some(l) if l.trim_start() == FENCE => break,
564                    Some(_) => {}
565                    None => return false,
566                }
567            }
568            any = true;
569        } else if lines.next().map(str::trim_start) == Some(MARKER) {
570            any = true; // an alt caption line, then its marker
571        } else {
572            return false;
573        }
574    }
575    any
576}
577
578fn render_one(node: &Node, blocks: &mut Vec<String>, ctx: &mut Ctx) {
579    match node {
580        Node::Heading { level, text } => {
581            let text = heading_line_breaks(&strict_text(text, ctx.strict));
582            if ctx.in_table_cell {
583                // docling-core#540: no `#` markers inside a table cell.
584                blocks.push(text);
585            } else {
586                let hashes = "#".repeat((*level).clamp(1, 6) as usize);
587                blocks.push(format!("{hashes} {text}"));
588            }
589        }
590        // An empty body paragraph (docling's blank-line text item) contributes
591        // nothing to Markdown — only DocLang/JSON keep it.
592        Node::Paragraph { text } if text.is_empty() => {}
593        Node::Paragraph { text } => blocks.push(md_line_breaks(&strict_text(text, ctx.strict))),
594        // A standalone caption item renders like a text item; its hyperlink
595        // annotation becomes a Markdown link around the whole caption.
596        Node::Caption { text, .. } if text.is_empty() => {}
597        Node::Caption { text, href } => {
598            let body = md_line_breaks(&strict_text(text, ctx.strict));
599            blocks.push(match href {
600                Some(url) => format!("[{body}]({url})"),
601                None => body,
602            });
603        }
604        Node::CheckboxItem { checked, text } => {
605            let mark = if *checked { "- [x] " } else { "- [ ] " };
606            blocks.push(md_line_breaks(&strict_text(
607                &format!("{mark}{text}"),
608                ctx.strict,
609            )));
610        }
611        Node::Code {
612            language,
613            text,
614            pretty,
615            ..
616        } => {
617            // Legacy docling never emits a language on the fence; strict keeps it.
618            let lang = match language {
619                Some(l) if ctx.strict => l.as_str(),
620                _ => "",
621            };
622            // Strict prefers the line-preserving rendering when the backend
623            // supplied one (PDF); legacy stays on docling's flat `text`.
624            let body = match pretty {
625                Some(p) if ctx.strict => p.as_str(),
626                _ => text.as_str(),
627            };
628            blocks.push(format!("```{lang}\n{body}\n```"));
629        }
630        // A CodeFormula-decoded display formula renders as docling's `$$…$$`
631        // (the un-enriched pipeline emits a placeholder paragraph instead).
632        Node::Formula { latex, .. } => blocks.push(format!("$${latex}$$")),
633        Node::Table(table) => {
634            // docling renders a table's caption as a text line before the grid.
635            // `caption` is already escaped (backend convention), like a paragraph.
636            if let Some(cap) = &table.caption {
637                if !cap.is_empty() {
638                    blocks.push(md_line_breaks(&strict_text(cap, ctx.strict)));
639                }
640            }
641            let rendered = render_table(table, ctx.compact_tables);
642            if !rendered.is_empty() {
643                blocks.push(rendered);
644            }
645        }
646        // Classification predictions don't affect docling's Markdown output.
647        Node::Picture { caption, image, .. } => {
648            if let Some(cap) = caption {
649                if !cap.is_empty() {
650                    blocks.push(md_line_breaks(cap));
651                }
652            }
653            blocks.push(picture_marker(image.as_ref(), ctx));
654        }
655        // A chart renders as docling's picture-with-meta markdown: the caption,
656        // the placeholder, the humanized classification ("line_chart" ->
657        // "Line chart"), then the chart's data grid as a regular table.
658        Node::Chart {
659            kind,
660            table,
661            caption,
662            ..
663        } => {
664            if let Some(cap) = caption {
665                if !cap.is_empty() {
666                    blocks.push(md_line_breaks(cap));
667                }
668            }
669            blocks.push(picture_marker(None, ctx));
670            blocks.push(humanize_label(kind));
671            let rendered = render_table(table, false);
672            if !rendered.is_empty() {
673                blocks.push(rendered);
674            }
675        }
676        // A DocLang-only node is omitted from Markdown.
677        Node::DoclangOnly(_) => {}
678        // A group on a non-body layer (a hidden spreadsheet sheet) renders
679        // nothing, like every other non-body item.
680        Node::Group { layer: Some(_), .. } => {}
681        Node::Group { children, .. } => render(children, blocks, ctx),
682        Node::FieldRegion { items } => {
683            // The region container and each field item carry no text of their
684            // own; docling-core 2.93 (#724) serializes them to nothing (older
685            // releases emitted a `<!-- missing-text -->` marker for each), so
686            // only an item's marker/key/value appear, as separate paragraphs.
687            for item in items {
688                for part in [&item.marker, &item.key, &item.value].into_iter().flatten() {
689                    blocks.push(md_line_breaks(&strict_text(part, ctx.strict)));
690                }
691            }
692        }
693        // docling's Markdown serializer has no component for a `KeyValueItem`
694        // and writes its fallback placeholder in the item's place.
695        Node::KeyValueGraph { .. } => blocks.push(MISSING_KEY_VALUE_ITEM.to_string()),
696        // A rich inline group renders exactly like a paragraph of its Markdown
697        // text — the structured runs are DocLang-only.
698        Node::InlineGroup { md_text, .. } => {
699            blocks.push(md_line_breaks(&strict_text(md_text, ctx.strict)))
700        }
701        // A plain-text backend dump renders verbatim as a single block.
702        Node::TextDump(text) => {
703            if !text.is_empty() {
704                blocks.push(text.clone());
705            }
706        }
707        // Furniture (page headers/footers, HTML `<title>`) is excluded from
708        // Markdown by default, mirroring docling.
709        Node::Furniture { .. } => {}
710        Node::PageFurniture { .. } => {}
711        // A picture's contained text is JSON-only: docling's Markdown picture
712        // serializer prints the caption and the image, never the children.
713        Node::PictureChildren(_) => {}
714        // A comment lives in the notes layer — omitted like other furniture;
715        // the annotation on a body item is JSON-only, so render the item.
716        Node::CommentSection { .. } => {}
717        Node::Commented { inner, .. } => render_one(inner, blocks, ctx),
718        // Layout provenance is DocLang-only; render the wrapped node.
719        Node::Located { inner, .. } | Node::Prov { inner, .. } => render_one(inner, blocks, ctx),
720        // Page breaks are DocLang-only; docling omits them from Markdown.
721        Node::PageBreak => {}
722        // Page markers feed the JSON export only.
723        Node::PageInfo { .. } => {}
724        // Runs of adjacent list items are merged by `render`; a stray single
725        // item (a hand-built document, or a `Located` wrapper around one)
726        // still renders as its own one-item list instead of panicking —
727        // `nodes` is public API, so every representable tree must serialize.
728        Node::ListItem { .. } => render_list_run(std::slice::from_ref(node), blocks, ctx.strict),
729    }
730}
731
732/// The Markdown for a picture under the active [`ImageMode`]; Referenced mode also
733/// records the bytes in `ctx.artifacts` for the caller to write.
734/// docling-core's `_humanize_text`: underscores to spaces, first letter
735/// capitalized ("line_chart" -> "Line chart").
736fn humanize_label(label: &str) -> String {
737    let text = label.replace('_', " ");
738    let mut chars = text.chars();
739    match chars.next() {
740        Some(f) => f.to_uppercase().collect::<String>() + chars.as_str(),
741        None => text,
742    }
743}
744
745fn picture_marker(image: Option<&crate::PictureImage>, ctx: &mut Ctx) -> String {
746    match (ctx.images, image) {
747        (ImageMode::Embedded, Some(img)) => format!("![Image]({})", img.data_uri()),
748        (ImageMode::Referenced, Some(img)) => {
749            let path = format!(
750                "{}/image_{:06}.{}",
751                ctx.artifacts_dir,
752                ctx.pic_index,
753                ext_for(&img.mimetype)
754            );
755            ctx.pic_index += 1;
756            ctx.artifacts.push((path.clone(), img.data.clone()));
757            format!("![Image]({})", escape_uri_path(&path))
758        }
759        // Placeholder, or any mode with no extracted image.
760        _ => "<!-- image -->".to_string(),
761    }
762}
763
764/// Encode a URL or filesystem path as a Markdown link destination —
765/// docling-core's `MarkdownPictureSerializer._escape_uri_path`
766/// (docling-core#698, 2.94). Handles URLs of any scheme as well as POSIX and
767/// Windows paths, keeps relative paths relative and never double-encodes:
768/// backslashes become `/` (a backslash is both the Windows separator and a
769/// Markdown escape), a UNC share `//host/…` and an absolute Windows path
770/// `C:/…` become RFC 8089 `file://` URLs (the one spelling a renderer cannot
771/// misread as a scheme-relative URL or a `C:` scheme), a URL keeps its
772/// scheme / authority / delimiters with only the components encoded, and
773/// everything else is percent-encoded as a path. `%` is kept so an
774/// already-encoded destination stays as it is; spaces and parentheses are
775/// encoded because they would end (or unbalance) a Markdown inline link.
776pub(crate) fn escape_uri_path(value: &str) -> String {
777    const KEEP: &str = "/%:@+,;=~$!&'*";
778    let s = value.replace('\\', "/");
779    if let Some(rest) = s.strip_prefix("//") {
780        // A fileshare: `file://<host>/<path>`, the host possibly empty.
781        let rest = rest.trim_start_matches('/');
782        let (host, tail) = rest.split_once('/').unwrap_or((rest, ""));
783        return format!("file://{host}{}", percent_quote(&format!("/{tail}"), KEEP));
784    }
785    let bytes = s.as_bytes();
786    if bytes.len() >= 3 && bytes[0].is_ascii_alphabetic() && bytes[1] == b':' && bytes[2] == b'/' {
787        // A Windows path with a drive letter: `file:///C:/…`.
788        return format!("file:///{}", percent_quote(&s, KEEP));
789    }
790    // A URL keeps its scheme, authority and delimiters; only its components are
791    // encoded. A single-character scheme cannot be real (it is a drive letter,
792    // handled above), so it is read as a path — like `urlsplit`.
793    if let Some((scheme, rest)) = s.split_once(':') {
794        let valid_scheme = scheme.len() > 1
795            && scheme.as_bytes()[0].is_ascii_alphabetic()
796            && scheme
797                .bytes()
798                .all(|b| b.is_ascii_alphanumeric() || matches!(b, b'+' | b'-' | b'.'));
799        if valid_scheme {
800            let (authority, rest) = match rest.strip_prefix("//") {
801                Some(r) => {
802                    let end = r.find(['/', '?', '#']).unwrap_or(r.len());
803                    (Some(&r[..end]), &r[end..])
804                }
805                None => (None, rest),
806            };
807            let (before_frag, fragment) = rest.split_once('#').unwrap_or((rest, ""));
808            let (path, query) = before_frag.split_once('?').unwrap_or((before_frag, ""));
809            let mut out = format!("{scheme}:");
810            if let Some(a) = authority {
811                out.push_str("//");
812                out.push_str(a);
813            }
814            out.push_str(&percent_quote(path, KEEP));
815            if !query.is_empty() {
816                out.push('?');
817                out.push_str(&percent_quote(query, KEEP));
818            }
819            if !fragment.is_empty() {
820                out.push('#');
821                out.push_str(&percent_quote(fragment, KEEP));
822            }
823            return out;
824        }
825    }
826    // A relative or root-relative local path.
827    percent_quote(&s, KEEP)
828}
829
830/// `urllib.parse.quote(s, safe)`: unreserved ASCII (`A–Z a–z 0–9 _ . - ~`) and
831/// the `safe` set stay, every other byte of the UTF-8 encoding becomes `%XX`.
832fn percent_quote(s: &str, safe: &str) -> String {
833    let mut out = String::with_capacity(s.len());
834    for &b in s.as_bytes() {
835        let keep = b.is_ascii_alphanumeric()
836            || matches!(b, b'_' | b'.' | b'-' | b'~')
837            || (b.is_ascii() && safe.contains(b as char));
838        if keep {
839            out.push(b as char);
840        } else {
841            out.push_str(&format!("%{b:02X}"));
842        }
843    }
844    out
845}
846
847fn ext_for(mimetype: &str) -> &str {
848    match mimetype {
849        "image/jpeg" => "jpg",
850        "image/gif" => "gif",
851        "image/webp" => "webp",
852        "image/bmp" => "bmp",
853        "image/tiff" => "tif",
854        _ => "png",
855    }
856}
857
858/// Render a table. `compact` selects between two serializers:
859///
860/// - **padded** (default) — docling-core's `tabulate(tablefmt="github")`: columns
861///   are padded to a fixed width (header width + a minimum padding of 2, or the
862///   widest data cell); numeric columns (every data cell parses as a number) are
863///   right-aligned, others left-aligned; separators are plain dashes of
864///   `width + 2`. Matches current published docling (DOCX/HTML conformance).
865/// - **compact** — `| a | b |` cells with single-dash `| - | - |` separators, no
866///   width padding. Matches the committed PDF groundtruth corpus, which predates
867///   the padded serializer.
868///
869/// Each cell is first escaped (`\n` → space, `|` → `&#124;`) so it can't break the
870/// table. The header row is the table's leading `column_header` block flattened
871/// to one row ([`Table::header_row_count`] + [`flatten_header_rows`],
872/// docling-core#723); alignment and widths are computed over the body rows.
873/// Whether a table cell counts as a number for column alignment, matching
874/// `tabulate`'s detection: an ordinary float/int (`f64`-parseable, covering
875/// `1e2`/`inf`/`+1.5`) **or** a thousands-separated number like `7,015`.
876fn is_number_cell(t: &str) -> bool {
877    t.parse::<f64>().is_ok() || is_thousands_number(t)
878}
879
880/// A number with comma thousands-separators, per `tabulate`'s
881/// `_float_with_thousands_separators` regex
882/// (`^(([+-]?[0-9]{1,3})(?:,([0-9]{3}))*)?(?(1)\.[0-9]*|\.[0-9]+)?$`): the
883/// integer part is 1–3 digits then any number of `,ddd` groups; the fraction is
884/// optional (and, without an integer part, must have at least one digit).
885fn is_thousands_number(t: &str) -> bool {
886    let b = t.as_bytes();
887    let mut i = 0;
888    let start = i;
889    if i < b.len() && (b[i] == b'+' || b[i] == b'-') {
890        i += 1;
891    }
892    // First digit chunk: 1–3 digits.
893    let d0 = i;
894    while i < b.len() && b[i].is_ascii_digit() && i - d0 < 3 {
895        i += 1;
896    }
897    let has_int = i > d0;
898    if has_int {
899        // Subsequent `,ddd` groups (exactly three digits each).
900        while i + 3 < b.len() + 1
901            && b.get(i) == Some(&b',')
902            && b.get(i + 1).is_some_and(u8::is_ascii_digit)
903            && b.get(i + 2).is_some_and(u8::is_ascii_digit)
904            && b.get(i + 3).is_some_and(u8::is_ascii_digit)
905        {
906            i += 4;
907        }
908    } else {
909        // A sign only counts with an integer part.
910        i = start;
911    }
912    // Optional fraction.
913    if i < b.len() && b[i] == b'.' {
914        i += 1;
915        let f0 = i;
916        while i < b.len() && b[i].is_ascii_digit() {
917            i += 1;
918        }
919        if !has_int && i == f0 {
920            return false; // `.` with no digits and no integer part
921        }
922    } else if !has_int {
923        return false; // neither integer nor fractional part
924    }
925    i == b.len()
926}
927
928/// The single GFM header row for a table: the leading header rows (see
929/// [`Table::header_row_count`]) flattened per column, texts joined with
930/// `" - "` after dropping consecutive duplicates — docling-core's
931/// `_flatten_header_rows` (docling-core#723). The duplicate rule is what
932/// keeps a row-spanning header from being joined to itself (the grid repeats
933/// its text into every row it covers); it is position-based, so two stacked
934/// levels sharing a label collapse too — GFM has one header row, and upstream
935/// accepts that loss. No header rows → one empty header cell per column.
936fn flatten_header_rows(header_rows: &[Vec<String>], num_cols: usize) -> Vec<String> {
937    (0..num_cols)
938        .map(|c| {
939            let mut parts: Vec<&str> = Vec::new();
940            for row in header_rows {
941                let text = row.get(c).map(String::as_str).unwrap_or("");
942                if !text.is_empty() && parts.last() != Some(&text) {
943                    parts.push(text);
944                }
945            }
946            parts.join(" - ")
947        })
948        .collect()
949}
950
951pub(crate) fn render_table(table: &Table, compact: bool) -> String {
952    if table.rows.is_empty() {
953        return String::new();
954    }
955    let num_cols = table.rows.iter().map(Vec::len).max().unwrap_or(0);
956    if num_cols == 0 {
957        return String::new();
958    }
959
960    // Escaped, rectangular grid (ragged rows padded with empty cells). The
961    // header block is resolved to the one row GFM allows (docling-core#723);
962    // `tabulate` strips data cells of surrounding whitespace but leaves the
963    // header texts as-is.
964    let num_headers = table.header_row_count().min(table.rows.len());
965    let escaped = |r: usize| -> Vec<String> {
966        (0..num_cols)
967            .map(|c| escape_cell(table.rows[r].get(c).map(String::as_str).unwrap_or("")))
968            .collect()
969    };
970    let header_rows: Vec<Vec<String>> = (0..num_headers).map(escaped).collect();
971    let header = flatten_header_rows(&header_rows, num_cols);
972    let body: Vec<Vec<String>> = (num_headers..table.rows.len())
973        .map(|r| {
974            escaped(r)
975                .into_iter()
976                .map(|c| c.trim().to_string())
977                .collect()
978        })
979        .collect();
980
981    if compact {
982        // Compact: cells joined by " | ", no padding, single-dash separators.
983        let render_row = |row: &[String]| -> String { format!("| {} |", row.join(" | ")) };
984        let mut lines = Vec::with_capacity(body.len() + 2);
985        lines.push(render_row(&header));
986        let sep: Vec<&str> = (0..num_cols).map(|_| "-").collect();
987        lines.push(format!("| {} |", sep.join(" | ")));
988        for row in &body {
989            lines.push(render_row(row));
990        }
991        return lines.join("\n");
992    }
993
994    // Display width (Unicode scalar count — good enough for now).
995    let dw = |s: &str| s.chars().count();
996
997    // A column is right-aligned when at least one body cell is numeric and every
998    // non-empty body cell is numeric — matching `tabulate`'s column typing, where
999    // empty cells are "missing" (ignored) and a number may carry thousands
1000    // separators (`7,015`), which a plain `f64` parse rejects.
1001    let right: Vec<bool> = (0..num_cols)
1002        .map(|c| {
1003            let mut any = false;
1004            for row in &body {
1005                let t = row[c].trim();
1006                if t.is_empty() {
1007                    continue;
1008                }
1009                if !is_number_cell(t) {
1010                    return false;
1011                }
1012                any = true;
1013            }
1014            any
1015        })
1016        .collect();
1017
1018    // Column width = max(header_width + MIN_PADDING(2), max body-cell width).
1019    let width: Vec<usize> = (0..num_cols)
1020        .map(|c| {
1021            let mut w = dw(&header[c]) + 2;
1022            for row in &body {
1023                w = w.max(dw(&row[c]));
1024            }
1025            w
1026        })
1027        .collect();
1028
1029    let fmt_cell = |s: &str, c: usize| -> String {
1030        let pad = " ".repeat(width[c].saturating_sub(dw(s)));
1031        let body = if right[c] {
1032            format!("{pad}{s}")
1033        } else {
1034            format!("{s}{pad}")
1035        };
1036        format!(" {body} ")
1037    };
1038    let render_row = |row: &[String]| -> String {
1039        let cells: Vec<String> = (0..num_cols).map(|c| fmt_cell(&row[c], c)).collect();
1040        format!("|{}|", cells.join("|"))
1041    };
1042
1043    let mut lines = Vec::with_capacity(body.len() + 2);
1044    lines.push(render_row(&header));
1045    let sep: Vec<String> = (0..num_cols).map(|c| "-".repeat(width[c] + 2)).collect();
1046    lines.push(format!("|{}|", sep.join("|")));
1047    for row in &body {
1048        lines.push(render_row(row));
1049    }
1050    lines.join("\n")
1051}
1052
1053/// Escape a table cell so it can't break the markdown table: newlines become
1054/// spaces and pipes become the `&#124;` HTML entity (matches docling-core).
1055fn escape_cell(s: &str) -> String {
1056    s.replace('\n', " ").replace('|', "&#124;")
1057}
1058
1059#[cfg(test)]
1060mod tests {
1061    use super::*;
1062    use crate::{PictureImage, TableCell, TableStructure};
1063
1064    /// #385: where one list ends and the next begins is the backend's call
1065    /// (`first_in_list`), never the serializer's. An ordered run `1.` → `5.`
1066    /// is one list (an AsciiDoc numbered list around a nested one), and so are
1067    /// mixed bullet/ordered items the backend did not separate; only a flagged
1068    /// item opens a new list and earns the blank line.
1069    #[test]
1070    fn list_boundaries_come_from_the_backend_not_the_numbering() {
1071        let item = |ordered: bool, number: u64, first_in_list: bool, text: &str| Node::ListItem {
1072            ordered,
1073            number,
1074            first_in_list,
1075            text: text.into(),
1076            level: 0,
1077            marker: None,
1078            location: None,
1079            dclx: None,
1080            href: None,
1081            layer: None,
1082        };
1083        let md = |items: Vec<Node>| {
1084            let mut doc = DoclingDocument::new("t");
1085            for n in items {
1086                doc.push(n);
1087            }
1088            doc.export_to_markdown()
1089        };
1090        // A number gap alone is not a boundary.
1091        assert_eq!(
1092            md(vec![
1093                item(true, 1, true, "one"),
1094                item(true, 5, false, "five")
1095            ]),
1096            "1. one\n5. five\n"
1097        );
1098        // Nor is a kind flip the backend did not flag …
1099        assert_eq!(
1100            md(vec![
1101                item(false, 0, true, "bullet"),
1102                item(true, 1, false, "one"),
1103                item(false, 0, false, "bullet two"),
1104            ]),
1105            "- bullet\n1. one\n- bullet two\n"
1106        );
1107        // … while a flagged item is one, whatever its number says.
1108        assert_eq!(
1109            md(vec![
1110                item(true, 1, true, "a"),
1111                item(true, 2, false, "b"),
1112                item(true, 3, true, "new list, continuing count"),
1113            ]),
1114            "1. a\n2. b\n\n3. new list, continuing count\n"
1115        );
1116    }
1117
1118    #[test]
1119    fn renders_headings_paragraphs_and_lists() {
1120        let mut doc = DoclingDocument::new("demo");
1121        doc.add_heading(1, "Title");
1122        doc.add_paragraph("Hello world.");
1123        doc.push(Node::ListItem {
1124            ordered: false,
1125            number: 1,
1126            first_in_list: true,
1127            text: "first".into(),
1128            level: 0,
1129            marker: None,
1130            location: None,
1131            dclx: None,
1132            href: None,
1133            layer: None,
1134        });
1135        doc.push(Node::ListItem {
1136            ordered: false,
1137            number: 2,
1138            first_in_list: false,
1139            text: "second".into(),
1140            level: 0,
1141            marker: None,
1142            location: None,
1143            dclx: None,
1144            href: None,
1145            layer: None,
1146        });
1147        let md = doc.export_to_markdown();
1148        assert_eq!(md, "# Title\n\nHello world.\n\n- first\n- second\n");
1149    }
1150
1151    /// docling-core 2.92 (#721): a single newline inside an item's text is a
1152    /// GFM hard line break, a blank line stays a paragraph break, and a heading
1153    /// collapses its newline to a space. Nested-table dumps stay verbatim.
1154    #[test]
1155    fn single_newlines_become_gfm_hard_line_breaks() {
1156        let mut doc = DoclingDocument::new("t");
1157        doc.push(Node::Heading {
1158            level: 1,
1159            text: "Hello\nWorld".into(),
1160        });
1161        doc.push(Node::Paragraph {
1162            text: "line one\nline two\n\npara two".into(),
1163        });
1164        doc.push(Node::ListItem {
1165            ordered: false,
1166            number: 1,
1167            first_in_list: true,
1168            text: "item\ncontinued".into(),
1169            level: 0,
1170            marker: None,
1171            location: None,
1172            dclx: None,
1173            href: None,
1174            layer: None,
1175        });
1176        doc.push(Node::TextDump("A1 B1 \n\n\nC1".into()));
1177        assert_eq!(
1178            doc.export_to_markdown(),
1179            "# Hello World\n\nline one  \nline two\n\npara two\n\n- item  \ncontinued\n\nA1 B1 \n\n\nC1\n"
1180        );
1181    }
1182
1183    /// docling-core#540: inside a rich table cell a heading is plain text;
1184    /// docling-core#724: a field region renders only its items' key/value text.
1185    #[test]
1186    fn table_cell_mode_and_field_regions() {
1187        let mut doc = DoclingDocument::new("t");
1188        doc.push(Node::Heading {
1189            level: 2,
1190            text: "A  text".into(),
1191        });
1192        doc.push(Node::Paragraph {
1193            text: "body".into(),
1194        });
1195        assert_eq!(to_markdown_table_cell(&doc, false), "A  text\n\nbody");
1196        assert_eq!(doc.export_to_markdown(), "## A  text\n\nbody\n");
1197
1198        let mut doc = DoclingDocument::new("f");
1199        doc.push(Node::FieldRegion {
1200            items: vec![crate::FieldItem {
1201                marker: None,
1202                key: Some("Name:".into()),
1203                value: Some("John Doe".into()),
1204                value_kind: None,
1205            }],
1206        });
1207        assert_eq!(doc.export_to_markdown(), "Name:\n\nJohn Doe\n");
1208    }
1209
1210    #[test]
1211    fn strict_renders_recovered_links_legacy_does_not() {
1212        let mut doc = DoclingDocument::new("cv");
1213        doc.add_paragraph("Find me on LinkedIn or GitHub.");
1214        doc.links = vec![
1215            ("LinkedIn".into(), "https://www.linkedin.com/in/x/".into()),
1216            ("GitHub".into(), "https://github.com/x/".into()),
1217        ];
1218        // Legacy/docling mode: links are left untouched (conformance preserved).
1219        assert_eq!(doc.export_to_markdown(), "Find me on LinkedIn or GitHub.\n");
1220        // Strict mode: anchors become Markdown links.
1221        assert_eq!(
1222            doc.export_to_markdown_with(true),
1223            "Find me on [LinkedIn](https://www.linkedin.com/in/x/) or [GitHub](https://github.com/x/).\n"
1224        );
1225    }
1226
1227    #[test]
1228    fn strict_links_match_escaped_anchor_and_consume_in_order() {
1229        let mut doc = DoclingDocument::new("d");
1230        // The PDF assembler HTML-escapes prose, so by serialization time the body
1231        // already carries `&amp;`; the anchor is stored un-escaped. The matcher must
1232        // escape the anchor to find it. Two identical anchors link in document order.
1233        doc.add_paragraph("AI &amp; ML here, and issues here, then issues there.");
1234        doc.links = vec![
1235            ("AI & ML".into(), "https://a/".into()),
1236            ("issues".into(), "https://first/".into()),
1237            ("issues".into(), "https://second/".into()),
1238        ];
1239        assert_eq!(
1240            doc.export_to_markdown_with(true),
1241            "[AI &amp; ML](https://a/) here, and [issues](https://first/) here, then [issues](https://second/) there.\n"
1242        );
1243    }
1244
1245    /// docling-core#698: the referenced-image destination is percent-encoded —
1246    /// upstream's own case table (paths, Windows flavours, UNC, URLs) plus
1247    /// idempotency on the encoded result.
1248    #[test]
1249    fn referenced_image_destinations_are_escaped() {
1250        let cases = [
1251            (
1252                "doc_artifacts/image_000001_ab12.png",
1253                "doc_artifacts/image_000001_ab12.png",
1254            ),
1255            (
1256                "My Report_artifacts/img.png",
1257                "My%20Report_artifacts/img.png",
1258            ),
1259            ("artifacts/img (1).png", "artifacts/img%20%281%29.png"),
1260            ("100%_scale/a#b?c.png", "100%_scale/a%23b%3Fc.png"),
1261            ("/home/a b/img.png", "/home/a%20b/img.png"),
1262            (
1263                "My Report_artifacts\\img.png",
1264                "My%20Report_artifacts/img.png",
1265            ),
1266            (
1267                "C:/Users/me/My Docs/img.png",
1268                "file:///C:/Users/me/My%20Docs/img.png",
1269            ),
1270            ("C:\\Users\\me\\img.png", "file:///C:/Users/me/img.png"),
1271            (
1272                "//server/share/My Docs/img.png",
1273                "file://server/share/My%20Docs/img.png",
1274            ),
1275            ("\\\\server\\share\\img.png", "file://server/share/img.png"),
1276            ("file:///home/a b/img.png", "file:///home/a%20b/img.png"),
1277            (
1278                "s3://bucket/My Report_artifacts/img.png",
1279                "s3://bucket/My%20Report_artifacts/img.png",
1280            ),
1281            (
1282                "https://example.com:8080/a b.png?w=1&h=2#frag",
1283                "https://example.com:8080/a%20b.png?w=1&h=2#frag",
1284            ),
1285            (
1286                "https://example.com/img (1).png",
1287                "https://example.com/img%20%281%29.png",
1288            ),
1289            ("caf\u{e9}/im\u{e4}ge.png", "caf%C3%A9/im%C3%A4ge.png"),
1290        ];
1291        for (input, expected) in cases {
1292            assert_eq!(escape_uri_path(input), expected, "input {input:?}");
1293            assert_eq!(
1294                escape_uri_path(expected),
1295                expected,
1296                "idempotent {expected:?}"
1297            );
1298        }
1299        // The whole marker, through the referenced-image export.
1300        let mut doc = DoclingDocument::new("t");
1301        doc.push(Node::Picture {
1302            caption: None,
1303            caption_href: None,
1304            image: Some(PictureImage {
1305                mimetype: "image/png".into(),
1306                width: 1,
1307                height: 1,
1308                data: b"x".to_vec(),
1309            }),
1310            classification: None,
1311            caption_parent: Default::default(),
1312        });
1313        let (md, files) = doc
1314            .export_to_markdown_with_images(ImageMode::Referenced, "My Report (final)_artifacts");
1315        assert!(
1316            md.contains("![Image](My%20Report%20%28final%29_artifacts/image_000000.png)"),
1317            "got:\n{md}"
1318        );
1319        // The file path handed back for writing stays unescaped.
1320        assert_eq!(files[0].0, "My Report (final)_artifacts/image_000000.png");
1321    }
1322
1323    /// Pictures the HTML backend folds into a list item print after the item
1324    /// line with plain newlines; a `<br>` newline in the item's own text is
1325    /// still a GFM hard line break.
1326    #[test]
1327    fn folded_list_item_pictures_keep_plain_newlines() {
1328        assert_eq!(
1329            list_item_text("Step\n<!-- image -->", false),
1330            "Step\n<!-- image -->"
1331        );
1332        assert_eq!(
1333            list_item_text("Step\nAlt text\n<!-- image -->\n<!-- image -->", false),
1334            "Step\nAlt text\n<!-- image -->\n<!-- image -->"
1335        );
1336        assert_eq!(
1337            list_item_text("line one\nline two", false),
1338            "line one  \nline two"
1339        );
1340    }
1341
1342    /// docling-core#723: the header block is the leading run of rows on which a
1343    /// `column_header` cell starts, flattened per column with " - ".
1344    #[test]
1345    fn stacked_header_rows_flatten_into_one() {
1346        let mut t = Table {
1347            rows: vec![
1348                vec!["".into(), "% of Total".into(), "% of Total".into()],
1349                vec!["class".into(), "Train".into(), "Test".into()],
1350                vec!["Caption".into(), "2.04".into(), "1.77".into()],
1351            ],
1352            ..Default::default()
1353        };
1354        t.structure = Some(TableStructure {
1355            header_row: vec![true, true, false],
1356            col_continuation: vec![
1357                vec![false, false, true],
1358                vec![false, false, false],
1359                vec![false, false, false],
1360            ],
1361            ..Default::default()
1362        });
1363        assert_eq!(t.header_row_count(), 2);
1364        assert_eq!(
1365            render_table(&t, true),
1366            "| class | % of Total - Train | % of Total - Test |\n| - | - | - |\n| Caption | 2.04 | 1.77 |"
1367        );
1368        // padded: widths from the flattened header, alignment from body rows
1369        assert_eq!(
1370            render_table(&t, false),
1371            "| class   |   % of Total - Train |   % of Total - Test |\n\
1372             |---------|----------------------|---------------------|\n\
1373             | Caption |                 2.04 |                1.77 |"
1374        );
1375    }
1376
1377    /// A header spanning two rows is repeated into the second row by the grid;
1378    /// that row is not a header row unless another header cell starts there.
1379    #[test]
1380    fn vertically_spanning_header_does_not_extend_the_block() {
1381        let mut t = Table {
1382            rows: vec![
1383                vec!["Name".into(), "Value".into()],
1384                vec!["Name".into(), "1".into()],
1385                vec!["x".into(), "2".into()],
1386            ],
1387            ..Default::default()
1388        };
1389        t.structure = Some(TableStructure {
1390            col_header: vec![vec![true, true], vec![true, false], vec![false, false]],
1391            row_continuation: vec![vec![false, false], vec![true, false], vec![false, false]],
1392            ..Default::default()
1393        });
1394        assert_eq!(t.header_row_count(), 1);
1395        assert_eq!(
1396            render_table(&t, true),
1397            "| Name | Value |\n| - | - |\n| Name | 1 |\n| x | 2 |"
1398        );
1399    }
1400
1401    /// Flags that begin on a later row promote nothing: every row stays in the
1402    /// body under an empty header row (tabulate's `headers=["", ""]`).
1403    #[test]
1404    fn header_flags_not_on_row_zero_keep_all_rows_in_the_body() {
1405        let mut t = Table {
1406            rows: vec![
1407                vec!["1".into(), "2".into()],
1408                vec!["a".into(), "b".into()],
1409                vec!["333".into(), "4".into()],
1410            ],
1411            ..Default::default()
1412        };
1413        t.structure = Some(TableStructure {
1414            header_row: vec![false, true, false],
1415            ..Default::default()
1416        });
1417        assert_eq!(t.header_row_count(), 0);
1418        assert_eq!(
1419            render_table(&t, false),
1420            "|     |    |\n|-----|----|\n| 1   | 2  |\n| a   | b  |\n| 333 | 4  |"
1421        );
1422    }
1423
1424    /// A pivot table's row headers (`<th rowspan>`) carry `row_header`, not
1425    /// `column_header` (docling#4216), so the data row beside them is not
1426    /// pulled into the header block — what this port used to reach with a
1427    /// deviation now falls out of the flags themselves.
1428    #[test]
1429    fn pivot_row_headers_do_not_extend_the_header() {
1430        let mut t = Table {
1431            rows: vec![
1432                vec!["Year".into(), "Month".into()],
1433                vec!["2025".into(), "January".into()],
1434                vec!["2025".into(), "February".into()],
1435            ],
1436            ..Default::default()
1437        };
1438        t.structure = Some(TableStructure {
1439            col_header: vec![vec![true, true], vec![false, false], vec![false, false]],
1440            row_header: vec![vec![false, false], vec![true, false], vec![true, false]],
1441            row_continuation: vec![vec![false, false], vec![false, false], vec![true, false]],
1442            ..Default::default()
1443        });
1444        assert_eq!(t.header_row_count(), 1);
1445        assert_eq!(
1446            render_table(&t, true),
1447            "| Year | Month |\n| - | - |\n| 2025 | January |\n| 2025 | February |"
1448        );
1449    }
1450
1451    /// No `column_header` anywhere (first-class cells without flags) → row 0
1452    /// stays the header, as before.
1453    #[test]
1454    fn unflagged_cells_keep_row_zero_as_header() {
1455        let mut t = Table {
1456            rows: vec![vec!["h".into()], vec!["d".into()]],
1457            ..Default::default()
1458        };
1459        t.cells = Some(
1460            [(0usize, "h"), (1, "d")]
1461                .into_iter()
1462                .map(|(r, text)| TableCell {
1463                    text: text.into(),
1464                    bbox: None,
1465                    start_row: r,
1466                    start_col: 0,
1467                    row_span: 1,
1468                    col_span: 1,
1469                    column_header: false,
1470                    row_header: false,
1471                    row_section: false,
1472                })
1473                .collect(),
1474        );
1475        assert_eq!(t.header_row_count(), 1);
1476        assert_eq!(render_table(&t, true), "| h |\n| - |\n| d |");
1477    }
1478
1479    #[test]
1480    fn renders_compact_table() {
1481        let mut doc = DoclingDocument::new("t");
1482        // The compact form is opt-in (the PDF backend sets it); default output uses
1483        // the padded GitHub serializer (covered by the regression fixtures).
1484        doc.compact_tables = true;
1485        doc.push(Node::Table(Table {
1486            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
1487            location: None,
1488            structure: None,
1489            cell_blocks: None,
1490            cells: None,
1491            caption: None,
1492            caption_parent: Default::default(),
1493        }));
1494        let md = doc.export_to_markdown();
1495        assert_eq!(md, "| a | b |\n| - | - |\n| 1 | 2 |\n");
1496    }
1497
1498    #[test]
1499    fn renders_padded_github_table_by_default() {
1500        let mut doc = DoclingDocument::new("t");
1501        doc.push(Node::Table(Table {
1502            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
1503            location: None,
1504            structure: None,
1505            cell_blocks: None,
1506            cells: None,
1507            caption: None,
1508            caption_parent: Default::default(),
1509        }));
1510        let md = doc.export_to_markdown();
1511        // Numeric data columns are right-aligned; columns padded to header+2.
1512        assert_eq!(md, "|   a |   b |\n|-----|-----|\n|   1 |   2 |\n");
1513    }
1514
1515    #[test]
1516    fn strict_unescapes_inline_underscores_legacy_keeps_them() {
1517        let mut doc = DoclingDocument::new("t");
1518        doc.add_heading(1, "a\\_b");
1519        doc.add_paragraph("x\\_y");
1520        doc.push(Node::ListItem {
1521            ordered: false,
1522            number: 1,
1523            first_in_list: true,
1524            text: "i\\_j".into(),
1525            level: 0,
1526            marker: None,
1527            location: None,
1528            dclx: None,
1529            href: None,
1530            layer: None,
1531        });
1532        // Legacy reproduces docling's `\_` escaping byte-for-byte.
1533        assert_eq!(doc.export_to_markdown(), "# a\\_b\n\nx\\_y\n\n- i\\_j\n");
1534        // Strict prefers literal underscores (Rust-only readability mode).
1535        assert_eq!(doc.export_to_markdown_with(true), "# a_b\n\nx_y\n\n- i_j\n");
1536    }
1537
1538    /// Drive a document's nodes through [`MarkdownStreamer`] in the given page
1539    /// splits and assert the concatenated chunks equal the buffered serializer.
1540    fn assert_stream_matches(
1541        doc: &DoclingDocument,
1542        strict: bool,
1543        images: ImageMode,
1544        splits: &[usize],
1545    ) {
1546        let (want, want_artifacts) = to_markdown_images(doc, strict, images, "artifacts");
1547        let mut streamer =
1548            MarkdownStreamer::with_artifacts(strict, images, doc.compact_tables, "artifacts")
1549                .with_page_break_placeholder(doc.page_break_placeholder.clone());
1550        let mut got = String::new();
1551        let mut got_artifacts = Vec::new();
1552        let mut start = 0;
1553        for &end in splits {
1554            // Links only matter in strict mode; feed them all with the first batch
1555            // that has content (document order is preserved by the queue).
1556            let links = if start == 0 {
1557                doc.links.as_slice()
1558            } else {
1559                &[]
1560            };
1561            got.push_str(&streamer.push(&doc.nodes[start..end], links));
1562            // Referenced mode: drain per push, as a real caller writing files
1563            // page by page would — numbering must continue across drains.
1564            got_artifacts.extend(streamer.take_artifacts());
1565            start = end;
1566        }
1567        got.push_str(&streamer.push(
1568            &doc.nodes[start..],
1569            if start == 0 {
1570                doc.links.as_slice()
1571            } else {
1572                &[]
1573            },
1574        ));
1575        got_artifacts.extend(streamer.take_artifacts());
1576        got.push_str(&streamer.finish());
1577        assert_eq!(
1578            got, want,
1579            "streamed output diverged (splits={splits:?}, strict={strict})"
1580        );
1581        assert_eq!(
1582            got_artifacts, want_artifacts,
1583            "streamed artifacts diverged (splits={splits:?}, strict={strict})"
1584        );
1585    }
1586
1587    #[test]
1588    fn streaming_is_byte_identical_to_buffered() {
1589        let mut doc = DoclingDocument::new("d");
1590        doc.add_heading(1, "Title");
1591        doc.add_paragraph("First paragraph.");
1592        doc.push(Node::ListItem {
1593            ordered: false,
1594            number: 1,
1595            first_in_list: true,
1596            text: "a".into(),
1597            level: 0,
1598            marker: None,
1599            location: None,
1600            dclx: None,
1601            href: None,
1602            layer: None,
1603        });
1604        doc.push(Node::ListItem {
1605            ordered: false,
1606            number: 2,
1607            first_in_list: false,
1608            text: "b".into(),
1609            level: 0,
1610            marker: None,
1611            location: None,
1612            dclx: None,
1613            href: None,
1614            layer: None,
1615        });
1616        doc.push(Node::Code {
1617            language: Some("rust".into()),
1618            text: "let x = 1;".into(),
1619            orig: None,
1620            pretty: None,
1621        });
1622        doc.push(Node::Table(Table {
1623            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
1624            location: None,
1625            structure: None,
1626            cell_blocks: None,
1627            cells: None,
1628            caption: None,
1629            caption_parent: Default::default(),
1630        }));
1631        doc.push(Node::Picture {
1632            caption: Some("Fig 1".into()),
1633            caption_href: None,
1634            image: Some(PictureImage {
1635                mimetype: "image/png".into(),
1636                width: 2,
1637                height: 2,
1638                data: b"png-one".to_vec(),
1639            }),
1640            classification: None,
1641            caption_parent: Default::default(),
1642        });
1643        doc.add_paragraph("Last paragraph.");
1644        // A second embedded picture, so referenced mode must keep numbering
1645        // (`image_000001`) across chunk boundaries.
1646        doc.push(Node::Picture {
1647            caption: None,
1648            caption_href: None,
1649            image: Some(PictureImage {
1650                mimetype: "image/png".into(),
1651                width: 2,
1652                height: 2,
1653                data: b"png-two".to_vec(),
1654            }),
1655            classification: None,
1656            caption_parent: Default::default(),
1657        });
1658
1659        // A run of list items must never straddle a split, so try splits that fall
1660        // on safe block boundaries (the streaming PDF assembler guarantees this).
1661        for &strict in &[false, true] {
1662            for &images in &[
1663                ImageMode::Placeholder,
1664                ImageMode::Embedded,
1665                ImageMode::Referenced,
1666            ] {
1667                for splits in [&[][..], &[1][..], &[2][..], &[4][..], &[1, 4, 6, 7][..]] {
1668                    assert_stream_matches(&doc, strict, images, splits);
1669                }
1670            }
1671        }
1672    }
1673
1674    #[test]
1675    fn streaming_applies_recovered_links_in_strict_mode() {
1676        let mut doc = DoclingDocument::new("d");
1677        doc.add_paragraph("See LinkedIn for details.");
1678        doc.add_paragraph("And GitHub too.");
1679        doc.links = vec![
1680            ("LinkedIn".into(), "https://lnkd/".into()),
1681            ("GitHub".into(), "https://gh/".into()),
1682        ];
1683        // The second anchor lives in the second block, so it must be carried across
1684        // the page boundary and placed when that block streams out.
1685        assert_stream_matches(&doc, true, ImageMode::Placeholder, &[1]);
1686    }
1687
1688    /// A three-page document with one empty page in the middle and page
1689    /// markers of both kinds, as the backends emit them.
1690    fn paged_doc() -> DoclingDocument {
1691        let mut doc = DoclingDocument::new("p");
1692        doc.push(Node::PageInfo {
1693            page_no: 1,
1694            width: 100.0,
1695            height: 100.0,
1696        });
1697        doc.add_heading(1, "Title");
1698        doc.add_paragraph("Page one.");
1699        // Page two: a marker plus furniture only — renders nothing.
1700        doc.push(Node::PageBreak);
1701        doc.push(Node::PageInfo {
1702            page_no: 2,
1703            width: 100.0,
1704            height: 100.0,
1705        });
1706        doc.push(Node::PageFurniture {
1707            footer: true,
1708            location: [0, 500, 511, 511],
1709            text: "2".into(),
1710        });
1711        doc.push(Node::PageBreak);
1712        doc.push(Node::PageInfo {
1713            page_no: 3,
1714            width: 100.0,
1715            height: 100.0,
1716        });
1717        doc.add_paragraph("Page three.");
1718        // A trailing boundary with nothing after it.
1719        doc.push(Node::PageBreak);
1720        doc
1721    }
1722
1723    #[test]
1724    fn page_break_placeholder_lands_between_pages_only() {
1725        let mut doc = paged_doc();
1726        // Off by default: docling's Markdown carries no page breaks.
1727        assert_eq!(
1728            doc.export_to_markdown(),
1729            "# Title\n\nPage one.\n\nPage three.\n"
1730        );
1731        doc.page_break_placeholder = Some("<!-- page break -->".into());
1732        // One break for the 1→3 transition (the empty page 2 and the doubled
1733        // PageBreak+PageInfo markers collapse), none before the first block,
1734        // none for the trailing boundary.
1735        assert_eq!(
1736            doc.export_to_markdown(),
1737            "# Title\n\nPage one.\n\n<!-- page break -->\n\nPage three.\n"
1738        );
1739        // An empty placeholder is still a (blank) part, as upstream's
1740        // `str.replace(marker, "")` leaves the delimiters around it.
1741        doc.page_break_placeholder = Some(String::new());
1742        assert_eq!(
1743            doc.export_to_markdown(),
1744            "# Title\n\nPage one.\n\n\n\nPage three.\n"
1745        );
1746    }
1747
1748    #[test]
1749    fn page_break_placeholder_never_leads_a_single_page() {
1750        let mut doc = DoclingDocument::new("one");
1751        doc.page_break_placeholder = Some("---".into());
1752        doc.push(Node::PageBreak);
1753        doc.push(Node::PageInfo {
1754            page_no: 1,
1755            width: 10.0,
1756            height: 10.0,
1757        });
1758        doc.add_paragraph("Only page.");
1759        assert_eq!(doc.export_to_markdown(), "Only page.\n");
1760        // Two boundaries with no content between them: still one break.
1761        doc.push(Node::PageBreak);
1762        doc.push(Node::PageBreak);
1763        doc.add_paragraph("Next.");
1764        assert_eq!(doc.export_to_markdown(), "Only page.\n\n---\n\nNext.\n");
1765    }
1766
1767    #[test]
1768    fn page_break_placeholder_streams_byte_identical() {
1769        let mut doc = paged_doc();
1770        doc.page_break_placeholder = Some("<!-- page break -->".into());
1771        // Split at every page marker (how the PDF pipeline pushes page batches)
1772        // and at odd places inside a page: the pending break must survive a
1773        // push that renders nothing (page two) and land on page three's block.
1774        for splits in [
1775            &[3usize][..],
1776            &[3, 6],
1777            &[3, 6, 8],
1778            &[1, 2, 3, 4, 5, 6, 7, 8, 9],
1779            &[8],
1780        ] {
1781            assert_stream_matches(&doc, false, ImageMode::Placeholder, splits);
1782            assert_stream_matches(&doc, true, ImageMode::Placeholder, splits);
1783        }
1784    }
1785
1786    #[test]
1787    fn strict_tightens_punctuation_spacing_legacy_keeps_it() {
1788        let mut doc = DoclingDocument::new("t");
1789        doc.add_paragraph("see [ 37 , 36 ] and ( x ) .");
1790        // Legacy keeps docling's spacing byte-for-byte.
1791        assert_eq!(doc.export_to_markdown(), "see [ 37 , 36 ] and ( x ) .\n");
1792        // Strict tightens punctuation for readable Markdown.
1793        assert_eq!(doc.export_to_markdown_with(true), "see [37, 36] and (x).\n");
1794    }
1795}