Skip to main content

docling_core/
markdown.rs

1//! Markdown serializer for [`DoclingDocument`].
2
3use crate::document::{DoclingDocument, Node, Table};
4
5/// What docling's Markdown serializer writes for an item it has no component
6/// for — a `KeyValueItem` (the XBRL fact graph) is the one such item a backend
7/// produces.
8const MISSING_KEY_VALUE_ITEM: &str = "<!-- missing-key-value-item -->";
9
10/// How pictures are rendered (mirrors docling-core's `ImageRefMode`).
11#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
12pub enum ImageMode {
13    /// `<!-- image -->` (docling's default, and the only mode without image data).
14    #[default]
15    Placeholder,
16    /// `![Image](data:<mime>;base64,…)` — self-contained.
17    Embedded,
18    /// `![Image](<artifacts>/image_NNNNNN.<ext>)`; the bytes are returned for the
19    /// caller to write.
20    Referenced,
21}
22
23/// Serializer state threaded through the render walk.
24struct Ctx {
25    strict: bool,
26    /// Emit compact `| a | b |` tables instead of the padded GitHub serializer.
27    compact_tables: bool,
28    images: ImageMode,
29    artifacts_dir: String,
30    /// (relative path, bytes) for each referenced image — written by the caller.
31    artifacts: Vec<(String, Vec<u8>)>,
32    pic_index: usize,
33    /// Rendering the block content of a rich table cell (docling-core 2.94's
34    /// `in_table_cell`, docling-core#540): a heading has no valid Markdown
35    /// form inside a table, so it renders as plain text without `#` markers.
36    in_table_cell: bool,
37    /// docling-core's `MarkdownParams.page_break_placeholder`: the text that
38    /// separates two pages ([`DoclingDocument::page_break_placeholder`]).
39    /// `None` omits page breaks, docling's default.
40    page_break: Option<String>,
41    /// A page boundary has been crossed since the last rendered block, so the
42    /// next block is preceded by the placeholder. docling yields its
43    /// `_PageBreakNode` between two *items* whose `prov.page_no` differ, so a
44    /// boundary before the first block or after the last one emits nothing,
45    /// and a run of empty pages collapses into a single break.
46    pending_page_break: bool,
47    /// Whether any block has been rendered yet — for a streamer, across every
48    /// earlier push too — which is what makes a boundary a *pending* break.
49    emitted_any: bool,
50}
51
52/// Render a document to a Markdown string (pictures as placeholders).
53///
54/// `strict` selects the serializer-level behaviours that differ between
55/// docling-legacy output and cleaner Markdown — currently the code-fence
56/// language (legacy drops it, strict keeps it).
57pub fn to_markdown(doc: &DoclingDocument, strict: bool) -> String {
58    to_markdown_images(doc, strict, ImageMode::Placeholder, "artifacts").0
59}
60
61/// Render to Markdown with an explicit picture [`ImageMode`]. Returns the
62/// Markdown and, for [`ImageMode::Referenced`], the `(path, bytes)` of each image
63/// the caller should write (relative to the Markdown file).
64pub fn to_markdown_images(
65    doc: &DoclingDocument,
66    strict: bool,
67    images: ImageMode,
68    artifacts_dir: &str,
69) -> (String, Vec<(String, Vec<u8>)>) {
70    let mut ctx = Ctx {
71        strict,
72        compact_tables: doc.compact_tables,
73        images,
74        artifacts_dir: artifacts_dir.to_string(),
75        artifacts: Vec::new(),
76        pic_index: 0,
77        in_table_cell: false,
78        page_break: doc.page_break_placeholder.clone(),
79        pending_page_break: false,
80        emitted_any: false,
81    };
82    let mut blocks: Vec<String> = Vec::new();
83    render(&doc.nodes, &mut blocks, &mut ctx);
84    let mut body = blocks.join("\n\n");
85    // Strict mode only: turn recovered source hyperlinks into Markdown links.
86    // docling's standard pipeline drops them, so doing this in legacy mode would
87    // diverge from docling — hence strict-only, leaving conformance output intact.
88    if strict && !doc.links.is_empty() {
89        body = apply_links(&body, &doc.links);
90    }
91    let md = if body.is_empty() {
92        String::new()
93    } else {
94        format!("{body}\n")
95    };
96    (md, ctx.artifacts)
97}
98
99/// Render the block content of a *rich table cell* to Markdown — what
100/// docling-core's table serializer does for a `RichTableCell`
101/// (`doc_serializer.serialize(item, in_table_cell=True)`): the cell's
102/// paragraphs, lists and flattened nested tables render as in a document, but a
103/// heading loses its `#` markers (docling-core#540 — the Markdown spec has no
104/// headings inside tables). Pictures stay placeholders. The caller flattens the
105/// result into its cell text; the table serializer later turns the newlines
106/// into spaces.
107pub fn to_markdown_table_cell(doc: &DoclingDocument, strict: bool) -> String {
108    let mut ctx = Ctx {
109        strict,
110        compact_tables: doc.compact_tables,
111        images: ImageMode::Placeholder,
112        artifacts_dir: String::new(),
113        artifacts: Vec::new(),
114        pic_index: 0,
115        in_table_cell: true,
116        // A rich cell is one page's content; its sub-document carries no
117        // page boundaries and docling's `_iterate_items` runs the page-break
118        // scan over the document root only.
119        page_break: None,
120        pending_page_break: false,
121        emitted_any: false,
122    };
123    let mut blocks: Vec<String> = Vec::new();
124    render(&doc.nodes, &mut blocks, &mut ctx);
125    blocks.join("\n\n")
126}
127
128/// Wrap each recovered link's anchor text in Markdown `[anchor](href)`. Anchors
129/// arrive cleaned (curly quotes/dashes already normalized) but un-escaped, so we
130/// match against the body's HTML-escaped (`&`/`<`/`>`) form, the way prose nodes
131/// were serialized. Links are consumed in document order from a moving cursor, so
132/// a repeated anchor (e.g. two "issues") links its successive occurrences rather
133/// than all pointing at the first. An anchor that can't be located is skipped
134/// (its text may have been split across a line wrap or table cell).
135fn apply_links(body: &str, links: &[(String, String)]) -> String {
136    let mut out = body.to_string();
137    let mut cursor = 0usize;
138    for (anchor, href) in links {
139        let anchor = anchor
140            .replace('&', "&amp;")
141            .replace('<', "&lt;")
142            .replace('>', "&gt;");
143        if anchor.is_empty() {
144            continue;
145        }
146        if let Some(rel) = out[cursor..].find(&anchor) {
147            let at = cursor + rel;
148            // Don't relink inside an already-emitted `](` Markdown link target.
149            let replacement = format!("[{anchor}]({href})");
150            out.replace_range(at..at + anchor.len(), &replacement);
151            cursor = at + replacement.len();
152        }
153    }
154    out
155}
156
157/// Like [`apply_links`] but over a single chunk, consuming from a shared queue so
158/// the same `[anchor](href)` rewriting can be applied incrementally as Markdown is
159/// streamed out. Each queued link is matched (in document order) against `chunk`
160/// and rewritten in place; a link whose anchor is not in this chunk is carried
161/// forward in the queue for a later chunk. Anchors are recovered in document
162/// order and a chunk is always a contiguous run of whole blocks, so this
163/// reproduces [`apply_links`]' single moving cursor: the link lands in whichever
164/// chunk contains its anchor, identically to the buffered path. (A link whose
165/// anchor never appears is carried to the end and dropped — the same no-op
166/// `apply_links` performs for an unlocatable anchor.)
167fn apply_links_chunk(chunk: &str, queue: &mut Vec<(String, String)>) -> String {
168    let mut out = chunk.to_string();
169    let mut cursor = 0usize;
170    let mut carried: Vec<(String, String)> = Vec::new();
171    for (anchor_raw, href) in std::mem::take(queue) {
172        let anchor = anchor_raw
173            .replace('&', "&amp;")
174            .replace('<', "&lt;")
175            .replace('>', "&gt;");
176        if anchor.is_empty() {
177            continue;
178        }
179        if let Some(rel) = out[cursor..].find(&anchor) {
180            let at = cursor + rel;
181            let replacement = format!("[{anchor}]({href})");
182            out.replace_range(at..at + anchor.len(), &replacement);
183            cursor = at + replacement.len();
184        } else {
185            // Not in this chunk; try again when its block is flushed.
186            carried.push((anchor_raw, href));
187        }
188    }
189    *queue = carried;
190    out
191}
192
193/// Incremental Markdown serializer: feed finalized, in-document-order batches of
194/// [`Node`]s and receive Markdown chunks whose concatenation is **byte-identical**
195/// to [`to_markdown_images`] over the same nodes. This is the streaming
196/// counterpart of the buffered serializer — used to emit a document's Markdown in
197/// chunks (e.g. page by page, as the parallel PDF pipeline finishes pages) instead
198/// of building the whole string up front.
199///
200/// [`ImageMode::Placeholder`] and [`ImageMode::Embedded`] render inline.
201/// [`ImageMode::Referenced`] additionally hands each picture's bytes out through
202/// [`take_artifacts`](Self::take_artifacts) — construct with
203/// [`with_artifacts`](Self::with_artifacts) and drain after every push so the
204/// bytes can be written to disk as pages finish instead of accumulating for the
205/// whole document (issue #80's memory-bounded image handling).
206///
207/// Each [`push`](Self::push) must contain whole blocks in reading order: a caller
208/// must not split a run of list items across two pushes (the run would render as
209/// two separate lists). Finalized PDF page batches already satisfy this.
210pub struct MarkdownStreamer {
211    strict: bool,
212    images: ImageMode,
213    compact_tables: bool,
214    /// Whether any non-empty chunk has been emitted yet (drives `\n\n` joins and
215    /// the trailing newline).
216    emitted_any: bool,
217    /// Recovered links not yet placed (strict mode), consumed in document order.
218    links: Vec<(String, String)>,
219    /// Referenced mode: the link prefix, the not-yet-drained `(path, bytes)`
220    /// artifacts, and the running image number (continues across pushes so the
221    /// stream matches the buffered serializer's `image_000000…` numbering).
222    artifacts_dir: String,
223    artifacts: Vec<(String, Vec<u8>)>,
224    pic_index: usize,
225    /// [`DoclingDocument::page_break_placeholder`] and the boundary carried
226    /// over from the previous push (a page batch opens with its page marker,
227    /// so the break it implies is paid by that batch's first block).
228    page_break: Option<String>,
229    pending_page_break: bool,
230}
231
232impl MarkdownStreamer {
233    /// Create a streamer. `compact_tables` mirrors [`DoclingDocument::compact_tables`].
234    /// For [`ImageMode::Referenced`] use [`with_artifacts`](Self::with_artifacts).
235    pub fn new(strict: bool, images: ImageMode, compact_tables: bool) -> Self {
236        debug_assert!(
237            images != ImageMode::Referenced,
238            "referenced image mode needs an artifacts dir; use with_artifacts"
239        );
240        Self::with_artifacts(strict, images, compact_tables, "artifacts")
241    }
242
243    /// Like [`new`](Self::new) but with the artifacts link prefix, allowing
244    /// [`ImageMode::Referenced`]: pictures render as
245    /// `![Image](<artifacts_dir>/image_NNNNNN.<ext>)` and each push's image
246    /// bytes wait in [`take_artifacts`](Self::take_artifacts) for the caller to
247    /// write. The concatenated chunks and the artifact list match the buffered
248    /// [`to_markdown_images`] byte-for-byte.
249    pub fn with_artifacts(
250        strict: bool,
251        images: ImageMode,
252        compact_tables: bool,
253        artifacts_dir: &str,
254    ) -> Self {
255        Self {
256            strict,
257            images,
258            compact_tables,
259            emitted_any: false,
260            links: Vec::new(),
261            artifacts_dir: artifacts_dir.to_string(),
262            artifacts: Vec::new(),
263            pic_index: 0,
264            page_break: None,
265            pending_page_break: false,
266        }
267    }
268
269    /// Insert `placeholder` between pages, mirroring
270    /// [`DoclingDocument::page_break_placeholder`] for the buffered path (the
271    /// concatenated chunks stay byte-identical to it). `None` — the default —
272    /// omits page breaks. Set before the first [`push`](Self::push).
273    pub fn with_page_break_placeholder(mut self, placeholder: Option<String>) -> Self {
274        self.page_break = placeholder;
275        self
276    }
277
278    /// The `(relative path, bytes)` of images rendered by pushes since the last
279    /// drain ([`ImageMode::Referenced`] only — empty otherwise). Paths are
280    /// relative to the Markdown file, i.e. they start with the configured
281    /// artifacts dir.
282    pub fn take_artifacts(&mut self) -> Vec<(String, Vec<u8>)> {
283        std::mem::take(&mut self.artifacts)
284    }
285
286    /// Render one finalized batch of nodes (plus any links recovered from the same
287    /// span, in document order) into the next Markdown chunk. Returns an empty
288    /// string when the batch produces no output (e.g. empty tables/pictures), in
289    /// which case nothing should be written.
290    pub fn push(&mut self, nodes: &[Node], links: &[(String, String)]) -> String {
291        self.links.extend(links.iter().cloned());
292        let mut ctx = Ctx {
293            strict: self.strict,
294            compact_tables: self.compact_tables,
295            images: self.images,
296            artifacts_dir: std::mem::take(&mut self.artifacts_dir),
297            artifacts: std::mem::take(&mut self.artifacts),
298            pic_index: self.pic_index,
299            in_table_cell: false,
300            page_break: std::mem::take(&mut self.page_break),
301            pending_page_break: self.pending_page_break,
302            emitted_any: self.emitted_any,
303        };
304        let mut blocks: Vec<String> = Vec::new();
305        render(nodes, &mut blocks, &mut ctx);
306        self.artifacts_dir = std::mem::take(&mut ctx.artifacts_dir);
307        self.artifacts = std::mem::take(&mut ctx.artifacts);
308        self.pic_index = ctx.pic_index;
309        self.page_break = std::mem::take(&mut ctx.page_break);
310        self.pending_page_break = ctx.pending_page_break;
311        if blocks.is_empty() {
312            return String::new();
313        }
314        let mut body = blocks.join("\n\n");
315        if self.strict && !self.links.is_empty() {
316            body = apply_links_chunk(&body, &mut self.links);
317        }
318        let chunk = if self.emitted_any {
319            format!("\n\n{body}")
320        } else {
321            body
322        };
323        self.emitted_any = true;
324        chunk
325    }
326
327    /// Emit the trailing newline that finishes the document (empty if no content
328    /// was produced). Call exactly once, after the final [`push`](Self::push).
329    pub fn finish(self) -> String {
330        if self.emitted_any {
331            "\n".to_string()
332        } else {
333            String::new()
334        }
335    }
336}
337
338/// In `strict` mode, rewrite inline text for readability rather than byte-for-byte
339/// docling fidelity: undo the legacy `\_` underscore escaping, and tighten stray
340/// spaces around punctuation (`[ 37 , 36 ]` → `[37, 36]`, `( x )` → `(x)`). This
341/// cleans up both the PDF backend's glyph-split spacing and the space the legacy
342/// emphasis serialization leaves before punctuation (`*a* ,` → `*a*,`).
343/// Legacy/default output keeps docling's spacing untouched. Only inline text
344/// nodes pass through here — code blocks and table cells are left alone.
345fn strict_text(text: &str, strict: bool) -> String {
346    if !strict {
347        return text.to_string();
348    }
349    text.replace("\\_", "_")
350        .replace(" ,", ",")
351        .replace(" .", ".")
352        .replace(" ;", ";")
353        .replace(" )", ")")
354        .replace("( ", "(")
355        .replace(" ]", "]")
356        .replace("[ ", "[")
357}
358
359/// docling-core 2.92's `_md_line_breaks` (docling-core#721): a single `\n`
360/// inside an item's text becomes a GFM hard line break (`"  \n"`, two trailing
361/// spaces) so renderers honour it; a blank line (`\n\n`) is a paragraph break
362/// and stays as is — the document scope already joins blocks with `\n\n`.
363/// Applied to body text, list items and captions, never to code/formulas.
364fn md_line_breaks(text: &str) -> String {
365    if !text.contains('\n') {
366        return text.to_string();
367    }
368    text.split("\n\n")
369        .map(|para| para.replace('\n', "  \n"))
370        .collect::<Vec<_>>()
371        .join("\n\n")
372}
373
374/// Undo [`md_line_breaks`] on a rich table cell's flattened Markdown so the
375/// non-Markdown exports (JSON `text`, LaTeX cells) see the cell's raw line
376/// breaks, as docling's do — a rich cell's text is its Markdown serialization
377/// in our model, and the two trailing spaces are a Markdown-only marker.
378pub(crate) fn strip_hard_breaks(text: &str) -> String {
379    if text.contains("  \n") {
380        text.replace("  \n", "\n")
381    } else {
382        text.to_string()
383    }
384}
385
386/// docling-core's `_heading_line_breaks`: a GFM heading cannot span lines, so a
387/// newline inside heading text collapses to a space (`# Hello World`, not a
388/// broken `# Hello\nWorld`).
389fn heading_line_breaks(text: &str) -> String {
390    text.replace('\n', " ")
391}
392
393fn render(nodes: &[Node], blocks: &mut Vec<String>, ctx: &mut Ctx) {
394    let mut i = 0;
395    while i < nodes.len() {
396        let before = blocks.len();
397        match &nodes[i] {
398            // A page boundary: the explicit `PageBreak` (slides, DjVu / DocTags
399            // pages) or the `PageInfo` marker that opens every PDF page and
400            // spreadsheet sheet — a sheet boundary carries both, and the flag
401            // absorbs the pair into one break. docling's `_iterate_items`
402            // yields a `_PageBreakNode` only between two items on different
403            // pages (a group's leading item counts for the group), which is
404            // exactly "a boundary between two rendered blocks": nothing before
405            // the first block, nothing after the last, consecutive boundaries
406            // — empty or furniture-only pages — collapsed into one.
407            Node::PageBreak | Node::PageInfo { .. } => {
408                if ctx.page_break.is_some() && ctx.emitted_any {
409                    ctx.pending_page_break = true;
410                }
411                i += 1;
412            }
413            Node::ListItem { .. } => {
414                let start = i;
415                i += 1;
416                loop {
417                    match nodes.get(i) {
418                        Some(Node::ListItem { .. }) => i += 1,
419                        // An empty paragraph between two list items is absorbed
420                        // into the run — docling keeps such a ListGroup
421                        // contiguous rather than splitting it.
422                        Some(Node::Paragraph { text })
423                            if text.is_empty()
424                                && matches!(nodes.get(i + 1), Some(Node::ListItem { .. })) =>
425                        {
426                            i += 1
427                        }
428                        _ => break,
429                    }
430                }
431                render_list_run(&nodes[start..i], blocks, ctx.strict);
432            }
433            other => {
434                render_one(other, blocks, ctx);
435                i += 1;
436            }
437        }
438        if blocks.len() > before {
439            if ctx.pending_page_break {
440                // The placeholder is a block of its own, joined by the document
441                // delimiter like docling's `_PageBreakSerResult` part — an
442                // empty placeholder therefore leaves the doubled `\n\n`
443                // upstream leaves too.
444                if let Some(placeholder) = &ctx.page_break {
445                    blocks.insert(before, placeholder.clone());
446                }
447                ctx.pending_page_break = false;
448            }
449            ctx.emitted_any = true;
450        }
451    }
452}
453
454/// Render a contiguous run of list items.
455///
456/// Ordered items use their explicit `number`. A new sibling list (marked by
457/// `first_in_list`) at the same depth is separated by a blank line, matching
458/// docling-core's serializer.
459fn render_list_run(items: &[Node], blocks: &mut Vec<String>, strict: bool) {
460    let mut lines: Vec<String> = Vec::new();
461    // Whether a top-level item has been rendered yet — a fresh-list flag on
462    // the very first item opens nothing.
463    let mut any_top = false;
464
465    for item in items {
466        let Node::ListItem {
467            ordered,
468            number,
469            first_in_list,
470            text,
471            level,
472            marker: orig_marker,
473            location: _,
474            dclx: _,
475            href: _,
476            layer,
477        } = item
478        else {
479            continue;
480        };
481        // A non-body (furniture) list item is omitted from Markdown, matching
482        // docling's content-layer filtering.
483        if layer.is_some() {
484            continue;
485        }
486        let level = *level as usize;
487
488        // A new sibling list at the top level gets a blank line — and only the
489        // backend knows where one starts (`first_in_list`: Word's `numId`
490        // changing, an HTML `<ul>` closing, a Markdown bullet switching
491        // `-`→`*`). The serializer used to guess it from a kind flip or a
492        // number gap as well, which split lists docling keeps whole (an
493        // AsciiDoc `1.` … `5.`, mixed `*`/`1.` markers) — #385. Only at the
494        // top level: nested sibling groups are children of a list item, and
495        // docling joins an item's children without blank lines.
496        if level == 0 {
497            if any_top && *first_in_list {
498                lines.push(String::new());
499            }
500            any_top = true;
501        }
502
503        let indent = "    ".repeat(level);
504        // docling-core's `case_already_valid`: a marker of digits and a dot
505        // prints verbatim — Python's `\d+\.` admits every Unicode decimal
506        // digit, so a DOCX `decimalFullWidth` marker (`1.`, docling#4336)
507        // is kept as it is rather than renumbered in ASCII.
508        let verbatim = orig_marker.as_deref().filter(|m| {
509            m.strip_suffix('.')
510                .is_some_and(|d| !d.is_empty() && d.chars().all(char::is_numeric))
511        });
512        let marker = match verbatim {
513            Some(m) if *ordered => m.to_string(),
514            _ if *ordered => format!("{number}."),
515            _ => "-".to_string(),
516        };
517        lines.push(format!("{indent}{marker} {}", list_item_text(text, strict)));
518    }
519
520    // A run consisting only of furniture (content-layer-filtered) items yields no
521    // lines; pushing an empty block here would surface as a stray blank line.
522    if !lines.is_empty() {
523        blocks.push(lines.join("\n"));
524    }
525}
526
527/// A list item's Markdown body. The GFM hard-line-break rule (docling-core#721)
528/// applies to the item's own text; pictures the HTML backend folded into the
529/// item (`"\n[alt\n]<!-- image -->"` per `<img>` inside the `<li>`) are
530/// docling's picture *children* of the item, which its serializer prints after
531/// the item line with plain newlines — so a folded tail keeps its newlines
532/// unmarked. The tail is recognised structurally: every line after the first is
533/// an image marker or an alt caption directly followed by one.
534fn list_item_text(text: &str, strict: bool) -> String {
535    let escaped = strict_text(text, strict);
536    if let Some((own, tail)) = escaped.split_once('\n') {
537        if is_folded_child_tail(tail) {
538            return format!("{}\n{tail}", md_line_breaks(own));
539        }
540    }
541    md_line_breaks(&escaped)
542}
543
544/// Whether everything after a list item's own first line is a folded *child*
545/// block rather than a continuation of the item's text: an image marker
546/// (optionally preceded by its caption/alt line) or a fenced code block. The
547/// AsciiDoc backend indents such a block to the item's own depth (as
548/// docling-core's list serializer does for each part it emits), so a leading
549/// indent is ignored here.
550fn is_folded_child_tail(tail: &str) -> bool {
551    const MARKER: &str = "<!-- image -->";
552    const FENCE: &str = "```";
553    let mut lines = tail.split('\n').peekable();
554    let mut any = false;
555    while let Some(line) = lines.next() {
556        let line = line.trim_start();
557        if line == MARKER {
558            any = true;
559        } else if line == FENCE {
560            // Skip the block's body; an unclosed fence is not a folded child.
561            loop {
562                match lines.next() {
563                    Some(l) if l.trim_start() == FENCE => break,
564                    Some(_) => {}
565                    None => return false,
566                }
567            }
568            any = true;
569        } else if lines.next().map(str::trim_start) == Some(MARKER) {
570            any = true; // an alt caption line, then its marker
571        } else {
572            return false;
573        }
574    }
575    any
576}
577
578fn render_one(node: &Node, blocks: &mut Vec<String>, ctx: &mut Ctx) {
579    match node {
580        Node::Heading { level, text } => {
581            let text = heading_line_breaks(&strict_text(text, ctx.strict));
582            if ctx.in_table_cell {
583                // docling-core#540: no `#` markers inside a table cell.
584                blocks.push(text);
585            } else {
586                let hashes = "#".repeat((*level).clamp(1, 6) as usize);
587                blocks.push(format!("{hashes} {text}"));
588            }
589        }
590        // An empty body paragraph (docling's blank-line text item) contributes
591        // nothing to Markdown — only DocLang/JSON keep it.
592        Node::Paragraph { text } if text.is_empty() => {}
593        Node::Paragraph { text } => blocks.push(md_line_breaks(&strict_text(text, ctx.strict))),
594        // A standalone caption item renders like a text item; its hyperlink
595        // annotation becomes a Markdown link around the whole caption.
596        Node::Caption { text, .. } if text.is_empty() => {}
597        Node::Caption { text, href } => {
598            let body = md_line_breaks(&strict_text(text, ctx.strict));
599            blocks.push(match href {
600                Some(url) => format!("[{body}]({url})"),
601                None => body,
602            });
603        }
604        Node::CheckboxItem { checked, text } => {
605            let mark = if *checked { "- [x] " } else { "- [ ] " };
606            blocks.push(md_line_breaks(&strict_text(
607                &format!("{mark}{text}"),
608                ctx.strict,
609            )));
610        }
611        Node::Code {
612            language,
613            text,
614            pretty,
615            ..
616        } => {
617            // Legacy docling never emits a language on the fence; strict keeps it.
618            let lang = match language {
619                Some(l) if ctx.strict => l.as_str(),
620                _ => "",
621            };
622            // Strict prefers the line-preserving rendering when the backend
623            // supplied one (PDF); legacy stays on docling's flat `text`.
624            let body = match pretty {
625                Some(p) if ctx.strict => p.as_str(),
626                _ => text.as_str(),
627            };
628            blocks.push(format!("```{lang}\n{body}\n```"));
629        }
630        // A CodeFormula-decoded display formula renders as docling's `$$…$$`
631        // (the un-enriched pipeline emits a placeholder paragraph instead).
632        Node::Formula { latex, .. } => blocks.push(format!("$${latex}$$")),
633        Node::Table(table) => {
634            // docling renders a table's caption as a text line before the grid.
635            // `caption` is already escaped (backend convention), like a paragraph.
636            if let Some(cap) = &table.caption {
637                if !cap.is_empty() {
638                    blocks.push(md_line_breaks(&strict_text(cap, ctx.strict)));
639                }
640            }
641            let rendered = render_table(table, ctx.compact_tables);
642            if !rendered.is_empty() {
643                blocks.push(rendered);
644            }
645        }
646        // Classification predictions don't affect docling's Markdown output.
647        Node::Picture { caption, image, .. } => {
648            if let Some(cap) = caption {
649                if !cap.is_empty() {
650                    blocks.push(md_line_breaks(cap));
651                }
652            }
653            blocks.push(picture_marker(image.as_ref(), ctx));
654        }
655        // A chart renders as docling's picture-with-meta markdown: the caption,
656        // the placeholder, the humanized classification ("line_chart" ->
657        // "Line chart"), then the chart's data grid as a regular table.
658        Node::Chart {
659            kind,
660            table,
661            caption,
662            ..
663        } => {
664            if let Some(cap) = caption {
665                if !cap.is_empty() {
666                    blocks.push(md_line_breaks(cap));
667                }
668            }
669            blocks.push(picture_marker(None, ctx));
670            blocks.push(humanize_label(kind));
671            let rendered = render_table(table, false);
672            if !rendered.is_empty() {
673                blocks.push(rendered);
674            }
675        }
676        // A DocLang-only node is omitted from Markdown.
677        Node::DoclangOnly(_) => {}
678        // A group on a non-body layer (a hidden spreadsheet sheet) renders
679        // nothing, like every other non-body item.
680        Node::Group { layer: Some(_), .. } => {}
681        Node::Group { children, .. } => render(children, blocks, ctx),
682        Node::FieldRegion { items } => {
683            // The region container and each field item carry no text of their
684            // own; docling-core 2.93 (#724) serializes them to nothing (older
685            // releases emitted a `<!-- missing-text -->` marker for each), so
686            // only an item's marker/key/value appear, as separate paragraphs.
687            for item in items {
688                for part in [&item.marker, &item.key, &item.value].into_iter().flatten() {
689                    blocks.push(md_line_breaks(&strict_text(part, ctx.strict)));
690                }
691            }
692        }
693        // docling's Markdown serializer has no component for a `KeyValueItem`
694        // and writes its fallback placeholder in the item's place.
695        Node::KeyValueGraph { .. } => blocks.push(MISSING_KEY_VALUE_ITEM.to_string()),
696        // A rich inline group renders exactly like a paragraph of its Markdown
697        // text — the structured runs are DocLang-only.
698        Node::InlineGroup { md_text, .. } => {
699            blocks.push(md_line_breaks(&strict_text(md_text, ctx.strict)))
700        }
701        // A plain-text backend dump renders verbatim as a single block.
702        Node::TextDump(text) => {
703            if !text.is_empty() {
704                blocks.push(text.clone());
705            }
706        }
707        // Furniture (page headers/footers, HTML `<title>`) is excluded from
708        // Markdown by default, mirroring docling.
709        Node::Furniture { .. } => {}
710        Node::PageFurniture { .. } => {}
711        // A comment lives in the notes layer — omitted like other furniture;
712        // the annotation on a body item is JSON-only, so render the item.
713        Node::CommentSection { .. } => {}
714        Node::Commented { inner, .. } => render_one(inner, blocks, ctx),
715        // Layout provenance is DocLang-only; render the wrapped node.
716        Node::Located { inner, .. } | Node::Prov { inner, .. } => render_one(inner, blocks, ctx),
717        // Page breaks are DocLang-only; docling omits them from Markdown.
718        Node::PageBreak => {}
719        // Page markers feed the JSON export only.
720        Node::PageInfo { .. } => {}
721        // Runs of adjacent list items are merged by `render`; a stray single
722        // item (a hand-built document, or a `Located` wrapper around one)
723        // still renders as its own one-item list instead of panicking —
724        // `nodes` is public API, so every representable tree must serialize.
725        Node::ListItem { .. } => render_list_run(std::slice::from_ref(node), blocks, ctx.strict),
726    }
727}
728
729/// The Markdown for a picture under the active [`ImageMode`]; Referenced mode also
730/// records the bytes in `ctx.artifacts` for the caller to write.
731/// docling-core's `_humanize_text`: underscores to spaces, first letter
732/// capitalized ("line_chart" -> "Line chart").
733fn humanize_label(label: &str) -> String {
734    let text = label.replace('_', " ");
735    let mut chars = text.chars();
736    match chars.next() {
737        Some(f) => f.to_uppercase().collect::<String>() + chars.as_str(),
738        None => text,
739    }
740}
741
742fn picture_marker(image: Option<&crate::PictureImage>, ctx: &mut Ctx) -> String {
743    match (ctx.images, image) {
744        (ImageMode::Embedded, Some(img)) => format!("![Image]({})", img.data_uri()),
745        (ImageMode::Referenced, Some(img)) => {
746            let path = format!(
747                "{}/image_{:06}.{}",
748                ctx.artifacts_dir,
749                ctx.pic_index,
750                ext_for(&img.mimetype)
751            );
752            ctx.pic_index += 1;
753            ctx.artifacts.push((path.clone(), img.data.clone()));
754            format!("![Image]({})", escape_uri_path(&path))
755        }
756        // Placeholder, or any mode with no extracted image.
757        _ => "<!-- image -->".to_string(),
758    }
759}
760
761/// Encode a URL or filesystem path as a Markdown link destination —
762/// docling-core's `MarkdownPictureSerializer._escape_uri_path`
763/// (docling-core#698, 2.94). Handles URLs of any scheme as well as POSIX and
764/// Windows paths, keeps relative paths relative and never double-encodes:
765/// backslashes become `/` (a backslash is both the Windows separator and a
766/// Markdown escape), a UNC share `//host/…` and an absolute Windows path
767/// `C:/…` become RFC 8089 `file://` URLs (the one spelling a renderer cannot
768/// misread as a scheme-relative URL or a `C:` scheme), a URL keeps its
769/// scheme / authority / delimiters with only the components encoded, and
770/// everything else is percent-encoded as a path. `%` is kept so an
771/// already-encoded destination stays as it is; spaces and parentheses are
772/// encoded because they would end (or unbalance) a Markdown inline link.
773pub(crate) fn escape_uri_path(value: &str) -> String {
774    const KEEP: &str = "/%:@+,;=~$!&'*";
775    let s = value.replace('\\', "/");
776    if let Some(rest) = s.strip_prefix("//") {
777        // A fileshare: `file://<host>/<path>`, the host possibly empty.
778        let rest = rest.trim_start_matches('/');
779        let (host, tail) = rest.split_once('/').unwrap_or((rest, ""));
780        return format!("file://{host}{}", percent_quote(&format!("/{tail}"), KEEP));
781    }
782    let bytes = s.as_bytes();
783    if bytes.len() >= 3 && bytes[0].is_ascii_alphabetic() && bytes[1] == b':' && bytes[2] == b'/' {
784        // A Windows path with a drive letter: `file:///C:/…`.
785        return format!("file:///{}", percent_quote(&s, KEEP));
786    }
787    // A URL keeps its scheme, authority and delimiters; only its components are
788    // encoded. A single-character scheme cannot be real (it is a drive letter,
789    // handled above), so it is read as a path — like `urlsplit`.
790    if let Some((scheme, rest)) = s.split_once(':') {
791        let valid_scheme = scheme.len() > 1
792            && scheme.as_bytes()[0].is_ascii_alphabetic()
793            && scheme
794                .bytes()
795                .all(|b| b.is_ascii_alphanumeric() || matches!(b, b'+' | b'-' | b'.'));
796        if valid_scheme {
797            let (authority, rest) = match rest.strip_prefix("//") {
798                Some(r) => {
799                    let end = r.find(['/', '?', '#']).unwrap_or(r.len());
800                    (Some(&r[..end]), &r[end..])
801                }
802                None => (None, rest),
803            };
804            let (before_frag, fragment) = rest.split_once('#').unwrap_or((rest, ""));
805            let (path, query) = before_frag.split_once('?').unwrap_or((before_frag, ""));
806            let mut out = format!("{scheme}:");
807            if let Some(a) = authority {
808                out.push_str("//");
809                out.push_str(a);
810            }
811            out.push_str(&percent_quote(path, KEEP));
812            if !query.is_empty() {
813                out.push('?');
814                out.push_str(&percent_quote(query, KEEP));
815            }
816            if !fragment.is_empty() {
817                out.push('#');
818                out.push_str(&percent_quote(fragment, KEEP));
819            }
820            return out;
821        }
822    }
823    // A relative or root-relative local path.
824    percent_quote(&s, KEEP)
825}
826
827/// `urllib.parse.quote(s, safe)`: unreserved ASCII (`A–Z a–z 0–9 _ . - ~`) and
828/// the `safe` set stay, every other byte of the UTF-8 encoding becomes `%XX`.
829fn percent_quote(s: &str, safe: &str) -> String {
830    let mut out = String::with_capacity(s.len());
831    for &b in s.as_bytes() {
832        let keep = b.is_ascii_alphanumeric()
833            || matches!(b, b'_' | b'.' | b'-' | b'~')
834            || (b.is_ascii() && safe.contains(b as char));
835        if keep {
836            out.push(b as char);
837        } else {
838            out.push_str(&format!("%{b:02X}"));
839        }
840    }
841    out
842}
843
844fn ext_for(mimetype: &str) -> &str {
845    match mimetype {
846        "image/jpeg" => "jpg",
847        "image/gif" => "gif",
848        "image/webp" => "webp",
849        "image/bmp" => "bmp",
850        "image/tiff" => "tif",
851        _ => "png",
852    }
853}
854
855/// Render a table. `compact` selects between two serializers:
856///
857/// - **padded** (default) — docling-core's `tabulate(tablefmt="github")`: columns
858///   are padded to a fixed width (header width + a minimum padding of 2, or the
859///   widest data cell); numeric columns (every data cell parses as a number) are
860///   right-aligned, others left-aligned; separators are plain dashes of
861///   `width + 2`. Matches current published docling (DOCX/HTML conformance).
862/// - **compact** — `| a | b |` cells with single-dash `| - | - |` separators, no
863///   width padding. Matches the committed PDF groundtruth corpus, which predates
864///   the padded serializer.
865///
866/// Each cell is first escaped (`\n` → space, `|` → `&#124;`) so it can't break the
867/// table. The header row is the table's leading `column_header` block flattened
868/// to one row ([`Table::header_row_count`] + [`flatten_header_rows`],
869/// docling-core#723); alignment and widths are computed over the body rows.
870/// Whether a table cell counts as a number for column alignment, matching
871/// `tabulate`'s detection: an ordinary float/int (`f64`-parseable, covering
872/// `1e2`/`inf`/`+1.5`) **or** a thousands-separated number like `7,015`.
873fn is_number_cell(t: &str) -> bool {
874    t.parse::<f64>().is_ok() || is_thousands_number(t)
875}
876
877/// A number with comma thousands-separators, per `tabulate`'s
878/// `_float_with_thousands_separators` regex
879/// (`^(([+-]?[0-9]{1,3})(?:,([0-9]{3}))*)?(?(1)\.[0-9]*|\.[0-9]+)?$`): the
880/// integer part is 1–3 digits then any number of `,ddd` groups; the fraction is
881/// optional (and, without an integer part, must have at least one digit).
882fn is_thousands_number(t: &str) -> bool {
883    let b = t.as_bytes();
884    let mut i = 0;
885    let start = i;
886    if i < b.len() && (b[i] == b'+' || b[i] == b'-') {
887        i += 1;
888    }
889    // First digit chunk: 1–3 digits.
890    let d0 = i;
891    while i < b.len() && b[i].is_ascii_digit() && i - d0 < 3 {
892        i += 1;
893    }
894    let has_int = i > d0;
895    if has_int {
896        // Subsequent `,ddd` groups (exactly three digits each).
897        while i + 3 < b.len() + 1
898            && b.get(i) == Some(&b',')
899            && b.get(i + 1).is_some_and(u8::is_ascii_digit)
900            && b.get(i + 2).is_some_and(u8::is_ascii_digit)
901            && b.get(i + 3).is_some_and(u8::is_ascii_digit)
902        {
903            i += 4;
904        }
905    } else {
906        // A sign only counts with an integer part.
907        i = start;
908    }
909    // Optional fraction.
910    if i < b.len() && b[i] == b'.' {
911        i += 1;
912        let f0 = i;
913        while i < b.len() && b[i].is_ascii_digit() {
914            i += 1;
915        }
916        if !has_int && i == f0 {
917            return false; // `.` with no digits and no integer part
918        }
919    } else if !has_int {
920        return false; // neither integer nor fractional part
921    }
922    i == b.len()
923}
924
925/// The single GFM header row for a table: the leading header rows (see
926/// [`Table::header_row_count`]) flattened per column, texts joined with
927/// `" - "` after dropping consecutive duplicates — docling-core's
928/// `_flatten_header_rows` (docling-core#723). The duplicate rule is what
929/// keeps a row-spanning header from being joined to itself (the grid repeats
930/// its text into every row it covers); it is position-based, so two stacked
931/// levels sharing a label collapse too — GFM has one header row, and upstream
932/// accepts that loss. No header rows → one empty header cell per column.
933fn flatten_header_rows(header_rows: &[Vec<String>], num_cols: usize) -> Vec<String> {
934    (0..num_cols)
935        .map(|c| {
936            let mut parts: Vec<&str> = Vec::new();
937            for row in header_rows {
938                let text = row.get(c).map(String::as_str).unwrap_or("");
939                if !text.is_empty() && parts.last() != Some(&text) {
940                    parts.push(text);
941                }
942            }
943            parts.join(" - ")
944        })
945        .collect()
946}
947
948pub(crate) fn render_table(table: &Table, compact: bool) -> String {
949    if table.rows.is_empty() {
950        return String::new();
951    }
952    let num_cols = table.rows.iter().map(Vec::len).max().unwrap_or(0);
953    if num_cols == 0 {
954        return String::new();
955    }
956
957    // Escaped, rectangular grid (ragged rows padded with empty cells). The
958    // header block is resolved to the one row GFM allows (docling-core#723);
959    // `tabulate` strips data cells of surrounding whitespace but leaves the
960    // header texts as-is.
961    let num_headers = table.header_row_count().min(table.rows.len());
962    let escaped = |r: usize| -> Vec<String> {
963        (0..num_cols)
964            .map(|c| escape_cell(table.rows[r].get(c).map(String::as_str).unwrap_or("")))
965            .collect()
966    };
967    let header_rows: Vec<Vec<String>> = (0..num_headers).map(escaped).collect();
968    let header = flatten_header_rows(&header_rows, num_cols);
969    let body: Vec<Vec<String>> = (num_headers..table.rows.len())
970        .map(|r| {
971            escaped(r)
972                .into_iter()
973                .map(|c| c.trim().to_string())
974                .collect()
975        })
976        .collect();
977
978    if compact {
979        // Compact: cells joined by " | ", no padding, single-dash separators.
980        let render_row = |row: &[String]| -> String { format!("| {} |", row.join(" | ")) };
981        let mut lines = Vec::with_capacity(body.len() + 2);
982        lines.push(render_row(&header));
983        let sep: Vec<&str> = (0..num_cols).map(|_| "-").collect();
984        lines.push(format!("| {} |", sep.join(" | ")));
985        for row in &body {
986            lines.push(render_row(row));
987        }
988        return lines.join("\n");
989    }
990
991    // Display width (Unicode scalar count — good enough for now).
992    let dw = |s: &str| s.chars().count();
993
994    // A column is right-aligned when at least one body cell is numeric and every
995    // non-empty body cell is numeric — matching `tabulate`'s column typing, where
996    // empty cells are "missing" (ignored) and a number may carry thousands
997    // separators (`7,015`), which a plain `f64` parse rejects.
998    let right: Vec<bool> = (0..num_cols)
999        .map(|c| {
1000            let mut any = false;
1001            for row in &body {
1002                let t = row[c].trim();
1003                if t.is_empty() {
1004                    continue;
1005                }
1006                if !is_number_cell(t) {
1007                    return false;
1008                }
1009                any = true;
1010            }
1011            any
1012        })
1013        .collect();
1014
1015    // Column width = max(header_width + MIN_PADDING(2), max body-cell width).
1016    let width: Vec<usize> = (0..num_cols)
1017        .map(|c| {
1018            let mut w = dw(&header[c]) + 2;
1019            for row in &body {
1020                w = w.max(dw(&row[c]));
1021            }
1022            w
1023        })
1024        .collect();
1025
1026    let fmt_cell = |s: &str, c: usize| -> String {
1027        let pad = " ".repeat(width[c].saturating_sub(dw(s)));
1028        let body = if right[c] {
1029            format!("{pad}{s}")
1030        } else {
1031            format!("{s}{pad}")
1032        };
1033        format!(" {body} ")
1034    };
1035    let render_row = |row: &[String]| -> String {
1036        let cells: Vec<String> = (0..num_cols).map(|c| fmt_cell(&row[c], c)).collect();
1037        format!("|{}|", cells.join("|"))
1038    };
1039
1040    let mut lines = Vec::with_capacity(body.len() + 2);
1041    lines.push(render_row(&header));
1042    let sep: Vec<String> = (0..num_cols).map(|c| "-".repeat(width[c] + 2)).collect();
1043    lines.push(format!("|{}|", sep.join("|")));
1044    for row in &body {
1045        lines.push(render_row(row));
1046    }
1047    lines.join("\n")
1048}
1049
1050/// Escape a table cell so it can't break the markdown table: newlines become
1051/// spaces and pipes become the `&#124;` HTML entity (matches docling-core).
1052fn escape_cell(s: &str) -> String {
1053    s.replace('\n', " ").replace('|', "&#124;")
1054}
1055
1056#[cfg(test)]
1057mod tests {
1058    use super::*;
1059    use crate::{PictureImage, TableCell, TableStructure};
1060
1061    /// #385: where one list ends and the next begins is the backend's call
1062    /// (`first_in_list`), never the serializer's. An ordered run `1.` → `5.`
1063    /// is one list (an AsciiDoc numbered list around a nested one), and so are
1064    /// mixed bullet/ordered items the backend did not separate; only a flagged
1065    /// item opens a new list and earns the blank line.
1066    #[test]
1067    fn list_boundaries_come_from_the_backend_not_the_numbering() {
1068        let item = |ordered: bool, number: u64, first_in_list: bool, text: &str| Node::ListItem {
1069            ordered,
1070            number,
1071            first_in_list,
1072            text: text.into(),
1073            level: 0,
1074            marker: None,
1075            location: None,
1076            dclx: None,
1077            href: None,
1078            layer: None,
1079        };
1080        let md = |items: Vec<Node>| {
1081            let mut doc = DoclingDocument::new("t");
1082            for n in items {
1083                doc.push(n);
1084            }
1085            doc.export_to_markdown()
1086        };
1087        // A number gap alone is not a boundary.
1088        assert_eq!(
1089            md(vec![
1090                item(true, 1, true, "one"),
1091                item(true, 5, false, "five")
1092            ]),
1093            "1. one\n5. five\n"
1094        );
1095        // Nor is a kind flip the backend did not flag …
1096        assert_eq!(
1097            md(vec![
1098                item(false, 0, true, "bullet"),
1099                item(true, 1, false, "one"),
1100                item(false, 0, false, "bullet two"),
1101            ]),
1102            "- bullet\n1. one\n- bullet two\n"
1103        );
1104        // … while a flagged item is one, whatever its number says.
1105        assert_eq!(
1106            md(vec![
1107                item(true, 1, true, "a"),
1108                item(true, 2, false, "b"),
1109                item(true, 3, true, "new list, continuing count"),
1110            ]),
1111            "1. a\n2. b\n\n3. new list, continuing count\n"
1112        );
1113    }
1114
1115    #[test]
1116    fn renders_headings_paragraphs_and_lists() {
1117        let mut doc = DoclingDocument::new("demo");
1118        doc.add_heading(1, "Title");
1119        doc.add_paragraph("Hello world.");
1120        doc.push(Node::ListItem {
1121            ordered: false,
1122            number: 1,
1123            first_in_list: true,
1124            text: "first".into(),
1125            level: 0,
1126            marker: None,
1127            location: None,
1128            dclx: None,
1129            href: None,
1130            layer: None,
1131        });
1132        doc.push(Node::ListItem {
1133            ordered: false,
1134            number: 2,
1135            first_in_list: false,
1136            text: "second".into(),
1137            level: 0,
1138            marker: None,
1139            location: None,
1140            dclx: None,
1141            href: None,
1142            layer: None,
1143        });
1144        let md = doc.export_to_markdown();
1145        assert_eq!(md, "# Title\n\nHello world.\n\n- first\n- second\n");
1146    }
1147
1148    /// docling-core 2.92 (#721): a single newline inside an item's text is a
1149    /// GFM hard line break, a blank line stays a paragraph break, and a heading
1150    /// collapses its newline to a space. Nested-table dumps stay verbatim.
1151    #[test]
1152    fn single_newlines_become_gfm_hard_line_breaks() {
1153        let mut doc = DoclingDocument::new("t");
1154        doc.push(Node::Heading {
1155            level: 1,
1156            text: "Hello\nWorld".into(),
1157        });
1158        doc.push(Node::Paragraph {
1159            text: "line one\nline two\n\npara two".into(),
1160        });
1161        doc.push(Node::ListItem {
1162            ordered: false,
1163            number: 1,
1164            first_in_list: true,
1165            text: "item\ncontinued".into(),
1166            level: 0,
1167            marker: None,
1168            location: None,
1169            dclx: None,
1170            href: None,
1171            layer: None,
1172        });
1173        doc.push(Node::TextDump("A1 B1 \n\n\nC1".into()));
1174        assert_eq!(
1175            doc.export_to_markdown(),
1176            "# Hello World\n\nline one  \nline two\n\npara two\n\n- item  \ncontinued\n\nA1 B1 \n\n\nC1\n"
1177        );
1178    }
1179
1180    /// docling-core#540: inside a rich table cell a heading is plain text;
1181    /// docling-core#724: a field region renders only its items' key/value text.
1182    #[test]
1183    fn table_cell_mode_and_field_regions() {
1184        let mut doc = DoclingDocument::new("t");
1185        doc.push(Node::Heading {
1186            level: 2,
1187            text: "A  text".into(),
1188        });
1189        doc.push(Node::Paragraph {
1190            text: "body".into(),
1191        });
1192        assert_eq!(to_markdown_table_cell(&doc, false), "A  text\n\nbody");
1193        assert_eq!(doc.export_to_markdown(), "## A  text\n\nbody\n");
1194
1195        let mut doc = DoclingDocument::new("f");
1196        doc.push(Node::FieldRegion {
1197            items: vec![crate::FieldItem {
1198                marker: None,
1199                key: Some("Name:".into()),
1200                value: Some("John Doe".into()),
1201                value_kind: None,
1202            }],
1203        });
1204        assert_eq!(doc.export_to_markdown(), "Name:\n\nJohn Doe\n");
1205    }
1206
1207    #[test]
1208    fn strict_renders_recovered_links_legacy_does_not() {
1209        let mut doc = DoclingDocument::new("cv");
1210        doc.add_paragraph("Find me on LinkedIn or GitHub.");
1211        doc.links = vec![
1212            ("LinkedIn".into(), "https://www.linkedin.com/in/x/".into()),
1213            ("GitHub".into(), "https://github.com/x/".into()),
1214        ];
1215        // Legacy/docling mode: links are left untouched (conformance preserved).
1216        assert_eq!(doc.export_to_markdown(), "Find me on LinkedIn or GitHub.\n");
1217        // Strict mode: anchors become Markdown links.
1218        assert_eq!(
1219            doc.export_to_markdown_with(true),
1220            "Find me on [LinkedIn](https://www.linkedin.com/in/x/) or [GitHub](https://github.com/x/).\n"
1221        );
1222    }
1223
1224    #[test]
1225    fn strict_links_match_escaped_anchor_and_consume_in_order() {
1226        let mut doc = DoclingDocument::new("d");
1227        // The PDF assembler HTML-escapes prose, so by serialization time the body
1228        // already carries `&amp;`; the anchor is stored un-escaped. The matcher must
1229        // escape the anchor to find it. Two identical anchors link in document order.
1230        doc.add_paragraph("AI &amp; ML here, and issues here, then issues there.");
1231        doc.links = vec![
1232            ("AI & ML".into(), "https://a/".into()),
1233            ("issues".into(), "https://first/".into()),
1234            ("issues".into(), "https://second/".into()),
1235        ];
1236        assert_eq!(
1237            doc.export_to_markdown_with(true),
1238            "[AI &amp; ML](https://a/) here, and [issues](https://first/) here, then [issues](https://second/) there.\n"
1239        );
1240    }
1241
1242    /// docling-core#698: the referenced-image destination is percent-encoded —
1243    /// upstream's own case table (paths, Windows flavours, UNC, URLs) plus
1244    /// idempotency on the encoded result.
1245    #[test]
1246    fn referenced_image_destinations_are_escaped() {
1247        let cases = [
1248            (
1249                "doc_artifacts/image_000001_ab12.png",
1250                "doc_artifacts/image_000001_ab12.png",
1251            ),
1252            (
1253                "My Report_artifacts/img.png",
1254                "My%20Report_artifacts/img.png",
1255            ),
1256            ("artifacts/img (1).png", "artifacts/img%20%281%29.png"),
1257            ("100%_scale/a#b?c.png", "100%_scale/a%23b%3Fc.png"),
1258            ("/home/a b/img.png", "/home/a%20b/img.png"),
1259            (
1260                "My Report_artifacts\\img.png",
1261                "My%20Report_artifacts/img.png",
1262            ),
1263            (
1264                "C:/Users/me/My Docs/img.png",
1265                "file:///C:/Users/me/My%20Docs/img.png",
1266            ),
1267            ("C:\\Users\\me\\img.png", "file:///C:/Users/me/img.png"),
1268            (
1269                "//server/share/My Docs/img.png",
1270                "file://server/share/My%20Docs/img.png",
1271            ),
1272            ("\\\\server\\share\\img.png", "file://server/share/img.png"),
1273            ("file:///home/a b/img.png", "file:///home/a%20b/img.png"),
1274            (
1275                "s3://bucket/My Report_artifacts/img.png",
1276                "s3://bucket/My%20Report_artifacts/img.png",
1277            ),
1278            (
1279                "https://example.com:8080/a b.png?w=1&h=2#frag",
1280                "https://example.com:8080/a%20b.png?w=1&h=2#frag",
1281            ),
1282            (
1283                "https://example.com/img (1).png",
1284                "https://example.com/img%20%281%29.png",
1285            ),
1286            ("caf\u{e9}/im\u{e4}ge.png", "caf%C3%A9/im%C3%A4ge.png"),
1287        ];
1288        for (input, expected) in cases {
1289            assert_eq!(escape_uri_path(input), expected, "input {input:?}");
1290            assert_eq!(
1291                escape_uri_path(expected),
1292                expected,
1293                "idempotent {expected:?}"
1294            );
1295        }
1296        // The whole marker, through the referenced-image export.
1297        let mut doc = DoclingDocument::new("t");
1298        doc.push(Node::Picture {
1299            caption: None,
1300            caption_href: None,
1301            image: Some(PictureImage {
1302                mimetype: "image/png".into(),
1303                width: 1,
1304                height: 1,
1305                data: b"x".to_vec(),
1306            }),
1307            classification: None,
1308            caption_parent: Default::default(),
1309        });
1310        let (md, files) = doc
1311            .export_to_markdown_with_images(ImageMode::Referenced, "My Report (final)_artifacts");
1312        assert!(
1313            md.contains("![Image](My%20Report%20%28final%29_artifacts/image_000000.png)"),
1314            "got:\n{md}"
1315        );
1316        // The file path handed back for writing stays unescaped.
1317        assert_eq!(files[0].0, "My Report (final)_artifacts/image_000000.png");
1318    }
1319
1320    /// Pictures the HTML backend folds into a list item print after the item
1321    /// line with plain newlines; a `<br>` newline in the item's own text is
1322    /// still a GFM hard line break.
1323    #[test]
1324    fn folded_list_item_pictures_keep_plain_newlines() {
1325        assert_eq!(
1326            list_item_text("Step\n<!-- image -->", false),
1327            "Step\n<!-- image -->"
1328        );
1329        assert_eq!(
1330            list_item_text("Step\nAlt text\n<!-- image -->\n<!-- image -->", false),
1331            "Step\nAlt text\n<!-- image -->\n<!-- image -->"
1332        );
1333        assert_eq!(
1334            list_item_text("line one\nline two", false),
1335            "line one  \nline two"
1336        );
1337    }
1338
1339    /// docling-core#723: the header block is the leading run of rows on which a
1340    /// `column_header` cell starts, flattened per column with " - ".
1341    #[test]
1342    fn stacked_header_rows_flatten_into_one() {
1343        let mut t = Table {
1344            rows: vec![
1345                vec!["".into(), "% of Total".into(), "% of Total".into()],
1346                vec!["class".into(), "Train".into(), "Test".into()],
1347                vec!["Caption".into(), "2.04".into(), "1.77".into()],
1348            ],
1349            ..Default::default()
1350        };
1351        t.structure = Some(TableStructure {
1352            header_row: vec![true, true, false],
1353            col_continuation: vec![
1354                vec![false, false, true],
1355                vec![false, false, false],
1356                vec![false, false, false],
1357            ],
1358            ..Default::default()
1359        });
1360        assert_eq!(t.header_row_count(), 2);
1361        assert_eq!(
1362            render_table(&t, true),
1363            "| class | % of Total - Train | % of Total - Test |\n| - | - | - |\n| Caption | 2.04 | 1.77 |"
1364        );
1365        // padded: widths from the flattened header, alignment from body rows
1366        assert_eq!(
1367            render_table(&t, false),
1368            "| class   |   % of Total - Train |   % of Total - Test |\n\
1369             |---------|----------------------|---------------------|\n\
1370             | Caption |                 2.04 |                1.77 |"
1371        );
1372    }
1373
1374    /// A header spanning two rows is repeated into the second row by the grid;
1375    /// that row is not a header row unless another header cell starts there.
1376    #[test]
1377    fn vertically_spanning_header_does_not_extend_the_block() {
1378        let mut t = Table {
1379            rows: vec![
1380                vec!["Name".into(), "Value".into()],
1381                vec!["Name".into(), "1".into()],
1382                vec!["x".into(), "2".into()],
1383            ],
1384            ..Default::default()
1385        };
1386        t.structure = Some(TableStructure {
1387            col_header: vec![vec![true, true], vec![true, false], vec![false, false]],
1388            row_continuation: vec![vec![false, false], vec![true, false], vec![false, false]],
1389            ..Default::default()
1390        });
1391        assert_eq!(t.header_row_count(), 1);
1392        assert_eq!(
1393            render_table(&t, true),
1394            "| Name | Value |\n| - | - |\n| Name | 1 |\n| x | 2 |"
1395        );
1396    }
1397
1398    /// Flags that begin on a later row promote nothing: every row stays in the
1399    /// body under an empty header row (tabulate's `headers=["", ""]`).
1400    #[test]
1401    fn header_flags_not_on_row_zero_keep_all_rows_in_the_body() {
1402        let mut t = Table {
1403            rows: vec![
1404                vec!["1".into(), "2".into()],
1405                vec!["a".into(), "b".into()],
1406                vec!["333".into(), "4".into()],
1407            ],
1408            ..Default::default()
1409        };
1410        t.structure = Some(TableStructure {
1411            header_row: vec![false, true, false],
1412            ..Default::default()
1413        });
1414        assert_eq!(t.header_row_count(), 0);
1415        assert_eq!(
1416            render_table(&t, false),
1417            "|     |    |\n|-----|----|\n| 1   | 2  |\n| a   | b  |\n| 333 | 4  |"
1418        );
1419    }
1420
1421    /// A pivot table's row headers (`<th rowspan>`) carry `row_header`, not
1422    /// `column_header` (docling#4216), so the data row beside them is not
1423    /// pulled into the header block — what this port used to reach with a
1424    /// deviation now falls out of the flags themselves.
1425    #[test]
1426    fn pivot_row_headers_do_not_extend_the_header() {
1427        let mut t = Table {
1428            rows: vec![
1429                vec!["Year".into(), "Month".into()],
1430                vec!["2025".into(), "January".into()],
1431                vec!["2025".into(), "February".into()],
1432            ],
1433            ..Default::default()
1434        };
1435        t.structure = Some(TableStructure {
1436            col_header: vec![vec![true, true], vec![false, false], vec![false, false]],
1437            row_header: vec![vec![false, false], vec![true, false], vec![true, false]],
1438            row_continuation: vec![vec![false, false], vec![false, false], vec![true, false]],
1439            ..Default::default()
1440        });
1441        assert_eq!(t.header_row_count(), 1);
1442        assert_eq!(
1443            render_table(&t, true),
1444            "| Year | Month |\n| - | - |\n| 2025 | January |\n| 2025 | February |"
1445        );
1446    }
1447
1448    /// No `column_header` anywhere (first-class cells without flags) → row 0
1449    /// stays the header, as before.
1450    #[test]
1451    fn unflagged_cells_keep_row_zero_as_header() {
1452        let mut t = Table {
1453            rows: vec![vec!["h".into()], vec!["d".into()]],
1454            ..Default::default()
1455        };
1456        t.cells = Some(
1457            [(0usize, "h"), (1, "d")]
1458                .into_iter()
1459                .map(|(r, text)| TableCell {
1460                    text: text.into(),
1461                    bbox: None,
1462                    start_row: r,
1463                    start_col: 0,
1464                    row_span: 1,
1465                    col_span: 1,
1466                    column_header: false,
1467                    row_header: false,
1468                    row_section: false,
1469                })
1470                .collect(),
1471        );
1472        assert_eq!(t.header_row_count(), 1);
1473        assert_eq!(render_table(&t, true), "| h |\n| - |\n| d |");
1474    }
1475
1476    #[test]
1477    fn renders_compact_table() {
1478        let mut doc = DoclingDocument::new("t");
1479        // The compact form is opt-in (the PDF backend sets it); default output uses
1480        // the padded GitHub serializer (covered by the regression fixtures).
1481        doc.compact_tables = true;
1482        doc.push(Node::Table(Table {
1483            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
1484            location: None,
1485            structure: None,
1486            cell_blocks: None,
1487            cells: None,
1488            caption: None,
1489            caption_parent: Default::default(),
1490        }));
1491        let md = doc.export_to_markdown();
1492        assert_eq!(md, "| a | b |\n| - | - |\n| 1 | 2 |\n");
1493    }
1494
1495    #[test]
1496    fn renders_padded_github_table_by_default() {
1497        let mut doc = DoclingDocument::new("t");
1498        doc.push(Node::Table(Table {
1499            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
1500            location: None,
1501            structure: None,
1502            cell_blocks: None,
1503            cells: None,
1504            caption: None,
1505            caption_parent: Default::default(),
1506        }));
1507        let md = doc.export_to_markdown();
1508        // Numeric data columns are right-aligned; columns padded to header+2.
1509        assert_eq!(md, "|   a |   b |\n|-----|-----|\n|   1 |   2 |\n");
1510    }
1511
1512    #[test]
1513    fn strict_unescapes_inline_underscores_legacy_keeps_them() {
1514        let mut doc = DoclingDocument::new("t");
1515        doc.add_heading(1, "a\\_b");
1516        doc.add_paragraph("x\\_y");
1517        doc.push(Node::ListItem {
1518            ordered: false,
1519            number: 1,
1520            first_in_list: true,
1521            text: "i\\_j".into(),
1522            level: 0,
1523            marker: None,
1524            location: None,
1525            dclx: None,
1526            href: None,
1527            layer: None,
1528        });
1529        // Legacy reproduces docling's `\_` escaping byte-for-byte.
1530        assert_eq!(doc.export_to_markdown(), "# a\\_b\n\nx\\_y\n\n- i\\_j\n");
1531        // Strict prefers literal underscores (Rust-only readability mode).
1532        assert_eq!(doc.export_to_markdown_with(true), "# a_b\n\nx_y\n\n- i_j\n");
1533    }
1534
1535    /// Drive a document's nodes through [`MarkdownStreamer`] in the given page
1536    /// splits and assert the concatenated chunks equal the buffered serializer.
1537    fn assert_stream_matches(
1538        doc: &DoclingDocument,
1539        strict: bool,
1540        images: ImageMode,
1541        splits: &[usize],
1542    ) {
1543        let (want, want_artifacts) = to_markdown_images(doc, strict, images, "artifacts");
1544        let mut streamer =
1545            MarkdownStreamer::with_artifacts(strict, images, doc.compact_tables, "artifacts")
1546                .with_page_break_placeholder(doc.page_break_placeholder.clone());
1547        let mut got = String::new();
1548        let mut got_artifacts = Vec::new();
1549        let mut start = 0;
1550        for &end in splits {
1551            // Links only matter in strict mode; feed them all with the first batch
1552            // that has content (document order is preserved by the queue).
1553            let links = if start == 0 {
1554                doc.links.as_slice()
1555            } else {
1556                &[]
1557            };
1558            got.push_str(&streamer.push(&doc.nodes[start..end], links));
1559            // Referenced mode: drain per push, as a real caller writing files
1560            // page by page would — numbering must continue across drains.
1561            got_artifacts.extend(streamer.take_artifacts());
1562            start = end;
1563        }
1564        got.push_str(&streamer.push(
1565            &doc.nodes[start..],
1566            if start == 0 {
1567                doc.links.as_slice()
1568            } else {
1569                &[]
1570            },
1571        ));
1572        got_artifacts.extend(streamer.take_artifacts());
1573        got.push_str(&streamer.finish());
1574        assert_eq!(
1575            got, want,
1576            "streamed output diverged (splits={splits:?}, strict={strict})"
1577        );
1578        assert_eq!(
1579            got_artifacts, want_artifacts,
1580            "streamed artifacts diverged (splits={splits:?}, strict={strict})"
1581        );
1582    }
1583
1584    #[test]
1585    fn streaming_is_byte_identical_to_buffered() {
1586        let mut doc = DoclingDocument::new("d");
1587        doc.add_heading(1, "Title");
1588        doc.add_paragraph("First paragraph.");
1589        doc.push(Node::ListItem {
1590            ordered: false,
1591            number: 1,
1592            first_in_list: true,
1593            text: "a".into(),
1594            level: 0,
1595            marker: None,
1596            location: None,
1597            dclx: None,
1598            href: None,
1599            layer: None,
1600        });
1601        doc.push(Node::ListItem {
1602            ordered: false,
1603            number: 2,
1604            first_in_list: false,
1605            text: "b".into(),
1606            level: 0,
1607            marker: None,
1608            location: None,
1609            dclx: None,
1610            href: None,
1611            layer: None,
1612        });
1613        doc.push(Node::Code {
1614            language: Some("rust".into()),
1615            text: "let x = 1;".into(),
1616            orig: None,
1617            pretty: None,
1618        });
1619        doc.push(Node::Table(Table {
1620            rows: vec![vec!["a".into(), "b".into()], vec!["1".into(), "2".into()]],
1621            location: None,
1622            structure: None,
1623            cell_blocks: None,
1624            cells: None,
1625            caption: None,
1626            caption_parent: Default::default(),
1627        }));
1628        doc.push(Node::Picture {
1629            caption: Some("Fig 1".into()),
1630            caption_href: None,
1631            image: Some(PictureImage {
1632                mimetype: "image/png".into(),
1633                width: 2,
1634                height: 2,
1635                data: b"png-one".to_vec(),
1636            }),
1637            classification: None,
1638            caption_parent: Default::default(),
1639        });
1640        doc.add_paragraph("Last paragraph.");
1641        // A second embedded picture, so referenced mode must keep numbering
1642        // (`image_000001`) across chunk boundaries.
1643        doc.push(Node::Picture {
1644            caption: None,
1645            caption_href: None,
1646            image: Some(PictureImage {
1647                mimetype: "image/png".into(),
1648                width: 2,
1649                height: 2,
1650                data: b"png-two".to_vec(),
1651            }),
1652            classification: None,
1653            caption_parent: Default::default(),
1654        });
1655
1656        // A run of list items must never straddle a split, so try splits that fall
1657        // on safe block boundaries (the streaming PDF assembler guarantees this).
1658        for &strict in &[false, true] {
1659            for &images in &[
1660                ImageMode::Placeholder,
1661                ImageMode::Embedded,
1662                ImageMode::Referenced,
1663            ] {
1664                for splits in [&[][..], &[1][..], &[2][..], &[4][..], &[1, 4, 6, 7][..]] {
1665                    assert_stream_matches(&doc, strict, images, splits);
1666                }
1667            }
1668        }
1669    }
1670
1671    #[test]
1672    fn streaming_applies_recovered_links_in_strict_mode() {
1673        let mut doc = DoclingDocument::new("d");
1674        doc.add_paragraph("See LinkedIn for details.");
1675        doc.add_paragraph("And GitHub too.");
1676        doc.links = vec![
1677            ("LinkedIn".into(), "https://lnkd/".into()),
1678            ("GitHub".into(), "https://gh/".into()),
1679        ];
1680        // The second anchor lives in the second block, so it must be carried across
1681        // the page boundary and placed when that block streams out.
1682        assert_stream_matches(&doc, true, ImageMode::Placeholder, &[1]);
1683    }
1684
1685    /// A three-page document with one empty page in the middle and page
1686    /// markers of both kinds, as the backends emit them.
1687    fn paged_doc() -> DoclingDocument {
1688        let mut doc = DoclingDocument::new("p");
1689        doc.push(Node::PageInfo {
1690            page_no: 1,
1691            width: 100.0,
1692            height: 100.0,
1693        });
1694        doc.add_heading(1, "Title");
1695        doc.add_paragraph("Page one.");
1696        // Page two: a marker plus furniture only — renders nothing.
1697        doc.push(Node::PageBreak);
1698        doc.push(Node::PageInfo {
1699            page_no: 2,
1700            width: 100.0,
1701            height: 100.0,
1702        });
1703        doc.push(Node::PageFurniture {
1704            footer: true,
1705            location: [0, 500, 511, 511],
1706            text: "2".into(),
1707        });
1708        doc.push(Node::PageBreak);
1709        doc.push(Node::PageInfo {
1710            page_no: 3,
1711            width: 100.0,
1712            height: 100.0,
1713        });
1714        doc.add_paragraph("Page three.");
1715        // A trailing boundary with nothing after it.
1716        doc.push(Node::PageBreak);
1717        doc
1718    }
1719
1720    #[test]
1721    fn page_break_placeholder_lands_between_pages_only() {
1722        let mut doc = paged_doc();
1723        // Off by default: docling's Markdown carries no page breaks.
1724        assert_eq!(
1725            doc.export_to_markdown(),
1726            "# Title\n\nPage one.\n\nPage three.\n"
1727        );
1728        doc.page_break_placeholder = Some("<!-- page break -->".into());
1729        // One break for the 1→3 transition (the empty page 2 and the doubled
1730        // PageBreak+PageInfo markers collapse), none before the first block,
1731        // none for the trailing boundary.
1732        assert_eq!(
1733            doc.export_to_markdown(),
1734            "# Title\n\nPage one.\n\n<!-- page break -->\n\nPage three.\n"
1735        );
1736        // An empty placeholder is still a (blank) part, as upstream's
1737        // `str.replace(marker, "")` leaves the delimiters around it.
1738        doc.page_break_placeholder = Some(String::new());
1739        assert_eq!(
1740            doc.export_to_markdown(),
1741            "# Title\n\nPage one.\n\n\n\nPage three.\n"
1742        );
1743    }
1744
1745    #[test]
1746    fn page_break_placeholder_never_leads_a_single_page() {
1747        let mut doc = DoclingDocument::new("one");
1748        doc.page_break_placeholder = Some("---".into());
1749        doc.push(Node::PageBreak);
1750        doc.push(Node::PageInfo {
1751            page_no: 1,
1752            width: 10.0,
1753            height: 10.0,
1754        });
1755        doc.add_paragraph("Only page.");
1756        assert_eq!(doc.export_to_markdown(), "Only page.\n");
1757        // Two boundaries with no content between them: still one break.
1758        doc.push(Node::PageBreak);
1759        doc.push(Node::PageBreak);
1760        doc.add_paragraph("Next.");
1761        assert_eq!(doc.export_to_markdown(), "Only page.\n\n---\n\nNext.\n");
1762    }
1763
1764    #[test]
1765    fn page_break_placeholder_streams_byte_identical() {
1766        let mut doc = paged_doc();
1767        doc.page_break_placeholder = Some("<!-- page break -->".into());
1768        // Split at every page marker (how the PDF pipeline pushes page batches)
1769        // and at odd places inside a page: the pending break must survive a
1770        // push that renders nothing (page two) and land on page three's block.
1771        for splits in [
1772            &[3usize][..],
1773            &[3, 6],
1774            &[3, 6, 8],
1775            &[1, 2, 3, 4, 5, 6, 7, 8, 9],
1776            &[8],
1777        ] {
1778            assert_stream_matches(&doc, false, ImageMode::Placeholder, splits);
1779            assert_stream_matches(&doc, true, ImageMode::Placeholder, splits);
1780        }
1781    }
1782
1783    #[test]
1784    fn strict_tightens_punctuation_spacing_legacy_keeps_it() {
1785        let mut doc = DoclingDocument::new("t");
1786        doc.add_paragraph("see [ 37 , 36 ] and ( x ) .");
1787        // Legacy keeps docling's spacing byte-for-byte.
1788        assert_eq!(doc.export_to_markdown(), "see [ 37 , 36 ] and ( x ) .\n");
1789        // Strict tightens punctuation for readable Markdown.
1790        assert_eq!(doc.export_to_markdown_with(true), "see [37, 36] and (x).\n");
1791    }
1792}