Skip to main content

docling/
converter.rs

1//! The top-level `DocumentConverter`.
2
3use std::collections::HashSet;
4use std::path::PathBuf;
5
6use crate::backend::{
7    is_deepseek_markdown, AbwBackend, AsciiDocBackend, CsvBackend, DeclarativeBackend,
8    DeepSeekBackend, DocBackend, DoclingJsonBackend, DocxBackend, EbcdicBackend, EmailBackend,
9    EpubBackend, InterchangeBackend, JatsBackend, LatexBackend, LotusBackend, MarkdownBackend,
10    MhtmlBackend, PptBackend, PptxBackend, QuattroBackend, RtfBackend, StarOffice5Backend,
11    UsptoBackend, VisioBackend, WebVttBackend, WpdBackend, WpsBackend, XlsBackend, XlsxBackend,
12};
13
14/// Whether `text` begins with an XML prolog — an `<?xml …?>` declaration or a
15/// non-HTML `<!DOCTYPE …>`. Used to route XML documents that arrived with a
16/// text/Markdown extension (e.g. a JATS article saved as `.txt`) to the XML
17/// backends. An HTML5 `<!DOCTYPE html>` is deliberately excluded.
18fn looks_like_xml(text: &str) -> bool {
19    let head = text.trim_start();
20    if head.starts_with("<?xml") {
21        return true;
22    }
23    if let Some(rest) = head.get(..9) {
24        if rest.eq_ignore_ascii_case("<!doctype") {
25            return !head[9..]
26                .trim_start()
27                .to_ascii_lowercase()
28                .starts_with("html");
29        }
30    }
31    false
32}
33
34/// Pick the concrete XML backend for a generic `.xml` source by sniffing its
35/// DOCTYPE / root element (the first part of the file).
36/// docling's `_guess_from_content` JATS rule for an `application/xml` input:
37/// the `<!DOCTYPE …>` declaration names a JATS DTD.
38fn has_jats_doctype(text: &str) -> bool {
39    let Some(start) = text.find("<!DOCTYPE ") else {
40        return false;
41    };
42    let Some(len) = text[start..].find('>') else {
43        return false;
44    };
45    let doctype = &text[start..start + len];
46    doctype.contains("JATS-journalpublishing") || doctype.contains("JATS-archive")
47}
48
49fn sniff_xml(bytes: &[u8]) -> InputFormat {
50    // Lossy over the raw head (docling#4038, 2.122): the window can cut a
51    // well-formed file mid-codepoint — a fixed-offset `&str` slice would
52    // panic on that char boundary — and an XML document may legitimately
53    // declare a non-UTF-8 encoding. Every marker matched below is ASCII, so
54    // replacement characters cannot change the outcome.
55    let head = String::from_utf8_lossy(&bytes[..bytes.len().min(4000)]);
56    let head = head.as_ref();
57    // Case-insensitive: USPTO DOCTYPE/root casing varies in the wild (docling
58    // PR #3801 — Grant Full Text v2.5 files were missed on casing).
59    let lower = head.to_ascii_lowercase();
60    if lower.contains("us-patent")
61        || lower.contains("patent-application-publication")
62        || lower.contains("patdoc")
63        || lower.contains("<pap-v1")
64    {
65        InputFormat::XmlUspto
66    } else if head.contains("<doclang") {
67        // A bare DocLang document saved as `.xml` (docling names them
68        // `*.dclg.xml`, whose final extension is plain `xml`).
69        InputFormat::XmlDoclang
70    } else if crate::backend::xbrl::looks_like_xbrl(head) {
71        InputFormat::XmlXbrl
72    } else {
73        InputFormat::XmlJats
74    }
75}
76use crate::error::ConversionError;
77use crate::format::InputFormat;
78use crate::result::{ConversionResult, ConversionStatus};
79use crate::source::SourceDocument;
80#[cfg(feature = "pdf")]
81use crate::stream::MarkdownStream;
82#[cfg(feature = "pdf")]
83use docling_core::ImageMode;
84
85/// Routes a [`SourceDocument`] to the backend for its format and returns a
86/// [`ConversionResult`].
87///
88/// The Rust analogue of `docling.document_converter.DocumentConverter`. In
89/// Phase 0 the format→backend dispatch is a direct match; the Python notion of
90/// per-format `FormatOption` (backend + pipeline + options) arrives with the
91/// PDF/ML pipeline in a later phase.
92#[derive(Debug, Clone)]
93pub struct DocumentConverter {
94    allowed_formats: Option<HashSet<InputFormat>>,
95    strict: bool,
96    fetch_images: bool,
97    list_attachments: bool,
98    /// Omit empty cells from sparse spreadsheet table grids (#271, XLSX/XLS
99    /// family; opt-in docling.rs extension).
100    skip_empty_cells: bool,
101    /// Emit Markdown tables in the compact `| a | b |` form instead of the
102    /// width-padded GitHub serializer (#271, opt-in docling.rs extension).
103    compact_tables: bool,
104    /// docling-core's `MarkdownParams.page_break_placeholder`: text inserted
105    /// between pages in the Markdown export; `None` (default) omits breaks.
106    page_break_placeholder: Option<String>,
107    /// EBCDIC copybook layout (#252): inline JSON or a file path. `None`
108    /// falls back to the `<stem>.layout.json` sidecar.
109    ebcdic_layout: Option<String>,
110    no_table_former: bool,
111    no_text_panels: bool,
112    no_ocr: bool,
113    skip_ocr: bool,
114    force_full_page_ocr: bool,
115    /// OCR mode id (docling's `OcrMode`, #254); parsed at the ML call sites.
116    ocr_mode: Option<String>,
117    /// OCR engine id (`ppocr` | `tesseract`, #460); parsed at the ML call
118    /// sites.
119    ocr_engine: Option<String>,
120    /// OCR render scale in px/pt (#254); validated at the ML call sites.
121    ocr_scale: Option<f32>,
122    /// Infer PDF/image section-header levels after assembly (#302, docling's
123    /// `HeadingHierarchyModel`): bookmarks > numbering > font style. Off by
124    /// default — heading levels then stay exactly as detected.
125    heading_hierarchy: bool,
126    use_web_browser: bool,
127    /// Named Whisper model preset for audio sources (docling's ASR model
128    /// specs, PR #3741): English-only / Distil-Whisper variants under
129    /// `.models/asr/<preset>/`. `None` = the default Whisper tiny.
130    asr_model: Option<String>,
131    asr_lang: Option<String>,
132    /// Max sampled frames per video (#138 Phase 2). `None` = the default
133    /// ([`DEFAULT_VIDEO_FRAMES`]); `Some(0)` disables frame extraction.
134    video_frames: Option<usize>,
135    /// The directory an XBRL instance's taxonomy is read from (docling's
136    /// `XBRLBackendOptions.taxonomy`); `None` = the instance's own directory.
137    xbrl_taxonomy: Option<PathBuf>,
138    /// Opt-in PDF/image enrichment models (docling's
139    /// `do_picture_classification` / `do_code_enrichment` /
140    /// `do_formula_enrichment`).
141    enrich: crate::EnrichmentOptions,
142    /// 1-based inclusive PDF page window (#80). See [`Self::page_range`].
143    page_range: Option<(usize, usize)>,
144    /// OCR recognition language for scanned PDF/image pages (`en`/`ch`).
145    /// `None` = the process default (`DOCLING_RS_OCR_LANG`, else English).
146    ocr_lang: Option<String>,
147    /// Character encoding for text inputs (docling's
148    /// `TextBackendOptions.encoding`); `None` = detect. See [`Self::encoding`].
149    encoding: Option<String>,
150    /// Directory referenced-mode streaming writes images into (#80).
151    /// See [`Self::artifacts_dir`].
152    artifacts_dir: String,
153}
154
155/// Default cap on sampled frames per video. Scene changes rarely exceed this
156/// in short clips, and uniform fallback at 8 keeps JSON/DCLX output (which
157/// embeds the PNGs) within sane bounds.
158pub const DEFAULT_VIDEO_FRAMES: usize = 8;
159
160/// Parse a user-facing page-range string (issue #80's `--pages`): `"A-B"` for
161/// an inclusive 1-based window, or a single `"N"` for one page. Whitespace
162/// around the numbers is tolerated. Validation against the actual page count
163/// happens at convert time; this only checks the spelling (`first >= 1`,
164/// `first <= last`).
165pub fn parse_page_range(s: &str) -> Result<(usize, usize), String> {
166    let parse_one = |part: &str| {
167        part.trim()
168            .parse::<usize>()
169            .map_err(|_| format!("invalid page number '{}'", part.trim()))
170    };
171    let (first, last) = match s.split_once('-') {
172        Some((a, b)) => (parse_one(a)?, parse_one(b)?),
173        None => {
174            let n = parse_one(s)?;
175            (n, n)
176        }
177    };
178    if first == 0 {
179        return Err("pages are 1-based; the range starts at 1".into());
180    }
181    if last < first {
182        return Err(format!("range {first}-{last} is inverted (first <= last)"));
183    }
184    Ok((first, last))
185}
186
187impl Default for DocumentConverter {
188    fn default() -> Self {
189        Self {
190            allowed_formats: None,
191            strict: false,
192            fetch_images: false,
193            list_attachments: false,
194            skip_empty_cells: false,
195            compact_tables: false,
196            page_break_placeholder: None,
197            ebcdic_layout: None,
198            no_table_former: false,
199            no_text_panels: false,
200            no_ocr: false,
201            skip_ocr: false,
202            force_full_page_ocr: false,
203            ocr_mode: None,
204            ocr_engine: None,
205            ocr_scale: None,
206            heading_hierarchy: false,
207            use_web_browser: false,
208            asr_model: None,
209            asr_lang: None,
210            video_frames: None,
211            xbrl_taxonomy: None,
212            enrich: crate::EnrichmentOptions::default(),
213            page_range: None,
214            ocr_lang: None,
215            encoding: None,
216            artifacts_dir: "artifacts".to_string(),
217        }
218    }
219}
220
221impl DocumentConverter {
222    /// A converter that accepts every supported format.
223    pub fn new() -> Self {
224        Self::default()
225    }
226
227    /// A converter restricted to an explicit set of formats. Sources of any
228    /// other format are rejected with [`ConversionError::UnsupportedFormat`].
229    pub fn with_allowed_formats(formats: impl IntoIterator<Item = InputFormat>) -> Self {
230        Self {
231            allowed_formats: Some(formats.into_iter().collect()),
232            ..Self::default()
233        }
234    }
235
236    /// Convert only PDF pages `first..=last` (**1-based** inclusive, the page
237    /// numbers a viewer shows — issue #80's `--pages A-B`). Out-of-window pages
238    /// are skipped before rasterization, so converting 3 pages of a 500-page
239    /// PDF costs 3 pages. `last` clamps to the document; a window that selects
240    /// no pages at all errors at convert time. Non-PDF formats ignore the
241    /// window (they convert whole).
242    pub fn page_range(mut self, first: usize, last: usize) -> Self {
243        self.page_range = Some((first, last));
244        self
245    }
246
247    /// OCR recognition language for scanned PDF/image pages: `"en"` (the
248    /// default — English PP-OCRv3, proper Latin word spacing) or `"ch"` (the
249    /// multilingual model docling conformance is measured with — glues Latin
250    /// words). An unknown value warns at conversion time and uses the
251    /// default; explicit `DOCLING_OCR_REC_ONNX`/`DOCLING_OCR_DICT` paths win
252    /// over this switch. Under [`ocr_engine`](Self::ocr_engine) `tesseract`
253    /// it is Tesseract's language list instead — tessdata stems (`"deu"`,
254    /// `"eng+fra"`, `"script/Cyrillic"`) or BCP-47 tags mapped onto them
255    /// (`"de"`, `"zh-Hant"`), see [`docling_pdf::tesseract_lang_arg`]; `en`
256    /// and `ch` keep working there. Formats that never OCR ignore it.
257    pub fn ocr_lang(mut self, lang: impl Into<String>) -> Self {
258        self.ocr_lang = Some(lang.into());
259        self
260    }
261
262    /// The configured ML pipeline for one conversion (models load per call —
263    /// callers that convert many files hold a warm [`docling_pdf::Pipeline`]
264    /// themselves). Grew out of docling-pdf's `convert_with_options` free
265    /// functions, whose fixed signatures couldn't take #244's `skip_ocr`.
266    #[cfg(feature = "pdf")]
267    fn ml_pipeline(&self) -> Result<docling_pdf::Pipeline, docling_pdf::PdfError> {
268        Ok(docling_pdf::Pipeline::new()?
269            .no_table_former(self.no_table_former)
270            .no_ocr(self.no_ocr)
271            .skip_ocr(self.skip_ocr)
272            .no_text_panels(self.no_text_panels)
273            .enrichments(self.enrich)
274            .ocr_lang(self.ocr_lang_choice())
275            .ocr_engine(self.ocr_engine_choice())
276            .tesseract_lang(self.tesseract_lang_choice())
277            .ocr_mode(self.ocr_mode_choice())
278            .ocr_scale(self.ocr_scale_choice())
279            .heading_hierarchy(docling_pdf::HeadingHierarchyOptions::enabled(
280                self.heading_hierarchy,
281            )))
282    }
283
284    /// DjVu (#434): the hidden text layer by default; a scan-only DjVu (no
285    /// text layer on any selected page) falls back to rasterize + OCR when the
286    /// ML pipeline is built and OCR is not disabled, otherwise it degrades to
287    /// an empty document with a warning. See [`crate::backend::djvu`].
288    fn convert_djvu(
289        &self,
290        source: &SourceDocument,
291    ) -> Result<docling_core::DoclingDocument, ConversionError> {
292        use crate::backend::djvu;
293        let text = djvu::convert_text_layer(source, self.page_range)?;
294        if !djvu::is_text_layer_empty(&text) {
295            return Ok(text);
296        }
297        // No text layer anywhere in the selection.
298        #[cfg(feature = "pdf")]
299        if !self.no_ocr && !self.skip_ocr {
300            let pngs = djvu::rasterize_pages(&source.bytes, self.page_range)?;
301            let mut pipeline = self
302                .ml_pipeline()
303                .map_err(|e| ConversionError::with_source("djvu", e))?;
304            let mut doc = docling_core::DoclingDocument::new(&source.name);
305            for (i, (page_no, png)) in pngs.iter().enumerate() {
306                if i > 0 {
307                    doc.push(docling_core::Node::PageBreak);
308                }
309                let mut page = pipeline
310                    .convert_image(png, &source.name)
311                    .map_err(|e| ConversionError::with_source("djvu", e))?;
312                // An image is its own page 1; the DjVu page keeps its number so
313                // a `--pages` window's JSON `pages` matches the text-layer path.
314                docling_pdf::assemble::stamp_page_no(&mut page.nodes, *page_no);
315                doc.nodes.extend(page.nodes);
316                doc.links.extend(page.links);
317            }
318            return Ok(doc);
319        }
320        eprintln!(
321            "docling: warning: DjVu '{}' has no text layer and OCR is unavailable/disabled; \
322             the document is empty",
323            source.name
324        );
325        Ok(text)
326    }
327
328    /// The parsed [`Self::ocr_lang`] choice for the ML call sites; a value
329    /// that parses to nothing warns here (once per conversion) rather than
330    /// erroring — same degradation the env selector applies.
331    #[cfg(feature = "pdf")]
332    fn ocr_lang_choice(&self) -> Option<docling_pdf::OcrLang> {
333        let raw = self.ocr_lang.as_deref()?;
334        // Under Tesseract the value is its language list, not a PP-OCR model
335        // — see `tesseract_lang_choice`.
336        if self.ocr_engine_choice() == Some(docling_pdf::OcrEngine::Tesseract) {
337            return None;
338        }
339        let parsed = docling_pdf::OcrLang::parse(raw);
340        if parsed.is_none() {
341            eprintln!(
342                "docling: ocr_lang {raw:?} names no language the OCR models read ({}); using \
343                 the default",
344                docling_pdf::OcrLang::ACCEPTED
345            );
346        }
347        parsed
348    }
349
350    /// The parsed [`Self::ocr_engine`] choice (#460), with the same
351    /// warn-and-default degradation as [`ocr_lang_choice`](Self::ocr_lang_choice).
352    #[cfg(feature = "pdf")]
353    fn ocr_engine_choice(&self) -> Option<docling_pdf::OcrEngine> {
354        let raw = self.ocr_engine.as_deref()?;
355        let parsed = docling_pdf::OcrEngine::parse(raw);
356        if parsed.is_none() {
357            eprintln!(
358                "docling: ocr_engine {raw:?} is not {}; using the default",
359                docling_pdf::OcrEngine::ACCEPTED
360            );
361        }
362        parsed
363    }
364
365    /// Tesseract's `-l` argument from [`Self::ocr_lang`] (#460) when the
366    /// engine is Tesseract: an unmappable value warns and leaves Tesseract's
367    /// default language.
368    #[cfg(feature = "pdf")]
369    fn tesseract_lang_choice(&self) -> Option<String> {
370        if self.ocr_engine_choice() != Some(docling_pdf::OcrEngine::Tesseract) {
371            return None;
372        }
373        let raw = self.ocr_lang.as_deref()?;
374        match docling_pdf::tesseract_lang_arg(raw) {
375            Ok(arg) => Some(arg),
376            Err(e) => {
377                eprintln!("docling: {e}; using Tesseract's default language");
378                None
379            }
380        }
381    }
382
383    /// The parsed [`Self::ocr_mode`] choice (#254), with the same
384    /// warn-and-default degradation as [`ocr_lang_choice`](Self::ocr_lang_choice).
385    #[cfg(feature = "pdf")]
386    fn ocr_mode_choice(&self) -> Option<docling_pdf::OcrMode> {
387        let raw = self.ocr_mode.as_deref()?;
388        let parsed = docling_pdf::OcrMode::parse(raw);
389        if parsed.is_none() {
390            eprintln!(
391                "docling: ocr_mode {raw:?} is not \
392                 default|full_page|layout_regions|pdf_aware_layout_regions; using the default"
393            );
394        }
395        parsed
396    }
397
398    /// The validated [`Self::ocr_scale`] (#254): non-positive/non-finite
399    /// values warn and fall back to the engine default.
400    #[cfg(feature = "pdf")]
401    fn ocr_scale_choice(&self) -> Option<f32> {
402        let s = self.ocr_scale?;
403        if !(s.is_finite() && s > 0.0) {
404            eprintln!("docling: ocr_scale {s} is not a positive number; using the default");
405            return None;
406        }
407        Some(s)
408    }
409
410    /// Where [`ImageMode::Referenced`] streaming writes image files, and the
411    /// link prefix used in the Markdown (default `artifacts`, matching the
412    /// buffered export's convention). Relative paths resolve against the
413    /// process working directory.
414    pub fn artifacts_dir(mut self, dir: impl Into<String>) -> Self {
415        self.artifacts_dir = dir.into();
416        self
417    }
418
419    /// Decode text inputs (Markdown, CSV, AsciiDoc, WebVTT, LaTeX, the XML
420    /// dialects, …) with this character encoding instead of detecting one —
421    /// docling's `TextBackendOptions.encoding`
422    /// (`MarkdownBackendOptions(encoding="shift_jis")`). A WHATWG encoding
423    /// label (`shift_jis`, `koi8-r`, `windows-1251`, `latin1`; Python codec
424    /// spellings with `_` are accepted). Nothing is guessed: bytes the
425    /// encoding cannot decode fail the conversion, as does an unknown label.
426    /// `None` (default) detects — a byte-order mark, then UTF-8, then
427    /// windows-1252 (see [`SourceDocument::text`]). A source that already
428    /// carries its own [`SourceDocument::encoding`] keeps it.
429    pub fn encoding(mut self, label: Option<String>) -> Self {
430        self.encoding = label;
431        self
432    }
433
434    /// The converter's [`encoding`](Self::encoding) applied to a source that
435    /// did not set its own.
436    fn with_encoding(&self, mut source: SourceDocument) -> SourceDocument {
437        if source.encoding.is_none() {
438            source.encoding = self.encoding.clone();
439        }
440        source
441    }
442
443    /// Cap the number of frames sampled from a video (#138 Phase 2); `0`
444    /// disables frame extraction entirely (Phase 1 behavior: transcript only).
445    /// Defaults to [`DEFAULT_VIDEO_FRAMES`]. Frames are extracted with the
446    /// `ffmpeg` binary when present (`DOCLING_FFMPEG` overrides the path);
447    /// without it a video converts to its transcript alone.
448    pub fn video_frames(mut self, max: usize) -> Self {
449        self.video_frames = Some(max);
450        self
451    }
452
453    /// The folder holding the taxonomy an XBRL instance refers to (docling's
454    /// `XBRLBackendOptions.taxonomy`): the filing's extension schema and
455    /// linkbases at the relative paths its `link:schemaRef` names, plus any
456    /// taxonomy packages (`.zip` with a `META-INF/catalog.xml`) that map the
457    /// base taxonomies' `http(s)` URLs to files for offline use. Without it the
458    /// instance's own directory is searched. Nothing is fetched remotely; a
459    /// document that cannot be found only costs the fact graph the hierarchy
460    /// it would have contributed.
461    pub fn xbrl_taxonomy(mut self, dir: impl Into<PathBuf>) -> Self {
462        self.xbrl_taxonomy = Some(dir.into());
463        self
464    }
465
466    /// Select a named Whisper model preset for audio sources — the
467    /// English-only (`whisper_tiny_en`, `whisper_base_en`, `whisper_small_en`)
468    /// and Distil-Whisper (`whisper_distil_small_en`) variants of docling's
469    /// ASR model specs. `None` (default) uses Whisper tiny (multilingual)
470    /// from `.models/asr/`; presets load from `.models/asr/<preset>/` (fetch
471    /// them with `download_dependencies.sh --asr-model <preset>`).
472    pub fn asr_model(mut self, model: Option<String>) -> Self {
473        self.asr_model = model;
474        self
475    }
476
477    /// Select the ASR transcription language for audio/video sources: a
478    /// Whisper code (`en`, `de`, `zh`, …) or `auto`. `None` (default) falls
479    /// back to `DOCLING_RS_ASR_LANG`, and — when that is unset too — to
480    /// per-file auto-detection from the first 30-second window (docling
481    /// 2.116 parity). English-only presets always transcribe English.
482    pub fn asr_lang(mut self, lang: Option<String>) -> Self {
483        self.asr_lang = lang;
484        self
485    }
486
487    /// Select the Markdown export mode for documents this converter produces.
488    ///
489    /// `false` (default) makes [`crate::DoclingDocument::export_to_markdown`]
490    /// reproduce docling's legacy output byte-for-byte; `true` makes it emit
491    /// cleaner, more conformant Markdown (code-fence languages preserved, no
492    /// inline-run spacing artifacts, no entity re-escaping). Rust-only — Python
493    /// docling has no such switch.
494    pub fn strict(mut self, strict: bool) -> Self {
495        self.strict = strict;
496        self
497    }
498
499    /// Fetch and embed external `<img>` images for HTML/EPUB/MHTML/JATS sources.
500    ///
501    /// Off by default (matching docling's `enable_*_fetch=False`), so output is
502    /// unchanged unless you opt in. When on, the HTML/EPUB/MHTML backends
503    /// resolve each `<img src>` — `data:` URIs, local files (relative to the
504    /// source file's directory), `http(s)` URLs, and EPUB/MHTML archive
505    /// entries — and the JATS backend reads a `<fig>`'s `<graphic xlink:href>`
506    /// from the source file's directory (#392), embedding the bytes so they
507    /// survive into JSON `ImageRef`s and
508    /// [`crate::DoclingDocument::export_to_markdown_with_images`].
509    ///
510    /// Remote `http(s)` URLs are fetched over the network; enable only for input
511    /// you trust (it can otherwise be used to make the process issue requests).
512    pub fn fetch_images(mut self, fetch: bool) -> Self {
513        self.fetch_images = fetch;
514        self
515    }
516
517    /// Append an `Attachments` section to converted emails (`.eml` / `.msg`):
518    /// one list item per attachment, `name (content/type)` — names and types
519    /// only, the payload is never embedded. docling's opt-in
520    /// `EmailBackendOptions.list_attachments` (#251); off by default.
521    pub fn list_attachments(mut self, list: bool) -> Self {
522        self.list_attachments = list;
523        self
524    }
525
526    /// Omit empty cells from sparse spreadsheet table grids (#271; XLSX/XLS
527    /// family, opt-in — a docling.rs extension, docling materialises the full
528    /// bounding box). A ragged region's box is mostly padding on sparse
529    /// sheets (~7× output inflation); with this on, each row keeps only its
530    /// occupied cells (merge-covered continuations included) and a table
531    /// that loses cells drops its span/structure overlay. Off by default —
532    /// default output stays byte-for-byte docling.
533    pub fn skip_empty_cells(mut self, skip: bool) -> Self {
534        self.skip_empty_cells = skip;
535        self
536    }
537
538    /// Emit Markdown tables in the compact `| a | b |` / `| - | - |` form
539    /// instead of docling-core's width-padded GitHub serializer (#271, all
540    /// formats; opt-in — a docling.rs extension). On sparse spreadsheets the
541    /// padding dominates the output size; compact rendering keeps the grid
542    /// semantics and drops only whitespace. Off by default — default output
543    /// stays byte-for-byte docling.
544    pub fn compact_tables(mut self, compact: bool) -> Self {
545        self.compact_tables = compact;
546        self
547    }
548
549    /// Insert `placeholder` between pages in the Markdown export — docling's
550    /// `export_to_markdown(page_break_placeholder=…)` (docling-core's
551    /// `MarkdownParams`, docling-serve's `md_page_break_placeholder`). Where
552    /// docling marks a break between two items whose `prov.page_no` differ,
553    /// the serializer here marks it between two rendered blocks separated by a
554    /// page boundary — the `PageInfo` marker opening every PDF page and sheet,
555    /// the `PageBreak` between slides and DjVu / DocTags pages — so a break
556    /// never leads or trails the document and empty pages collapse into one.
557    /// `None` (the default) keeps Markdown free of page breaks, as docling's
558    /// default export is. Markdown only; JSON carries pages in `prov`.
559    pub fn page_break_placeholder(mut self, placeholder: Option<String>) -> Self {
560        self.page_break_placeholder = placeholder;
561        self
562    }
563
564    /// The copybook layout for EBCDIC sources (#252): docling's
565    /// `EbcdicLayout` JSON, inline (a string starting with `{`) or as a file
566    /// path. Without it, a path-loaded source looks for a
567    /// `<stem>.layout.json` sidecar; converting EBCDIC with neither is an
568    /// error — the bytes are meaningless without their copybook.
569    pub fn ebcdic_layout(mut self, layout: impl Into<String>) -> Self {
570        self.ebcdic_layout = Some(layout.into());
571        self
572    }
573
574    /// Option-typed variant of [`ebcdic_layout`](Self::ebcdic_layout) for
575    /// call sites plumbing an optional flag through (`None` keeps the
576    /// sidecar fallback).
577    pub fn ebcdic_layout_opt(mut self, layout: Option<String>) -> Self {
578        self.ebcdic_layout = layout;
579        self
580    }
581
582    /// Skip loading and running the TableFormer table-structure model for
583    /// PDF/image/METS sources.
584    ///
585    /// Off by default. When enabled, table regions are still detected and
586    /// emitted, but their structure is reconstructed geometrically from cell
587    /// positions instead of the ONNX model's predicted structure — no model
588    /// load and no per-table inference, at the cost of table fidelity. Useful
589    /// when parsing speed matters more than exact table structure, especially
590    /// with [`convert_streaming`](Self::convert_streaming).
591    pub fn no_table_former(mut self, disable: bool) -> Self {
592        self.no_table_former = disable;
593        self
594    }
595
596    /// PDF/image: keep every detected picture as a picture — disable the
597    /// text-panel demotion that turns an uncaptioned, dense text-panel
598    /// "picture" into paragraphs (#157). The escape hatch for
599    /// image-extraction workflows and for charts the heuristic might still
600    /// misjudge on scanned pages (#173).
601    pub fn no_text_panels(mut self, disable: bool) -> Self {
602        self.no_text_panels = disable;
603        self
604    }
605
606    /// Infer PDF/image section-header levels after assembly (#302, docling's
607    /// `HeadingHierarchyModel` with its default options): the PDF outline
608    /// (bookmarks) is authoritative, legal/outline numbering covers headings
609    /// without a bookmark match, and font size/weight/slant/case rank the
610    /// rest. Off by default (docling parity): every detected heading then
611    /// keeps the flat level the assembler emits. Non-PDF/image formats
612    /// ignore it (their backends carry real heading levels already).
613    pub fn heading_hierarchy(mut self, enable: bool) -> Self {
614        self.heading_hierarchy = enable;
615        self
616    }
617
618    /// Skip layout detection, OCR, and TableFormer entirely for PDF/image/METS
619    /// sources — no model load, no inference of any kind.
620    ///
621    /// Off by default. When enabled, the PDF's embedded text cells are grouped by
622    /// line and emitted as plain paragraphs in reading order: no headings, lists,
623    /// tables, code blocks, or pictures, since that structure comes from the
624    /// layout model. The fastest possible PDF path, but pages with no embedded
625    /// text layer (scanned/image-only PDFs) yield no text at all — convert those
626    /// without this flag. Implies [`no_table_former`](Self::no_table_former).
627    pub fn no_ocr(mut self, disable: bool) -> Self {
628        self.no_ocr = disable;
629        self
630    }
631
632    /// Never run OCR, but keep layout detection and TableFormer — docling's
633    /// independent `do_ocr=False` (#244), the counterpart of
634    /// [`no_table_former`](Self::no_table_former). Unlike
635    /// [`no_ocr`](Self::no_ocr) (the skip-everything fast path), structured
636    /// output — headings, tables, pictures, reading order — is preserved; only
637    /// text that exists solely as pixels is lost (scanned pages come back with
638    /// empty regions, and the speculative OCR of large embedded images never
639    /// runs). The OCR model is never loaded, and independently of this flag a
640    /// *missing* OCR model now degrades to the same behavior with a warning
641    /// instead of failing the conversion. SVG inputs route to direct
642    /// `<text>` extraction (their text is native — skipping OCR must not lose
643    /// it), like `no_ocr`.
644    pub fn skip_ocr(mut self, disable: bool) -> Self {
645        self.skip_ocr = disable;
646        self
647    }
648
649    /// OCR every PDF page from its rendered image even when the page carries
650    /// an embedded text layer — docling's `force_full_page_ocr`. The escape
651    /// hatch for text layers that exist but lie (broken encodings, subset
652    /// fonts with garbage mappings, scanned forms with a few typed-in
653    /// fields). Off by default; ignored when [`no_ocr`](Self::no_ocr) is set,
654    /// mirroring docling, where it is a sub-option of `do_ocr`. Applies to
655    /// PDFs only — standalone images are always OCR'd.
656    pub fn force_full_page_ocr(mut self, force: bool) -> Self {
657        self.force_full_page_ocr = force;
658        self
659    }
660
661    /// Which document regions feed the OCR — docling's `OcrMode` (#254):
662    /// `default`, `full_page`, `layout_regions`, or
663    /// `pdf_aware_layout_regions`. The default is the text-layer-aware
664    /// behavior (docling's `pdf_aware_layout_regions`);
665    /// `full_page`/`layout_regions` discard the text layer like
666    /// [`force_full_page_ocr`](Self::force_full_page_ocr) — see
667    /// [`docling_pdf::OcrMode`] for the mapping. An unknown value warns at
668    /// conversion time and uses the default. PDF/image ML pipeline only.
669    pub fn ocr_mode(mut self, mode: impl Into<String>) -> Self {
670        self.ocr_mode = Some(mode.into());
671        self
672    }
673
674    /// Which OCR engine recognizes text on scanned pages (#460): `"ppocr"`
675    /// (the default — the built-in PP-OCRv3 recognizer, the conformance
676    /// engine) or `"tesseract"` (the system `tesseract` binary, docling's
677    /// `TesseractCliOcrOptions`; needs `tesseract-ocr` with a language pack
678    /// installed — `DOCLING_TESSERACT` names the binary,
679    /// `DOCLING_RS_TESSERACT_PSM` its page segmentation mode,
680    /// `DOCLING_RS_TESSDATA_DIR` / `TESSDATA_PREFIX` its data). Both read
681    /// the same layout-region crops; [`ocr_lang`](Self::ocr_lang) is the
682    /// engine's language. An unknown value warns at conversion time and uses
683    /// the default (`DOCLING_RS_OCR_ENGINE`, else PP-OCR). Formats that never
684    /// OCR ignore it.
685    pub fn ocr_engine(mut self, engine: impl Into<String>) -> Self {
686        self.ocr_engine = Some(engine.into());
687        self
688    }
689
690    /// OCR render scale in pixels per PDF point — docling's
691    /// `OcrOptions.scale` (#254; docling's default 3 = 216 dpi). Unset feeds
692    /// the recognizer the pipeline's own 2.0 px/pt page render; a different
693    /// value resamples that render for the OCR input only, leaving layout and
694    /// TableFormer pixels untouched. Non-positive values warn at conversion
695    /// time and are ignored. PDF/image ML pipeline only.
696    pub fn ocr_scale(mut self, scale: f32) -> Self {
697        self.ocr_scale = Some(scale);
698        self
699    }
700
701    /// Classify each detected picture with the DocumentFigureClassifier model
702    /// (docling's `do_picture_classification`). Off by default.
703    ///
704    /// The full 26-class prediction distribution (bar_chart, logo, signature,
705    /// …) lands on the picture item and is serialized into the docling JSON as
706    /// the `classification` annotation plus the `meta.classification` field.
707    /// Markdown output is unaffected. Needs `.models/picture_classifier.onnx`
708    /// (fetched by `scripts/install/download_dependencies.sh`); a missing
709    /// model warns once and skips classification.
710    pub fn do_picture_classification(mut self, enable: bool) -> Self {
711        self.enrich.picture_classification = enable;
712        self
713    }
714
715    /// Rewrite detected code blocks with the CodeFormulaV2 VLM (docling's
716    /// `do_code_enrichment`). Off by default.
717    ///
718    /// The model re-reads the code crop at ~120 dpi, emits the clean source
719    /// text (line breaks included) and identifies the language, which lands in
720    /// the JSON `code_language` field. Needs the `.models/code_formula/` graphs
721    /// (fetched by `scripts/install/download_dependencies.sh`); a missing
722    /// model warns once and leaves the block as extracted.
723    pub fn do_code_enrichment(mut self, enable: bool) -> Self {
724        self.enrich.code = enable;
725        self
726    }
727
728    /// Decode display formulas to LaTeX with the CodeFormulaV2 VLM (docling's
729    /// `do_formula_enrichment`). Off by default.
730    ///
731    /// An enriched formula renders as `$$latex$$` in Markdown and as a
732    /// `formula` text item in the JSON, replacing the
733    /// `<!-- formula-not-decoded -->` placeholder. Same model artifacts as
734    /// [`do_code_enrichment`](Self::do_code_enrichment).
735    pub fn do_formula_enrichment(mut self, enable: bool) -> Self {
736        self.enrich.formula = enable;
737        self
738    }
739
740    /// Pre-render HTML-routing input in a headless browser before parsing.
741    ///
742    /// Off by default. When enabled, HTML sources — and MHTML/EPUB, which
743    /// assemble HTML from their archives — are loaded in the system Chromium
744    /// (driven from Rust over the DevTools protocol — no Node/Playwright) so the
745    /// CSS cascade is resolved: elements the browser computes as `display:none`
746    /// (e.g. a stylesheet-collapsed nav menu) are removed before the normal HTML
747    /// backend runs. This is the one behaviour a pure-Rust parse can't reproduce;
748    /// everything else (structure, tables, KVP, formatting) is still handled in
749    /// Rust on the cleaned HTML.
750    ///
751    /// Requires the crate's `web-browser` Cargo feature; without it, converting
752    /// an HTML source with this enabled returns [`ConversionError::Browser`].
753    pub fn use_web_browser(mut self, enable: bool) -> Self {
754        self.use_web_browser = enable;
755        self
756    }
757
758    /// Return `html` unchanged, or — when [`use_web_browser`](Self::use_web_browser)
759    /// is on — its headless-browser-cleaned form (computed-hidden elements
760    /// removed). Borrows in the common (disabled) case; only allocates when the
761    /// browser actually runs.
762    fn maybe_prerender<'a>(
763        &self,
764        html: &'a str,
765    ) -> Result<std::borrow::Cow<'a, str>, ConversionError> {
766        crate::backend::maybe_prerender_html(html, self.use_web_browser)
767    }
768
769    /// Convert a source document to Markdown **incrementally**, returning an
770    /// iterator of Markdown chunks (with picture placeholders).
771    ///
772    /// Concatenating every `Ok` chunk reproduces
773    /// [`convert`](Self::convert)`(...).document.export_to_markdown()`
774    /// byte-for-byte. The win is for PDF, whose pages are processed in parallel:
775    /// each page's Markdown is emitted in document order as soon as it is ready, so
776    /// output starts before the whole document is converted. Other formats build
777    /// their document up front and stream it through the same interface.
778    ///
779    /// Streaming is Markdown-only — JSON needs the whole node tree, so there is no
780    /// streaming JSON. The conversion runs on a background thread; dropping the
781    /// returned [`MarkdownStream`] cancels it.
782    #[cfg(feature = "pdf")]
783    pub fn convert_streaming(
784        &self,
785        source: SourceDocument,
786    ) -> Result<MarkdownStream, ConversionError> {
787        self.convert_streaming_images(source, ImageMode::Placeholder)
788    }
789
790    /// Like [`convert_streaming`](Self::convert_streaming) but with an explicit
791    /// picture [`ImageMode`].
792    ///
793    /// [`ImageMode::Referenced`] streams too (issue #80): each page's images
794    /// are written to [`artifacts_dir`](Self::artifacts_dir) *as the page's
795    /// Markdown is emitted* and dropped from memory, so an image-heavy PDF
796    /// holds ~one page of images at a time instead of all of them until
797    /// export. The chunks and files match the buffered
798    /// `export_to_markdown_with_images(ImageMode::Referenced, ..)` output.
799    #[cfg(feature = "pdf")]
800    pub fn convert_streaming_images(
801        &self,
802        source: SourceDocument,
803        image_mode: ImageMode,
804    ) -> Result<MarkdownStream, ConversionError> {
805        if let Some(allowed) = &self.allowed_formats {
806            if !allowed.contains(&source.format) {
807                return Err(ConversionError::UnsupportedFormat(source.format));
808            }
809        }
810        let source = self.with_encoding(source);
811        Ok(crate::stream::spawn(self.clone(), source, image_mode))
812    }
813
814    /// Whether the heading-hierarchy stage (#302) is enabled — the streaming
815    /// front-end buffers PDF conversions when it is (the stage needs the whole
816    /// assembled document, and streamed output must stay byte-identical to
817    /// buffered output).
818    #[cfg(feature = "pdf")]
819    pub(crate) fn heading_hierarchy_enabled(&self) -> bool {
820        self.heading_hierarchy
821    }
822
823    /// Streaming internals ([`crate::stream`]) read the producer's settings
824    /// off the converter clone they receive.
825    #[cfg(feature = "pdf")]
826    pub(crate) fn stream_settings(&self) -> crate::stream::StreamSettings {
827        crate::stream::StreamSettings {
828            strict: self.strict,
829            no_table_former: self.no_table_former,
830            no_text_panels: self.no_text_panels,
831            no_ocr: self.no_ocr,
832            skip_ocr: self.skip_ocr,
833            force_full_page_ocr: self.force_full_page_ocr,
834            enrich: self.enrich,
835            page_range: self.page_range,
836            ocr_lang: self.ocr_lang_choice(),
837            ocr_engine: self.ocr_engine_choice(),
838            tesseract_lang: self.tesseract_lang_choice(),
839            ocr_mode: self.ocr_mode_choice(),
840            ocr_scale: self.ocr_scale_choice(),
841            artifacts_dir: self.artifacts_dir.clone(),
842            page_break_placeholder: self.page_break_placeholder.clone(),
843        }
844    }
845
846    /// Convert a single source document.
847    pub fn convert(&self, source: SourceDocument) -> Result<ConversionResult, ConversionError> {
848        if let Some(allowed) = &self.allowed_formats {
849            if !allowed.contains(&source.format) {
850                return Err(ConversionError::UnsupportedFormat(source.format));
851            }
852        }
853        let source = self.with_encoding(source);
854
855        let mut document = match source.format {
856            // A legacy APS (Automated Patent System) plain-text patent (`PATN`
857            // first record) is reconstructed verbatim, mirroring docling.
858            InputFormat::Md if crate::backend::uspto::looks_like_aps(&source.text()?) => {
859                crate::backend::uspto::convert_aps(&source)?
860            }
861            // A text/Markdown-typed file that is actually an XML document (e.g. a
862            // JATS article saved with a `.txt` extension) routes to the XML
863            // backends by content, mirroring docling's content-based detection.
864            InputFormat::Md if looks_like_xml(&source.text()?) => match sniff_xml(&source.bytes) {
865                InputFormat::XmlUspto => UsptoBackend.convert(&source)?,
866                InputFormat::XmlXbrl => {
867                    crate::backend::xbrl::convert_xbrl(&source, self.xbrl_taxonomy.as_deref())?
868                }
869                // docling's format detection reads an XML-looking `.txt` as
870                // `application/xml` and, when its DOCTYPE names a JATS DTD
871                // (`JATS-journalpublishing…` / `JATS-archive…`), converts it
872                // with the JATS backend like a real `.nxml`; any other XML
873                // saved as `.txt` is reconstructed generically
874                // (element-by-element).
875                _ if has_jats_doctype(&source.text()?) => JatsBackend {
876                    fetch_images: self.fetch_images,
877                }
878                .convert(&source)?,
879                _ => crate::backend::jats::convert_generic(&source)?,
880            },
881            // DeepSeek-OCR annotated Markdown (VLM token format) is detected by
882            // its `<|ref|>…[[bbox]]` annotations and parsed separately.
883            InputFormat::Md if is_deepseek_markdown(&source.text()?) => {
884                DeepSeekBackend.convert(&source)?
885            }
886            InputFormat::Md => MarkdownBackend {
887                strict: self.strict,
888            }
889            .convert(&source)?,
890            InputFormat::Csv => CsvBackend.convert(&source)?,
891            InputFormat::Html => {
892                // Optionally resolve the CSS cascade in a headless browser first
893                // (strips computed-hidden elements); everything else stays in the
894                // Rust HTML backend, which runs on the cleaned HTML.
895                // Bytes → text through docling's BeautifulSoup decoding order
896                // (#371): BOM, declared charset, UTF-8, windows-1252 — so a
897                // legacy windows-1252 page converts instead of failing the
898                // UTF-8 check every other text backend applies.
899                let decoded = crate::backend::decode_html_bytes(&source.bytes);
900                let html = self.maybe_prerender(&decoded)?;
901                if self.fetch_images {
902                    let resolver = crate::backend::FsImageResolver::new(
903                        source.base_dir().map(|p| p.to_path_buf()),
904                        source.base_url.clone(),
905                    );
906                    crate::backend::convert_html(&source.name, &html, &resolver)
907                } else {
908                    crate::backend::convert_html(&source.name, &html, &crate::backend::NoFetch)
909                }
910            }
911            InputFormat::Asciidoc => AsciiDocBackend {
912                fetch_images: self.fetch_images,
913            }
914            .convert(&source)?,
915            InputFormat::Xlsx => XlsxBackend {
916                skip_empty: self.skip_empty_cells,
917            }
918            .convert(&source)?,
919            InputFormat::Pptx => PptxBackend.convert(&source)?,
920            // RTF (#209): a docling.rs extension — docling reaches RTF only via
921            // LibreOffice; here it parses natively (hand-rolled tokenizer).
922            InputFormat::Rtf => RtfBackend.convert(&source)?,
923            InputFormat::Visio => VisioBackend.convert(&source)?,
924            // AbiWord (#216): docling.rs extension, native AWML parse.
925            InputFormat::Abiword => AbwBackend.convert(&source)?,
926            // WordPerfect 5.x/6.x+ (#216): docling.rs extension, native parse
927            // of the ÿWPC function-code stream.
928            InputFormat::WordPerfect => WpdBackend.convert(&source)?,
929            // Microsoft Works word processor (#216): docling.rs extension,
930            // native parse after libwps.
931            InputFormat::Works => WpsBackend.convert(&source)?,
932            // StarOffice 5 binaries (#215): docling.rs extension, native CFB
933            // parse (docling would go through LibreOffice).
934            InputFormat::StarOffice5 => StarOffice5Backend.convert(&source)?,
935            // DjVu (#434): docling.rs extension, pure-Rust decode (`djvu-rs`).
936            // The hidden text layer is the default; a scan-only DjVu falls back
937            // to rasterize + OCR when the ML pipeline is built.
938            InputFormat::Djvu => self.convert_djvu(&source)?,
939            // DIF/SYLK/dBase (#216): docling.rs extensions, one content-sniffing
940            // backend for the three table relics.
941            InputFormat::Dbf | InputFormat::Dif | InputFormat::Sylk => {
942                InterchangeBackend.convert(&source)?
943            }
944            // Lotus/Quattro/Works record streams (#216): one BOF-sniffing
945            // backend for the whole DOS-era family.
946            InputFormat::Lotus => LotusBackend.convert(&source)?,
947            // Quattro Pro (#216): docling.rs extension, native parse after
948            // libwps (DOS/Windows record streams, QPW OLE zones).
949            InputFormat::QuattroPro => QuattroBackend.convert(&source)?,
950            InputFormat::Docx => DocxBackend.convert(&source)?,
951            // Legacy binary Office (issue #127): parsed natively — docling
952            // proper converts these through LibreOffice first (PR #3804).
953            InputFormat::Xls => XlsBackend {
954                skip_empty: self.skip_empty_cells,
955            }
956            .convert(&source)?,
957            InputFormat::Ppt => PptBackend.convert(&source)?,
958            InputFormat::Doc => DocBackend.convert(&source)?,
959            InputFormat::Vtt => WebVttBackend.convert(&source)?,
960            InputFormat::Ebcdic => EbcdicBackend {
961                layout: self.ebcdic_layout.clone(),
962            }
963            .convert(&source)?,
964            InputFormat::Email => EmailBackend {
965                list_attachments: self.list_attachments,
966            }
967            .convert(&source)?,
968            InputFormat::Mhtml => MhtmlBackend {
969                fetch_images: self.fetch_images,
970                use_web_browser: self.use_web_browser,
971            }
972            .convert(&source)?,
973            InputFormat::Epub => EpubBackend {
974                fetch_images: self.fetch_images,
975                use_web_browser: self.use_web_browser,
976            }
977            .convert(&source)?,
978            InputFormat::JsonDocling => DoclingJsonBackend.convert(&source)?,
979            InputFormat::Latex => LatexBackend.convert(&source)?,
980            // A bare `.xml` defaults to XmlJats; sniff the content to route to the
981            // right XML backend (docling distinguishes by DOCTYPE / root element).
982            InputFormat::XmlJats | InputFormat::XmlUspto | InputFormat::XmlXbrl => {
983                match sniff_xml(&source.bytes) {
984                    InputFormat::XmlUspto => UsptoBackend.convert(&source)?,
985                    InputFormat::XmlXbrl => {
986                        crate::backend::xbrl::convert_xbrl(&source, self.xbrl_taxonomy.as_deref())?
987                    }
988                    _ => JatsBackend {
989                        fetch_images: self.fetch_images,
990                    }
991                    .convert(&source)?,
992                }
993            }
994            InputFormat::Odt | InputFormat::Ods | InputFormat::Odp => {
995                crate::backend::convert_odf(&source, self.fetch_images)?
996            }
997            // DocLang back in: bare XML (`.dclg`/`.dclg.xml`) or the OPC
998            // archive `--to dclx` writes.
999            InputFormat::XmlDoclang | InputFormat::Dclx => {
1000                crate::backend::DoclangBackend.convert(&source)?
1001            }
1002            // Raw DocTags (VLM token markup, #152): the tolerant docling-core
1003            // parser — never fails, best-effort document out.
1004            InputFormat::DocTags => {
1005                let mut doc = docling_core::doctags::parse(&source.text()?);
1006                doc.name = source.name.clone();
1007                doc
1008            }
1009            #[cfg(feature = "pdf")]
1010            InputFormat::Pdf => self
1011                .ml_pipeline()
1012                .map(|p| {
1013                    p.force_full_page_ocr(self.force_full_page_ocr)
1014                        .pages(self.page_range)
1015                })
1016                .and_then(|mut p| p.convert(&source.bytes, None, &source.name))
1017                .map_err(|e| ConversionError::with_source("pdf", e))?,
1018            // SVG (#212), the ML route: rasterize (resvg, white-backed PNG at
1019            // ~2048px long side) and ride the image pipeline. `--no-ocr` short-
1020            // circuits to direct <text> extraction instead — the SVG carries
1021            // its text natively, so skipping OCR must not mean losing it.
1022            #[cfg(feature = "pdf")]
1023            InputFormat::Svg if !self.no_ocr && !self.skip_ocr => {
1024                let png = crate::backend::svg::rasterize_png(&source.bytes)?;
1025                self.ml_pipeline()
1026                    .and_then(|mut p| p.convert_image(&png, &source.name))
1027                    .map_err(|e| ConversionError::with_source("svg", e))?
1028            }
1029            // SVG without the ML pipeline (pdf-text / wasm builds) or with
1030            // --no-ocr / --skip-ocr: pure-Rust <text> extraction, flat
1031            // paragraphs in reading order (the pdf / pdf-text split, applied
1032            // to SVG) — the SVG carries its text natively, so skipping OCR
1033            // must not mean losing it.
1034            InputFormat::Svg => crate::backend::SvgBackend.convert(&source)?,
1035            // Apple iWork (#213): pure-Rust IWA text extraction, all builds.
1036            InputFormat::Pages | InputFormat::Numbers | InputFormat::Keynote => {
1037                crate::backend::IworkBackend.convert(&source)?
1038            }
1039            #[cfg(feature = "pdf")]
1040            InputFormat::Image => self
1041                .ml_pipeline()
1042                .and_then(|mut p| p.convert_image(&source.bytes, &source.name))
1043                .map_err(|e| ConversionError::with_source("image", e))?,
1044            #[cfg(feature = "pdf")]
1045            InputFormat::MetsGbs => self
1046                .ml_pipeline()
1047                .and_then(|mut p| {
1048                    docling_pdf::convert_mets_gbs_with_pipeline(&source.bytes, &source.name, &mut p)
1049                })
1050                .map_err(|e| ConversionError::with_source("mets-gbs", e))?,
1051            // Audio → Whisper ASR (symphonia decode + ONNX inference); each
1052            // transcribed segment becomes a `[time: start-end] text` paragraph.
1053            #[cfg(feature = "asr")]
1054            InputFormat::Audio => docling_asr::convert_audio_with_options(
1055                &source.bytes,
1056                &source.name,
1057                self.asr_model.as_deref(),
1058                self.asr_lang.as_deref(),
1059            )
1060            .map_err(|e| ConversionError::with_source(source.format.as_str(), e))?,
1061            // Video (#138): the audio track transcribes through the same ASR
1062            // path (Phase 1), and — when the ffmpeg binary is available —
1063            // sampled frames interleave with the transcript as timestamped
1064            // pictures (Phase 2). Without ffmpeg: transcript only.
1065            #[cfg(feature = "asr")]
1066            InputFormat::Video => crate::video::convert_video(
1067                &source.bytes,
1068                &source.name,
1069                self.asr_model.as_deref(),
1070                self.asr_lang.as_deref(),
1071                self.video_frames.unwrap_or(DEFAULT_VIDEO_FRAMES),
1072            )
1073            .map_err(|e| ConversionError::with_source(source.format.as_str(), e))?,
1074            // Without the full ML pipeline, `pdf-text` still converts a PDF's
1075            // embedded text layer (pure Rust — the wasm32 path), equivalent to
1076            // `--no-ocr`: flat paragraphs, no headings/tables/pictures. A
1077            // scanned PDF has no text layer, so an empty document means "this
1078            // needs OCR" — say so instead of returning nothing.
1079            #[cfg(all(feature = "pdf-text", not(feature = "pdf")))]
1080            InputFormat::Pdf => {
1081                let doc = docling_pdf::convert_text_layer_pages(
1082                    &source.bytes,
1083                    &source.name,
1084                    self.page_range,
1085                )
1086                .map_err(|e| ConversionError::with_source("pdf", e))?;
1087                if doc.nodes.is_empty() {
1088                    return Err(ConversionError::Parse(
1089                        "PDF has no embedded text layer (scanned/image-only?); OCR needs a \
1090                         build with the `pdf` feature"
1091                            .into(),
1092                    ));
1093                }
1094                doc
1095            }
1096            // Compiled without the ML pipelines: the formats stay detectable,
1097            // but converting them needs a build with the matching feature.
1098            #[cfg(not(any(feature = "pdf", feature = "pdf-text")))]
1099            InputFormat::Pdf => {
1100                return Err(ConversionError::Parse(
1101                    "Pdf conversion is not compiled in (rebuild with the `pdf` feature, or \
1102                     `pdf-text` for text-layer-only extraction)"
1103                        .into(),
1104                ))
1105            }
1106            #[cfg(not(feature = "pdf"))]
1107            InputFormat::Image | InputFormat::MetsGbs => {
1108                return Err(ConversionError::Parse(format!(
1109                    "{:?} conversion is not compiled in (rebuild with the `pdf` feature)",
1110                    source.format
1111                )))
1112            }
1113            #[cfg(not(feature = "asr"))]
1114            InputFormat::Audio | InputFormat::Video => {
1115                return Err(ConversionError::Parse(format!(
1116                    "{} conversion is not compiled in (rebuild with the `asr` feature)",
1117                    source.format.as_str()
1118                )))
1119            }
1120        };
1121        // Carry the mode so `result.document.export_to_markdown()` reflects it.
1122        document.strict_markdown = self.strict;
1123        // Compact tables (#271) is additive: the PDF backend already turns it
1124        // on for its own corpus; never turn it back off here.
1125        if self.compact_tables {
1126            document.compact_tables = true;
1127        }
1128        // Page-break placeholder: a serializer knob like `strict`, carried on
1129        // the document so every Markdown export of it agrees.
1130        if self.page_break_placeholder.is_some() {
1131            document.page_break_placeholder = self.page_break_placeholder.clone();
1132        }
1133        // First-class cells for every table (#240): backends with page
1134        // geometry (the PDF TableFormer paths) set them; everything else —
1135        // declarative tables included — derives them from the grid plus the
1136        // structure overlay (real spans for DOCX/XLSX merges, HTML `th`
1137        // headers, ODF covered cells; 1×1 records otherwise), so the repair
1138        // API and the JSON `table_cells` are populated uniformly.
1139        for table in document.tables_mut() {
1140            if table.cells.is_none() {
1141                table.cells = Some(table.derive_cells());
1142            }
1143        }
1144
1145        Ok(ConversionResult {
1146            document,
1147            status: ConversionStatus::Success,
1148            input_name: source.name,
1149            format: source.format,
1150        })
1151    }
1152}
1153
1154#[cfg(test)]
1155mod tests {
1156    use super::*;
1157
1158    /// docling reads an XML-looking `.txt` as `application/xml` and converts
1159    /// it with the JATS backend when its DOCTYPE names a JATS DTD; other XML
1160    /// under `.txt` stays the generic element-by-element reconstruction.
1161    #[test]
1162    fn jats_doctype_text_file_uses_the_jats_backend() {
1163        let jats = "<!DOCTYPE article PUBLIC \"-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.2 20190208//EN\" \"JATS-archivearticle1.dtd\">\n<article><front><article-meta><title-group><article-title>T</article-title></title-group></article-meta></front><body><sec><title>S</title><p>Body.</p></sec></body></article>";
1164        let src = SourceDocument::from_bytes("a.txt", InputFormat::Md, jats.as_bytes().to_vec());
1165        let doc = DocumentConverter::new().convert(src).unwrap().document;
1166        assert!(doc.tree.is_some(), "JATS tree expected");
1167        assert_eq!(doc.export_to_markdown().trim(), "# T\n\n## S\n\nBody.");
1168        let other = "<?xml version=\"1.0\"?>\n<article><body><sec><title>S</title><p>Body.</p></sec></body></article>";
1169        let src = SourceDocument::from_bytes("b.txt", InputFormat::Md, other.as_bytes().to_vec());
1170        let doc = DocumentConverter::new().convert(src).unwrap().document;
1171        assert!(doc.tree.is_none(), "generic XML path expected");
1172    }
1173
1174    /// docling's `TextBackendOptions.encoding`: the converter's `encoding`
1175    /// decodes text inputs as named (a source's own setting wins), and an
1176    /// undecodable byte or unknown label is an error, not a guess.
1177    #[test]
1178    fn encoding_option_decodes_text_inputs() {
1179        let sjis = b"# \x93\xfa\x96\x7b\n".to_vec();
1180        let md = |c: &DocumentConverter, s: SourceDocument| {
1181            c.convert(s).map(|r| r.document.export_to_markdown())
1182        };
1183        let conv = DocumentConverter::new().encoding(Some("shift_jis".into()));
1184        let src = || SourceDocument::from_bytes("doc", InputFormat::Md, sjis.clone());
1185        assert_eq!(md(&conv, src()).unwrap().trim(), "# 日本");
1186        // Detection reads the same bytes as windows-1252.
1187        assert_ne!(
1188            md(&DocumentConverter::new(), src()).unwrap().trim(),
1189            "# 日本"
1190        );
1191        // The source's own encoding takes precedence over the converter's.
1192        let own = src().with_encoding(Some("shift_jis".into()));
1193        let latin = DocumentConverter::new().encoding(Some("latin1".into()));
1194        assert_eq!(md(&latin, own).unwrap().trim(), "# 日本");
1195        assert!(md(
1196            &DocumentConverter::new().encoding(Some("utf-8".into())),
1197            src()
1198        )
1199        .is_err());
1200        assert!(md(
1201            &DocumentConverter::new().encoding(Some("nope-1".into())),
1202            src()
1203        )
1204        .is_err());
1205    }
1206
1207    #[test]
1208    fn end_to_end_markdown() {
1209        let src =
1210            SourceDocument::from_bytes("doc", InputFormat::Md, b"# Hello\n\nWorld.\n".to_vec());
1211        let result = DocumentConverter::new().convert(src).unwrap();
1212        assert_eq!(result.status, ConversionStatus::Success);
1213        assert_eq!(result.document.export_to_markdown(), "# Hello\n\nWorld.\n");
1214    }
1215
1216    #[test]
1217    fn doctags_input_converts() {
1218        // Raw DocTags markup (#152) — the VLM token stream — as a first-class
1219        // input format (.doctags/.dt), through the tolerant docling-core
1220        // parser.
1221        let markup = b"<doctag><section_header_level_1><loc_1><loc_2><loc_3><loc_4>Intro</section_header_level_1><text>Body.</text></doctag>"
1222            .to_vec();
1223        let src = SourceDocument::from_bytes("page.doctags", InputFormat::DocTags, markup);
1224        let result = DocumentConverter::new().convert(src).unwrap();
1225        let md = result.document.export_to_markdown();
1226        assert!(md.contains("## Intro"), "{md}");
1227        assert!(md.contains("Body."), "{md}");
1228    }
1229
1230    #[test]
1231    fn doclang_xml_round_trips() {
1232        // Every input format now has a backend; DocLang XML reads back in and
1233        // re-exports as Markdown.
1234        let xml = b"<doclang version=\"0.7\">\n  <heading>Title</heading>\n  \
1235                    <text>Hello <bold>world</bold></text>\n</doclang>"
1236            .to_vec();
1237        let src = SourceDocument::from_bytes("doc.dclg", InputFormat::XmlDoclang, xml);
1238        let result = DocumentConverter::new().convert(src).unwrap();
1239        let md = result.document.export_to_markdown();
1240        assert!(md.contains("# Title"), "{md}");
1241        assert!(md.contains("**world**"), "{md}");
1242    }
1243
1244    #[test]
1245    fn sniffs_uspto_doctype_case_insensitively() {
1246        // docling PR #3801: Grant Full Text v2.5 files were missed when the
1247        // DOCTYPE casing differed.
1248        for head in [
1249            "<?xml version=\"1.0\"?><!DOCTYPE PATDOC SYSTEM \"ST32-US-Grant-025xml.dtd\"><PATDOC/>",
1250            "<?xml version=\"1.0\"?><!DOCTYPE patdoc SYSTEM \"st32-us-grant-025xml.dtd\"><patdoc/>",
1251            "<?xml version=\"1.0\"?><US-PATENT-GRANT-V4/>",
1252        ] {
1253            assert_eq!(
1254                super::sniff_xml(head.as_bytes()),
1255                InputFormat::XmlUspto,
1256                "head: {head}"
1257            );
1258        }
1259    }
1260}