docling/converter.rs
1//! The top-level `DocumentConverter`.
2
3use std::collections::HashSet;
4use std::path::PathBuf;
5
6use crate::backend::{
7 is_deepseek_markdown, AbwBackend, AsciiDocBackend, CsvBackend, DeclarativeBackend,
8 DeepSeekBackend, DocBackend, DoclingJsonBackend, DocxBackend, EbcdicBackend, EmailBackend,
9 EpubBackend, InterchangeBackend, JatsBackend, LatexBackend, LotusBackend, MarkdownBackend,
10 MhtmlBackend, PptBackend, PptxBackend, QuattroBackend, RtfBackend, StarOffice5Backend,
11 UsptoBackend, VisioBackend, WebVttBackend, WpdBackend, WpsBackend, XlsBackend, XlsxBackend,
12};
13
14/// Whether `text` begins with an XML prolog — an `<?xml …?>` declaration or a
15/// non-HTML `<!DOCTYPE …>`. Used to route XML documents that arrived with a
16/// text/Markdown extension (e.g. a JATS article saved as `.txt`) to the XML
17/// backends. An HTML5 `<!DOCTYPE html>` is deliberately excluded.
18fn looks_like_xml(text: &str) -> bool {
19 let head = text.trim_start();
20 if head.starts_with("<?xml") {
21 return true;
22 }
23 if let Some(rest) = head.get(..9) {
24 if rest.eq_ignore_ascii_case("<!doctype") {
25 return !head[9..]
26 .trim_start()
27 .to_ascii_lowercase()
28 .starts_with("html");
29 }
30 }
31 false
32}
33
34/// Pick the concrete XML backend for a generic `.xml` source by sniffing its
35/// DOCTYPE / root element (the first part of the file).
36/// docling's `_guess_from_content` JATS rule for an `application/xml` input:
37/// the `<!DOCTYPE …>` declaration names a JATS DTD.
38fn has_jats_doctype(text: &str) -> bool {
39 let Some(start) = text.find("<!DOCTYPE ") else {
40 return false;
41 };
42 let Some(len) = text[start..].find('>') else {
43 return false;
44 };
45 let doctype = &text[start..start + len];
46 doctype.contains("JATS-journalpublishing") || doctype.contains("JATS-archive")
47}
48
49fn sniff_xml(bytes: &[u8]) -> InputFormat {
50 // Lossy over the raw head (docling#4038, 2.122): the window can cut a
51 // well-formed file mid-codepoint — a fixed-offset `&str` slice would
52 // panic on that char boundary — and an XML document may legitimately
53 // declare a non-UTF-8 encoding. Every marker matched below is ASCII, so
54 // replacement characters cannot change the outcome.
55 let head = String::from_utf8_lossy(&bytes[..bytes.len().min(4000)]);
56 let head = head.as_ref();
57 // Case-insensitive: USPTO DOCTYPE/root casing varies in the wild (docling
58 // PR #3801 — Grant Full Text v2.5 files were missed on casing).
59 let lower = head.to_ascii_lowercase();
60 if lower.contains("us-patent")
61 || lower.contains("patent-application-publication")
62 || lower.contains("patdoc")
63 || lower.contains("<pap-v1")
64 {
65 InputFormat::XmlUspto
66 } else if head.contains("<doclang") {
67 // A bare DocLang document saved as `.xml` (docling names them
68 // `*.dclg.xml`, whose final extension is plain `xml`).
69 InputFormat::XmlDoclang
70 } else if crate::backend::xbrl::looks_like_xbrl(head) {
71 InputFormat::XmlXbrl
72 } else {
73 InputFormat::XmlJats
74 }
75}
76use crate::error::ConversionError;
77use crate::format::InputFormat;
78use crate::result::{ConversionResult, ConversionStatus};
79use crate::source::SourceDocument;
80#[cfg(feature = "pdf")]
81use crate::stream::MarkdownStream;
82#[cfg(feature = "pdf")]
83use docling_core::ImageMode;
84
85/// Routes a [`SourceDocument`] to the backend for its format and returns a
86/// [`ConversionResult`].
87///
88/// The Rust analogue of `docling.document_converter.DocumentConverter`. In
89/// Phase 0 the format→backend dispatch is a direct match; the Python notion of
90/// per-format `FormatOption` (backend + pipeline + options) arrives with the
91/// PDF/ML pipeline in a later phase.
92#[derive(Debug, Clone)]
93pub struct DocumentConverter {
94 allowed_formats: Option<HashSet<InputFormat>>,
95 strict: bool,
96 fetch_images: bool,
97 list_attachments: bool,
98 /// Omit empty cells from sparse spreadsheet table grids (#271, XLSX/XLS
99 /// family; opt-in docling.rs extension).
100 skip_empty_cells: bool,
101 /// Emit Markdown tables in the compact `| a | b |` form instead of the
102 /// width-padded GitHub serializer (#271, opt-in docling.rs extension).
103 compact_tables: bool,
104 /// docling-core's `MarkdownParams.page_break_placeholder`: text inserted
105 /// between pages in the Markdown export; `None` (default) omits breaks.
106 page_break_placeholder: Option<String>,
107 /// EBCDIC copybook layout (#252): inline JSON or a file path. `None`
108 /// falls back to the `<stem>.layout.json` sidecar.
109 ebcdic_layout: Option<String>,
110 no_table_former: bool,
111 no_text_panels: bool,
112 no_ocr: bool,
113 skip_ocr: bool,
114 force_full_page_ocr: bool,
115 /// OCR mode id (docling's `OcrMode`, #254); parsed at the ML call sites.
116 ocr_mode: Option<String>,
117 /// OCR engine id (`ppocr` | `tesseract`, #460); parsed at the ML call
118 /// sites.
119 ocr_engine: Option<String>,
120 /// OCR render scale in px/pt (#254); validated at the ML call sites.
121 ocr_scale: Option<f32>,
122 /// Infer PDF/image section-header levels after assembly (#302, docling's
123 /// `HeadingHierarchyModel`): bookmarks > numbering > font style. Off by
124 /// default — heading levels then stay exactly as detected.
125 heading_hierarchy: bool,
126 use_web_browser: bool,
127 /// Named Whisper model preset for audio sources (docling's ASR model
128 /// specs, PR #3741): English-only / Distil-Whisper variants under
129 /// `.models/asr/<preset>/`. `None` = the default Whisper tiny.
130 asr_model: Option<String>,
131 asr_lang: Option<String>,
132 /// Max sampled frames per video (#138 Phase 2). `None` = the default
133 /// ([`DEFAULT_VIDEO_FRAMES`]); `Some(0)` disables frame extraction.
134 video_frames: Option<usize>,
135 /// The directory an XBRL instance's taxonomy is read from (docling's
136 /// `XBRLBackendOptions.taxonomy`); `None` = the instance's own directory.
137 xbrl_taxonomy: Option<PathBuf>,
138 /// Opt-in PDF/image enrichment models (docling's
139 /// `do_picture_classification` / `do_code_enrichment` /
140 /// `do_formula_enrichment`).
141 enrich: crate::EnrichmentOptions,
142 /// 1-based inclusive PDF page window (#80). See [`Self::page_range`].
143 page_range: Option<(usize, usize)>,
144 /// OCR recognition language for scanned PDF/image pages (`en`/`ch`).
145 /// `None` = the process default (`DOCLING_RS_OCR_LANG`, else English).
146 ocr_lang: Option<String>,
147 /// Character encoding for text inputs (docling's
148 /// `TextBackendOptions.encoding`); `None` = detect. See [`Self::encoding`].
149 encoding: Option<String>,
150 /// Directory referenced-mode streaming writes images into (#80).
151 /// See [`Self::artifacts_dir`].
152 artifacts_dir: String,
153}
154
155/// Default cap on sampled frames per video. Scene changes rarely exceed this
156/// in short clips, and uniform fallback at 8 keeps JSON/DCLX output (which
157/// embeds the PNGs) within sane bounds.
158pub const DEFAULT_VIDEO_FRAMES: usize = 8;
159
160/// Parse a user-facing page-range string (issue #80's `--pages`): `"A-B"` for
161/// an inclusive 1-based window, or a single `"N"` for one page. Whitespace
162/// around the numbers is tolerated. Validation against the actual page count
163/// happens at convert time; this only checks the spelling (`first >= 1`,
164/// `first <= last`).
165pub fn parse_page_range(s: &str) -> Result<(usize, usize), String> {
166 let parse_one = |part: &str| {
167 part.trim()
168 .parse::<usize>()
169 .map_err(|_| format!("invalid page number '{}'", part.trim()))
170 };
171 let (first, last) = match s.split_once('-') {
172 Some((a, b)) => (parse_one(a)?, parse_one(b)?),
173 None => {
174 let n = parse_one(s)?;
175 (n, n)
176 }
177 };
178 if first == 0 {
179 return Err("pages are 1-based; the range starts at 1".into());
180 }
181 if last < first {
182 return Err(format!("range {first}-{last} is inverted (first <= last)"));
183 }
184 Ok((first, last))
185}
186
187impl Default for DocumentConverter {
188 fn default() -> Self {
189 Self {
190 allowed_formats: None,
191 strict: false,
192 fetch_images: false,
193 list_attachments: false,
194 skip_empty_cells: false,
195 compact_tables: false,
196 page_break_placeholder: None,
197 ebcdic_layout: None,
198 no_table_former: false,
199 no_text_panels: false,
200 no_ocr: false,
201 skip_ocr: false,
202 force_full_page_ocr: false,
203 ocr_mode: None,
204 ocr_engine: None,
205 ocr_scale: None,
206 heading_hierarchy: false,
207 use_web_browser: false,
208 asr_model: None,
209 asr_lang: None,
210 video_frames: None,
211 xbrl_taxonomy: None,
212 enrich: crate::EnrichmentOptions::default(),
213 page_range: None,
214 ocr_lang: None,
215 encoding: None,
216 artifacts_dir: "artifacts".to_string(),
217 }
218 }
219}
220
221impl DocumentConverter {
222 /// A converter that accepts every supported format.
223 pub fn new() -> Self {
224 Self::default()
225 }
226
227 /// A converter restricted to an explicit set of formats. Sources of any
228 /// other format are rejected with [`ConversionError::UnsupportedFormat`].
229 pub fn with_allowed_formats(formats: impl IntoIterator<Item = InputFormat>) -> Self {
230 Self {
231 allowed_formats: Some(formats.into_iter().collect()),
232 ..Self::default()
233 }
234 }
235
236 /// Convert only PDF pages `first..=last` (**1-based** inclusive, the page
237 /// numbers a viewer shows — issue #80's `--pages A-B`). Out-of-window pages
238 /// are skipped before rasterization, so converting 3 pages of a 500-page
239 /// PDF costs 3 pages. `last` clamps to the document; a window that selects
240 /// no pages at all errors at convert time. Non-PDF formats ignore the
241 /// window (they convert whole).
242 pub fn page_range(mut self, first: usize, last: usize) -> Self {
243 self.page_range = Some((first, last));
244 self
245 }
246
247 /// OCR recognition language for scanned PDF/image pages: `"en"` (the
248 /// default — English PP-OCRv3, proper Latin word spacing) or `"ch"` (the
249 /// multilingual model docling conformance is measured with — glues Latin
250 /// words). An unknown value warns at conversion time and uses the
251 /// default; explicit `DOCLING_OCR_REC_ONNX`/`DOCLING_OCR_DICT` paths win
252 /// over this switch. Under [`ocr_engine`](Self::ocr_engine) `tesseract`
253 /// it is Tesseract's language list instead — tessdata stems (`"deu"`,
254 /// `"eng+fra"`, `"script/Cyrillic"`) or BCP-47 tags mapped onto them
255 /// (`"de"`, `"zh-Hant"`), see [`docling_pdf::tesseract_lang_arg`]; `en`
256 /// and `ch` keep working there. Formats that never OCR ignore it.
257 pub fn ocr_lang(mut self, lang: impl Into<String>) -> Self {
258 self.ocr_lang = Some(lang.into());
259 self
260 }
261
262 /// The configured ML pipeline for one conversion (models load per call —
263 /// callers that convert many files hold a warm [`docling_pdf::Pipeline`]
264 /// themselves). Grew out of docling-pdf's `convert_with_options` free
265 /// functions, whose fixed signatures couldn't take #244's `skip_ocr`.
266 #[cfg(feature = "pdf")]
267 fn ml_pipeline(&self) -> Result<docling_pdf::Pipeline, docling_pdf::PdfError> {
268 Ok(docling_pdf::Pipeline::new()?
269 .no_table_former(self.no_table_former)
270 .no_ocr(self.no_ocr)
271 .skip_ocr(self.skip_ocr)
272 .no_text_panels(self.no_text_panels)
273 .enrichments(self.enrich)
274 .ocr_lang(self.ocr_lang_choice())
275 .ocr_engine(self.ocr_engine_choice())
276 .tesseract_lang(self.tesseract_lang_choice())
277 .ocr_mode(self.ocr_mode_choice())
278 .ocr_scale(self.ocr_scale_choice())
279 .heading_hierarchy(docling_pdf::HeadingHierarchyOptions::enabled(
280 self.heading_hierarchy,
281 )))
282 }
283
284 /// DjVu (#434): the hidden text layer by default; a scan-only DjVu (no
285 /// text layer on any selected page) falls back to rasterize + OCR when the
286 /// ML pipeline is built and OCR is not disabled, otherwise it degrades to
287 /// an empty document with a warning. See [`crate::backend::djvu`].
288 fn convert_djvu(
289 &self,
290 source: &SourceDocument,
291 ) -> Result<docling_core::DoclingDocument, ConversionError> {
292 use crate::backend::djvu;
293 let text = djvu::convert_text_layer(source, self.page_range)?;
294 if !djvu::is_text_layer_empty(&text) {
295 return Ok(text);
296 }
297 // No text layer anywhere in the selection.
298 #[cfg(feature = "pdf")]
299 if !self.no_ocr && !self.skip_ocr {
300 let pngs = djvu::rasterize_pages(&source.bytes, self.page_range)?;
301 let mut pipeline = self
302 .ml_pipeline()
303 .map_err(|e| ConversionError::with_source("djvu", e))?;
304 let mut doc = docling_core::DoclingDocument::new(&source.name);
305 for (i, (page_no, png)) in pngs.iter().enumerate() {
306 if i > 0 {
307 doc.push(docling_core::Node::PageBreak);
308 }
309 let mut page = pipeline
310 .convert_image(png, &source.name)
311 .map_err(|e| ConversionError::with_source("djvu", e))?;
312 // An image is its own page 1; the DjVu page keeps its number so
313 // a `--pages` window's JSON `pages` matches the text-layer path.
314 docling_pdf::assemble::stamp_page_no(&mut page.nodes, *page_no);
315 doc.nodes.extend(page.nodes);
316 doc.links.extend(page.links);
317 }
318 return Ok(doc);
319 }
320 eprintln!(
321 "docling: warning: DjVu '{}' has no text layer and OCR is unavailable/disabled; \
322 the document is empty",
323 source.name
324 );
325 Ok(text)
326 }
327
328 /// The parsed [`Self::ocr_lang`] choice for the ML call sites; a value
329 /// that parses to nothing warns here (once per conversion) rather than
330 /// erroring — same degradation the env selector applies.
331 #[cfg(feature = "pdf")]
332 fn ocr_lang_choice(&self) -> Option<docling_pdf::OcrLang> {
333 let raw = self.ocr_lang.as_deref()?;
334 // Under Tesseract the value is its language list, not a PP-OCR model
335 // — see `tesseract_lang_choice`.
336 if self.ocr_engine_choice() == Some(docling_pdf::OcrEngine::Tesseract) {
337 return None;
338 }
339 let parsed = docling_pdf::OcrLang::parse(raw);
340 if parsed.is_none() {
341 eprintln!(
342 "docling: ocr_lang {raw:?} names no language the OCR models read ({}); using \
343 the default",
344 docling_pdf::OcrLang::ACCEPTED
345 );
346 }
347 parsed
348 }
349
350 /// The parsed [`Self::ocr_engine`] choice (#460), with the same
351 /// warn-and-default degradation as [`ocr_lang_choice`](Self::ocr_lang_choice).
352 #[cfg(feature = "pdf")]
353 fn ocr_engine_choice(&self) -> Option<docling_pdf::OcrEngine> {
354 let raw = self.ocr_engine.as_deref()?;
355 let parsed = docling_pdf::OcrEngine::parse(raw);
356 if parsed.is_none() {
357 eprintln!(
358 "docling: ocr_engine {raw:?} is not {}; using the default",
359 docling_pdf::OcrEngine::ACCEPTED
360 );
361 }
362 parsed
363 }
364
365 /// Tesseract's `-l` argument from [`Self::ocr_lang`] (#460) when the
366 /// engine is Tesseract: an unmappable value warns and leaves Tesseract's
367 /// default language.
368 #[cfg(feature = "pdf")]
369 fn tesseract_lang_choice(&self) -> Option<String> {
370 if self.ocr_engine_choice() != Some(docling_pdf::OcrEngine::Tesseract) {
371 return None;
372 }
373 let raw = self.ocr_lang.as_deref()?;
374 match docling_pdf::tesseract_lang_arg(raw) {
375 Ok(arg) => Some(arg),
376 Err(e) => {
377 eprintln!("docling: {e}; using Tesseract's default language");
378 None
379 }
380 }
381 }
382
383 /// The parsed [`Self::ocr_mode`] choice (#254), with the same
384 /// warn-and-default degradation as [`ocr_lang_choice`](Self::ocr_lang_choice).
385 #[cfg(feature = "pdf")]
386 fn ocr_mode_choice(&self) -> Option<docling_pdf::OcrMode> {
387 let raw = self.ocr_mode.as_deref()?;
388 let parsed = docling_pdf::OcrMode::parse(raw);
389 if parsed.is_none() {
390 eprintln!(
391 "docling: ocr_mode {raw:?} is not \
392 default|full_page|layout_regions|pdf_aware_layout_regions; using the default"
393 );
394 }
395 parsed
396 }
397
398 /// The validated [`Self::ocr_scale`] (#254): non-positive/non-finite
399 /// values warn and fall back to the engine default.
400 #[cfg(feature = "pdf")]
401 fn ocr_scale_choice(&self) -> Option<f32> {
402 let s = self.ocr_scale?;
403 if !(s.is_finite() && s > 0.0) {
404 eprintln!("docling: ocr_scale {s} is not a positive number; using the default");
405 return None;
406 }
407 Some(s)
408 }
409
410 /// Where [`ImageMode::Referenced`] streaming writes image files, and the
411 /// link prefix used in the Markdown (default `artifacts`, matching the
412 /// buffered export's convention). Relative paths resolve against the
413 /// process working directory.
414 pub fn artifacts_dir(mut self, dir: impl Into<String>) -> Self {
415 self.artifacts_dir = dir.into();
416 self
417 }
418
419 /// Decode text inputs (Markdown, CSV, AsciiDoc, WebVTT, LaTeX, the XML
420 /// dialects, …) with this character encoding instead of detecting one —
421 /// docling's `TextBackendOptions.encoding`
422 /// (`MarkdownBackendOptions(encoding="shift_jis")`). A WHATWG encoding
423 /// label (`shift_jis`, `koi8-r`, `windows-1251`, `latin1`; Python codec
424 /// spellings with `_` are accepted). Nothing is guessed: bytes the
425 /// encoding cannot decode fail the conversion, as does an unknown label.
426 /// `None` (default) detects — a byte-order mark, then UTF-8, then
427 /// windows-1252 (see [`SourceDocument::text`]). A source that already
428 /// carries its own [`SourceDocument::encoding`] keeps it.
429 pub fn encoding(mut self, label: Option<String>) -> Self {
430 self.encoding = label;
431 self
432 }
433
434 /// The converter's [`encoding`](Self::encoding) applied to a source that
435 /// did not set its own.
436 fn with_encoding(&self, mut source: SourceDocument) -> SourceDocument {
437 if source.encoding.is_none() {
438 source.encoding = self.encoding.clone();
439 }
440 source
441 }
442
443 /// Cap the number of frames sampled from a video (#138 Phase 2); `0`
444 /// disables frame extraction entirely (Phase 1 behavior: transcript only).
445 /// Defaults to [`DEFAULT_VIDEO_FRAMES`]. Frames are extracted with the
446 /// `ffmpeg` binary when present (`DOCLING_FFMPEG` overrides the path);
447 /// without it a video converts to its transcript alone.
448 pub fn video_frames(mut self, max: usize) -> Self {
449 self.video_frames = Some(max);
450 self
451 }
452
453 /// The folder holding the taxonomy an XBRL instance refers to (docling's
454 /// `XBRLBackendOptions.taxonomy`): the filing's extension schema and
455 /// linkbases at the relative paths its `link:schemaRef` names, plus any
456 /// taxonomy packages (`.zip` with a `META-INF/catalog.xml`) that map the
457 /// base taxonomies' `http(s)` URLs to files for offline use. Without it the
458 /// instance's own directory is searched. Nothing is fetched remotely; a
459 /// document that cannot be found only costs the fact graph the hierarchy
460 /// it would have contributed.
461 pub fn xbrl_taxonomy(mut self, dir: impl Into<PathBuf>) -> Self {
462 self.xbrl_taxonomy = Some(dir.into());
463 self
464 }
465
466 /// Select a named Whisper model preset for audio sources — the
467 /// English-only (`whisper_tiny_en`, `whisper_base_en`, `whisper_small_en`)
468 /// and Distil-Whisper (`whisper_distil_small_en`) variants of docling's
469 /// ASR model specs. `None` (default) uses Whisper tiny (multilingual)
470 /// from `.models/asr/`; presets load from `.models/asr/<preset>/` (fetch
471 /// them with `download_dependencies.sh --asr-model <preset>`).
472 pub fn asr_model(mut self, model: Option<String>) -> Self {
473 self.asr_model = model;
474 self
475 }
476
477 /// Select the ASR transcription language for audio/video sources: a
478 /// Whisper code (`en`, `de`, `zh`, …) or `auto`. `None` (default) falls
479 /// back to `DOCLING_RS_ASR_LANG`, and — when that is unset too — to
480 /// per-file auto-detection from the first 30-second window (docling
481 /// 2.116 parity). English-only presets always transcribe English.
482 pub fn asr_lang(mut self, lang: Option<String>) -> Self {
483 self.asr_lang = lang;
484 self
485 }
486
487 /// Select the Markdown export mode for documents this converter produces.
488 ///
489 /// `false` (default) makes [`crate::DoclingDocument::export_to_markdown`]
490 /// reproduce docling's legacy output byte-for-byte; `true` makes it emit
491 /// cleaner, more conformant Markdown (code-fence languages preserved, no
492 /// inline-run spacing artifacts, no entity re-escaping). Rust-only — Python
493 /// docling has no such switch.
494 pub fn strict(mut self, strict: bool) -> Self {
495 self.strict = strict;
496 self
497 }
498
499 /// Fetch and embed external `<img>` images for HTML/EPUB/MHTML/JATS sources.
500 ///
501 /// Off by default (matching docling's `enable_*_fetch=False`), so output is
502 /// unchanged unless you opt in. When on, the HTML/EPUB/MHTML backends
503 /// resolve each `<img src>` — `data:` URIs, local files (relative to the
504 /// source file's directory), `http(s)` URLs, and EPUB/MHTML archive
505 /// entries — and the JATS backend reads a `<fig>`'s `<graphic xlink:href>`
506 /// from the source file's directory (#392), embedding the bytes so they
507 /// survive into JSON `ImageRef`s and
508 /// [`crate::DoclingDocument::export_to_markdown_with_images`].
509 ///
510 /// Remote `http(s)` URLs are fetched over the network; enable only for input
511 /// you trust (it can otherwise be used to make the process issue requests).
512 pub fn fetch_images(mut self, fetch: bool) -> Self {
513 self.fetch_images = fetch;
514 self
515 }
516
517 /// Append an `Attachments` section to converted emails (`.eml` / `.msg`):
518 /// one list item per attachment, `name (content/type)` — names and types
519 /// only, the payload is never embedded. docling's opt-in
520 /// `EmailBackendOptions.list_attachments` (#251); off by default.
521 pub fn list_attachments(mut self, list: bool) -> Self {
522 self.list_attachments = list;
523 self
524 }
525
526 /// Omit empty cells from sparse spreadsheet table grids (#271; XLSX/XLS
527 /// family, opt-in — a docling.rs extension, docling materialises the full
528 /// bounding box). A ragged region's box is mostly padding on sparse
529 /// sheets (~7× output inflation); with this on, each row keeps only its
530 /// occupied cells (merge-covered continuations included) and a table
531 /// that loses cells drops its span/structure overlay. Off by default —
532 /// default output stays byte-for-byte docling.
533 pub fn skip_empty_cells(mut self, skip: bool) -> Self {
534 self.skip_empty_cells = skip;
535 self
536 }
537
538 /// Emit Markdown tables in the compact `| a | b |` / `| - | - |` form
539 /// instead of docling-core's width-padded GitHub serializer (#271, all
540 /// formats; opt-in — a docling.rs extension). On sparse spreadsheets the
541 /// padding dominates the output size; compact rendering keeps the grid
542 /// semantics and drops only whitespace. Off by default — default output
543 /// stays byte-for-byte docling.
544 pub fn compact_tables(mut self, compact: bool) -> Self {
545 self.compact_tables = compact;
546 self
547 }
548
549 /// Insert `placeholder` between pages in the Markdown export — docling's
550 /// `export_to_markdown(page_break_placeholder=…)` (docling-core's
551 /// `MarkdownParams`, docling-serve's `md_page_break_placeholder`). Where
552 /// docling marks a break between two items whose `prov.page_no` differ,
553 /// the serializer here marks it between two rendered blocks separated by a
554 /// page boundary — the `PageInfo` marker opening every PDF page and sheet,
555 /// the `PageBreak` between slides and DjVu / DocTags pages — so a break
556 /// never leads or trails the document and empty pages collapse into one.
557 /// `None` (the default) keeps Markdown free of page breaks, as docling's
558 /// default export is. Markdown only; JSON carries pages in `prov`.
559 pub fn page_break_placeholder(mut self, placeholder: Option<String>) -> Self {
560 self.page_break_placeholder = placeholder;
561 self
562 }
563
564 /// The copybook layout for EBCDIC sources (#252): docling's
565 /// `EbcdicLayout` JSON, inline (a string starting with `{`) or as a file
566 /// path. Without it, a path-loaded source looks for a
567 /// `<stem>.layout.json` sidecar; converting EBCDIC with neither is an
568 /// error — the bytes are meaningless without their copybook.
569 pub fn ebcdic_layout(mut self, layout: impl Into<String>) -> Self {
570 self.ebcdic_layout = Some(layout.into());
571 self
572 }
573
574 /// Option-typed variant of [`ebcdic_layout`](Self::ebcdic_layout) for
575 /// call sites plumbing an optional flag through (`None` keeps the
576 /// sidecar fallback).
577 pub fn ebcdic_layout_opt(mut self, layout: Option<String>) -> Self {
578 self.ebcdic_layout = layout;
579 self
580 }
581
582 /// Skip loading and running the TableFormer table-structure model for
583 /// PDF/image/METS sources.
584 ///
585 /// Off by default. When enabled, table regions are still detected and
586 /// emitted, but their structure is reconstructed geometrically from cell
587 /// positions instead of the ONNX model's predicted structure — no model
588 /// load and no per-table inference, at the cost of table fidelity. Useful
589 /// when parsing speed matters more than exact table structure, especially
590 /// with [`convert_streaming`](Self::convert_streaming).
591 pub fn no_table_former(mut self, disable: bool) -> Self {
592 self.no_table_former = disable;
593 self
594 }
595
596 /// PDF/image: keep every detected picture as a picture — disable the
597 /// text-panel demotion that turns an uncaptioned, dense text-panel
598 /// "picture" into paragraphs (#157). The escape hatch for
599 /// image-extraction workflows and for charts the heuristic might still
600 /// misjudge on scanned pages (#173).
601 pub fn no_text_panels(mut self, disable: bool) -> Self {
602 self.no_text_panels = disable;
603 self
604 }
605
606 /// Infer PDF/image section-header levels after assembly (#302, docling's
607 /// `HeadingHierarchyModel` with its default options): the PDF outline
608 /// (bookmarks) is authoritative, legal/outline numbering covers headings
609 /// without a bookmark match, and font size/weight/slant/case rank the
610 /// rest. Off by default (docling parity): every detected heading then
611 /// keeps the flat level the assembler emits. Non-PDF/image formats
612 /// ignore it (their backends carry real heading levels already).
613 pub fn heading_hierarchy(mut self, enable: bool) -> Self {
614 self.heading_hierarchy = enable;
615 self
616 }
617
618 /// Skip layout detection, OCR, and TableFormer entirely for PDF/image/METS
619 /// sources — no model load, no inference of any kind.
620 ///
621 /// Off by default. When enabled, the PDF's embedded text cells are grouped by
622 /// line and emitted as plain paragraphs in reading order: no headings, lists,
623 /// tables, code blocks, or pictures, since that structure comes from the
624 /// layout model. The fastest possible PDF path, but pages with no embedded
625 /// text layer (scanned/image-only PDFs) yield no text at all — convert those
626 /// without this flag. Implies [`no_table_former`](Self::no_table_former).
627 pub fn no_ocr(mut self, disable: bool) -> Self {
628 self.no_ocr = disable;
629 self
630 }
631
632 /// Never run OCR, but keep layout detection and TableFormer — docling's
633 /// independent `do_ocr=False` (#244), the counterpart of
634 /// [`no_table_former`](Self::no_table_former). Unlike
635 /// [`no_ocr`](Self::no_ocr) (the skip-everything fast path), structured
636 /// output — headings, tables, pictures, reading order — is preserved; only
637 /// text that exists solely as pixels is lost (scanned pages come back with
638 /// empty regions, and the speculative OCR of large embedded images never
639 /// runs). The OCR model is never loaded, and independently of this flag a
640 /// *missing* OCR model now degrades to the same behavior with a warning
641 /// instead of failing the conversion. SVG inputs route to direct
642 /// `<text>` extraction (their text is native — skipping OCR must not lose
643 /// it), like `no_ocr`.
644 pub fn skip_ocr(mut self, disable: bool) -> Self {
645 self.skip_ocr = disable;
646 self
647 }
648
649 /// OCR every PDF page from its rendered image even when the page carries
650 /// an embedded text layer — docling's `force_full_page_ocr`. The escape
651 /// hatch for text layers that exist but lie (broken encodings, subset
652 /// fonts with garbage mappings, scanned forms with a few typed-in
653 /// fields). Off by default; ignored when [`no_ocr`](Self::no_ocr) is set,
654 /// mirroring docling, where it is a sub-option of `do_ocr`. Applies to
655 /// PDFs only — standalone images are always OCR'd.
656 pub fn force_full_page_ocr(mut self, force: bool) -> Self {
657 self.force_full_page_ocr = force;
658 self
659 }
660
661 /// Which document regions feed the OCR — docling's `OcrMode` (#254):
662 /// `default`, `full_page`, `layout_regions`, or
663 /// `pdf_aware_layout_regions`. The default is the text-layer-aware
664 /// behavior (docling's `pdf_aware_layout_regions`);
665 /// `full_page`/`layout_regions` discard the text layer like
666 /// [`force_full_page_ocr`](Self::force_full_page_ocr) — see
667 /// [`docling_pdf::OcrMode`] for the mapping. An unknown value warns at
668 /// conversion time and uses the default. PDF/image ML pipeline only.
669 pub fn ocr_mode(mut self, mode: impl Into<String>) -> Self {
670 self.ocr_mode = Some(mode.into());
671 self
672 }
673
674 /// Which OCR engine recognizes text on scanned pages (#460): `"ppocr"`
675 /// (the default — the built-in PP-OCRv3 recognizer, the conformance
676 /// engine) or `"tesseract"` (the system `tesseract` binary, docling's
677 /// `TesseractCliOcrOptions`; needs `tesseract-ocr` with a language pack
678 /// installed — `DOCLING_TESSERACT` names the binary,
679 /// `DOCLING_RS_TESSERACT_PSM` its page segmentation mode,
680 /// `DOCLING_RS_TESSDATA_DIR` / `TESSDATA_PREFIX` its data). Both read
681 /// the same layout-region crops; [`ocr_lang`](Self::ocr_lang) is the
682 /// engine's language. An unknown value warns at conversion time and uses
683 /// the default (`DOCLING_RS_OCR_ENGINE`, else PP-OCR). Formats that never
684 /// OCR ignore it.
685 pub fn ocr_engine(mut self, engine: impl Into<String>) -> Self {
686 self.ocr_engine = Some(engine.into());
687 self
688 }
689
690 /// OCR render scale in pixels per PDF point — docling's
691 /// `OcrOptions.scale` (#254; docling's default 3 = 216 dpi). Unset feeds
692 /// the recognizer the pipeline's own 2.0 px/pt page render; a different
693 /// value resamples that render for the OCR input only, leaving layout and
694 /// TableFormer pixels untouched. Non-positive values warn at conversion
695 /// time and are ignored. PDF/image ML pipeline only.
696 pub fn ocr_scale(mut self, scale: f32) -> Self {
697 self.ocr_scale = Some(scale);
698 self
699 }
700
701 /// Classify each detected picture with the DocumentFigureClassifier model
702 /// (docling's `do_picture_classification`). Off by default.
703 ///
704 /// The full 26-class prediction distribution (bar_chart, logo, signature,
705 /// …) lands on the picture item and is serialized into the docling JSON as
706 /// the `classification` annotation plus the `meta.classification` field.
707 /// Markdown output is unaffected. Needs `.models/picture_classifier.onnx`
708 /// (fetched by `scripts/install/download_dependencies.sh`); a missing
709 /// model warns once and skips classification.
710 pub fn do_picture_classification(mut self, enable: bool) -> Self {
711 self.enrich.picture_classification = enable;
712 self
713 }
714
715 /// Rewrite detected code blocks with the CodeFormulaV2 VLM (docling's
716 /// `do_code_enrichment`). Off by default.
717 ///
718 /// The model re-reads the code crop at ~120 dpi, emits the clean source
719 /// text (line breaks included) and identifies the language, which lands in
720 /// the JSON `code_language` field. Needs the `.models/code_formula/` graphs
721 /// (fetched by `scripts/install/download_dependencies.sh`); a missing
722 /// model warns once and leaves the block as extracted.
723 pub fn do_code_enrichment(mut self, enable: bool) -> Self {
724 self.enrich.code = enable;
725 self
726 }
727
728 /// Decode display formulas to LaTeX with the CodeFormulaV2 VLM (docling's
729 /// `do_formula_enrichment`). Off by default.
730 ///
731 /// An enriched formula renders as `$$latex$$` in Markdown and as a
732 /// `formula` text item in the JSON, replacing the
733 /// `<!-- formula-not-decoded -->` placeholder. Same model artifacts as
734 /// [`do_code_enrichment`](Self::do_code_enrichment).
735 pub fn do_formula_enrichment(mut self, enable: bool) -> Self {
736 self.enrich.formula = enable;
737 self
738 }
739
740 /// Pre-render HTML-routing input in a headless browser before parsing.
741 ///
742 /// Off by default. When enabled, HTML sources — and MHTML/EPUB, which
743 /// assemble HTML from their archives — are loaded in the system Chromium
744 /// (driven from Rust over the DevTools protocol — no Node/Playwright) so the
745 /// CSS cascade is resolved: elements the browser computes as `display:none`
746 /// (e.g. a stylesheet-collapsed nav menu) are removed before the normal HTML
747 /// backend runs. This is the one behaviour a pure-Rust parse can't reproduce;
748 /// everything else (structure, tables, KVP, formatting) is still handled in
749 /// Rust on the cleaned HTML.
750 ///
751 /// Requires the crate's `web-browser` Cargo feature; without it, converting
752 /// an HTML source with this enabled returns [`ConversionError::Browser`].
753 pub fn use_web_browser(mut self, enable: bool) -> Self {
754 self.use_web_browser = enable;
755 self
756 }
757
758 /// Return `html` unchanged, or — when [`use_web_browser`](Self::use_web_browser)
759 /// is on — its headless-browser-cleaned form (computed-hidden elements
760 /// removed). Borrows in the common (disabled) case; only allocates when the
761 /// browser actually runs.
762 fn maybe_prerender<'a>(
763 &self,
764 html: &'a str,
765 ) -> Result<std::borrow::Cow<'a, str>, ConversionError> {
766 crate::backend::maybe_prerender_html(html, self.use_web_browser)
767 }
768
769 /// Convert a source document to Markdown **incrementally**, returning an
770 /// iterator of Markdown chunks (with picture placeholders).
771 ///
772 /// Concatenating every `Ok` chunk reproduces
773 /// [`convert`](Self::convert)`(...).document.export_to_markdown()`
774 /// byte-for-byte. The win is for PDF, whose pages are processed in parallel:
775 /// each page's Markdown is emitted in document order as soon as it is ready, so
776 /// output starts before the whole document is converted. Other formats build
777 /// their document up front and stream it through the same interface.
778 ///
779 /// Streaming is Markdown-only — JSON needs the whole node tree, so there is no
780 /// streaming JSON. The conversion runs on a background thread; dropping the
781 /// returned [`MarkdownStream`] cancels it.
782 #[cfg(feature = "pdf")]
783 pub fn convert_streaming(
784 &self,
785 source: SourceDocument,
786 ) -> Result<MarkdownStream, ConversionError> {
787 self.convert_streaming_images(source, ImageMode::Placeholder)
788 }
789
790 /// Like [`convert_streaming`](Self::convert_streaming) but with an explicit
791 /// picture [`ImageMode`].
792 ///
793 /// [`ImageMode::Referenced`] streams too (issue #80): each page's images
794 /// are written to [`artifacts_dir`](Self::artifacts_dir) *as the page's
795 /// Markdown is emitted* and dropped from memory, so an image-heavy PDF
796 /// holds ~one page of images at a time instead of all of them until
797 /// export. The chunks and files match the buffered
798 /// `export_to_markdown_with_images(ImageMode::Referenced, ..)` output.
799 #[cfg(feature = "pdf")]
800 pub fn convert_streaming_images(
801 &self,
802 source: SourceDocument,
803 image_mode: ImageMode,
804 ) -> Result<MarkdownStream, ConversionError> {
805 if let Some(allowed) = &self.allowed_formats {
806 if !allowed.contains(&source.format) {
807 return Err(ConversionError::UnsupportedFormat(source.format));
808 }
809 }
810 let source = self.with_encoding(source);
811 Ok(crate::stream::spawn(self.clone(), source, image_mode))
812 }
813
814 /// Whether the heading-hierarchy stage (#302) is enabled — the streaming
815 /// front-end buffers PDF conversions when it is (the stage needs the whole
816 /// assembled document, and streamed output must stay byte-identical to
817 /// buffered output).
818 #[cfg(feature = "pdf")]
819 pub(crate) fn heading_hierarchy_enabled(&self) -> bool {
820 self.heading_hierarchy
821 }
822
823 /// Streaming internals ([`crate::stream`]) read the producer's settings
824 /// off the converter clone they receive.
825 #[cfg(feature = "pdf")]
826 pub(crate) fn stream_settings(&self) -> crate::stream::StreamSettings {
827 crate::stream::StreamSettings {
828 strict: self.strict,
829 no_table_former: self.no_table_former,
830 no_text_panels: self.no_text_panels,
831 no_ocr: self.no_ocr,
832 skip_ocr: self.skip_ocr,
833 force_full_page_ocr: self.force_full_page_ocr,
834 enrich: self.enrich,
835 page_range: self.page_range,
836 ocr_lang: self.ocr_lang_choice(),
837 ocr_engine: self.ocr_engine_choice(),
838 tesseract_lang: self.tesseract_lang_choice(),
839 ocr_mode: self.ocr_mode_choice(),
840 ocr_scale: self.ocr_scale_choice(),
841 artifacts_dir: self.artifacts_dir.clone(),
842 page_break_placeholder: self.page_break_placeholder.clone(),
843 }
844 }
845
846 /// Convert a single source document.
847 pub fn convert(&self, source: SourceDocument) -> Result<ConversionResult, ConversionError> {
848 if let Some(allowed) = &self.allowed_formats {
849 if !allowed.contains(&source.format) {
850 return Err(ConversionError::UnsupportedFormat(source.format));
851 }
852 }
853 let source = self.with_encoding(source);
854
855 let mut document = match source.format {
856 // A legacy APS (Automated Patent System) plain-text patent (`PATN`
857 // first record) is reconstructed verbatim, mirroring docling.
858 InputFormat::Md if crate::backend::uspto::looks_like_aps(&source.text()?) => {
859 crate::backend::uspto::convert_aps(&source)?
860 }
861 // A text/Markdown-typed file that is actually an XML document (e.g. a
862 // JATS article saved with a `.txt` extension) routes to the XML
863 // backends by content, mirroring docling's content-based detection.
864 InputFormat::Md if looks_like_xml(&source.text()?) => match sniff_xml(&source.bytes) {
865 InputFormat::XmlUspto => UsptoBackend.convert(&source)?,
866 InputFormat::XmlXbrl => {
867 crate::backend::xbrl::convert_xbrl(&source, self.xbrl_taxonomy.as_deref())?
868 }
869 // docling's format detection reads an XML-looking `.txt` as
870 // `application/xml` and, when its DOCTYPE names a JATS DTD
871 // (`JATS-journalpublishing…` / `JATS-archive…`), converts it
872 // with the JATS backend like a real `.nxml`; any other XML
873 // saved as `.txt` is reconstructed generically
874 // (element-by-element).
875 _ if has_jats_doctype(&source.text()?) => JatsBackend {
876 fetch_images: self.fetch_images,
877 }
878 .convert(&source)?,
879 _ => crate::backend::jats::convert_generic(&source)?,
880 },
881 // DeepSeek-OCR annotated Markdown (VLM token format) is detected by
882 // its `<|ref|>…[[bbox]]` annotations and parsed separately.
883 InputFormat::Md if is_deepseek_markdown(&source.text()?) => {
884 DeepSeekBackend.convert(&source)?
885 }
886 InputFormat::Md => MarkdownBackend {
887 strict: self.strict,
888 }
889 .convert(&source)?,
890 InputFormat::Csv => CsvBackend.convert(&source)?,
891 InputFormat::Html => {
892 // Optionally resolve the CSS cascade in a headless browser first
893 // (strips computed-hidden elements); everything else stays in the
894 // Rust HTML backend, which runs on the cleaned HTML.
895 // Bytes → text through docling's BeautifulSoup decoding order
896 // (#371): BOM, declared charset, UTF-8, windows-1252 — so a
897 // legacy windows-1252 page converts instead of failing the
898 // UTF-8 check every other text backend applies.
899 let decoded = crate::backend::decode_html_bytes(&source.bytes);
900 let html = self.maybe_prerender(&decoded)?;
901 if self.fetch_images {
902 let resolver = crate::backend::FsImageResolver::new(
903 source.base_dir().map(|p| p.to_path_buf()),
904 source.base_url.clone(),
905 );
906 crate::backend::convert_html(&source.name, &html, &resolver)
907 } else {
908 crate::backend::convert_html(&source.name, &html, &crate::backend::NoFetch)
909 }
910 }
911 InputFormat::Asciidoc => AsciiDocBackend {
912 fetch_images: self.fetch_images,
913 }
914 .convert(&source)?,
915 InputFormat::Xlsx => XlsxBackend {
916 skip_empty: self.skip_empty_cells,
917 }
918 .convert(&source)?,
919 InputFormat::Pptx => PptxBackend.convert(&source)?,
920 // RTF (#209): a docling.rs extension — docling reaches RTF only via
921 // LibreOffice; here it parses natively (hand-rolled tokenizer).
922 InputFormat::Rtf => RtfBackend.convert(&source)?,
923 InputFormat::Visio => VisioBackend.convert(&source)?,
924 // AbiWord (#216): docling.rs extension, native AWML parse.
925 InputFormat::Abiword => AbwBackend.convert(&source)?,
926 // WordPerfect 5.x/6.x+ (#216): docling.rs extension, native parse
927 // of the ÿWPC function-code stream.
928 InputFormat::WordPerfect => WpdBackend.convert(&source)?,
929 // Microsoft Works word processor (#216): docling.rs extension,
930 // native parse after libwps.
931 InputFormat::Works => WpsBackend.convert(&source)?,
932 // StarOffice 5 binaries (#215): docling.rs extension, native CFB
933 // parse (docling would go through LibreOffice).
934 InputFormat::StarOffice5 => StarOffice5Backend.convert(&source)?,
935 // DjVu (#434): docling.rs extension, pure-Rust decode (`djvu-rs`).
936 // The hidden text layer is the default; a scan-only DjVu falls back
937 // to rasterize + OCR when the ML pipeline is built.
938 InputFormat::Djvu => self.convert_djvu(&source)?,
939 // DIF/SYLK/dBase (#216): docling.rs extensions, one content-sniffing
940 // backend for the three table relics.
941 InputFormat::Dbf | InputFormat::Dif | InputFormat::Sylk => {
942 InterchangeBackend.convert(&source)?
943 }
944 // Lotus/Quattro/Works record streams (#216): one BOF-sniffing
945 // backend for the whole DOS-era family.
946 InputFormat::Lotus => LotusBackend.convert(&source)?,
947 // Quattro Pro (#216): docling.rs extension, native parse after
948 // libwps (DOS/Windows record streams, QPW OLE zones).
949 InputFormat::QuattroPro => QuattroBackend.convert(&source)?,
950 InputFormat::Docx => DocxBackend.convert(&source)?,
951 // Legacy binary Office (issue #127): parsed natively — docling
952 // proper converts these through LibreOffice first (PR #3804).
953 InputFormat::Xls => XlsBackend {
954 skip_empty: self.skip_empty_cells,
955 }
956 .convert(&source)?,
957 InputFormat::Ppt => PptBackend.convert(&source)?,
958 InputFormat::Doc => DocBackend.convert(&source)?,
959 InputFormat::Vtt => WebVttBackend.convert(&source)?,
960 InputFormat::Ebcdic => EbcdicBackend {
961 layout: self.ebcdic_layout.clone(),
962 }
963 .convert(&source)?,
964 InputFormat::Email => EmailBackend {
965 list_attachments: self.list_attachments,
966 }
967 .convert(&source)?,
968 InputFormat::Mhtml => MhtmlBackend {
969 fetch_images: self.fetch_images,
970 use_web_browser: self.use_web_browser,
971 }
972 .convert(&source)?,
973 InputFormat::Epub => EpubBackend {
974 fetch_images: self.fetch_images,
975 use_web_browser: self.use_web_browser,
976 }
977 .convert(&source)?,
978 InputFormat::JsonDocling => DoclingJsonBackend.convert(&source)?,
979 InputFormat::Latex => LatexBackend.convert(&source)?,
980 // A bare `.xml` defaults to XmlJats; sniff the content to route to the
981 // right XML backend (docling distinguishes by DOCTYPE / root element).
982 InputFormat::XmlJats | InputFormat::XmlUspto | InputFormat::XmlXbrl => {
983 match sniff_xml(&source.bytes) {
984 InputFormat::XmlUspto => UsptoBackend.convert(&source)?,
985 InputFormat::XmlXbrl => {
986 crate::backend::xbrl::convert_xbrl(&source, self.xbrl_taxonomy.as_deref())?
987 }
988 _ => JatsBackend {
989 fetch_images: self.fetch_images,
990 }
991 .convert(&source)?,
992 }
993 }
994 InputFormat::Odt | InputFormat::Ods | InputFormat::Odp => {
995 crate::backend::convert_odf(&source, self.fetch_images)?
996 }
997 // DocLang back in: bare XML (`.dclg`/`.dclg.xml`) or the OPC
998 // archive `--to dclx` writes.
999 InputFormat::XmlDoclang | InputFormat::Dclx => {
1000 crate::backend::DoclangBackend.convert(&source)?
1001 }
1002 // Raw DocTags (VLM token markup, #152): the tolerant docling-core
1003 // parser — never fails, best-effort document out.
1004 InputFormat::DocTags => {
1005 let mut doc = docling_core::doctags::parse(&source.text()?);
1006 doc.name = source.name.clone();
1007 doc
1008 }
1009 #[cfg(feature = "pdf")]
1010 InputFormat::Pdf => self
1011 .ml_pipeline()
1012 .map(|p| {
1013 p.force_full_page_ocr(self.force_full_page_ocr)
1014 .pages(self.page_range)
1015 })
1016 .and_then(|mut p| p.convert(&source.bytes, None, &source.name))
1017 .map_err(|e| ConversionError::with_source("pdf", e))?,
1018 // SVG (#212), the ML route: rasterize (resvg, white-backed PNG at
1019 // ~2048px long side) and ride the image pipeline. `--no-ocr` short-
1020 // circuits to direct <text> extraction instead — the SVG carries
1021 // its text natively, so skipping OCR must not mean losing it.
1022 #[cfg(feature = "pdf")]
1023 InputFormat::Svg if !self.no_ocr && !self.skip_ocr => {
1024 let png = crate::backend::svg::rasterize_png(&source.bytes)?;
1025 self.ml_pipeline()
1026 .and_then(|mut p| p.convert_image(&png, &source.name))
1027 .map_err(|e| ConversionError::with_source("svg", e))?
1028 }
1029 // SVG without the ML pipeline (pdf-text / wasm builds) or with
1030 // --no-ocr / --skip-ocr: pure-Rust <text> extraction, flat
1031 // paragraphs in reading order (the pdf / pdf-text split, applied
1032 // to SVG) — the SVG carries its text natively, so skipping OCR
1033 // must not mean losing it.
1034 InputFormat::Svg => crate::backend::SvgBackend.convert(&source)?,
1035 // Apple iWork (#213): pure-Rust IWA text extraction, all builds.
1036 InputFormat::Pages | InputFormat::Numbers | InputFormat::Keynote => {
1037 crate::backend::IworkBackend.convert(&source)?
1038 }
1039 #[cfg(feature = "pdf")]
1040 InputFormat::Image => self
1041 .ml_pipeline()
1042 .and_then(|mut p| p.convert_image(&source.bytes, &source.name))
1043 .map_err(|e| ConversionError::with_source("image", e))?,
1044 #[cfg(feature = "pdf")]
1045 InputFormat::MetsGbs => self
1046 .ml_pipeline()
1047 .and_then(|mut p| {
1048 docling_pdf::convert_mets_gbs_with_pipeline(&source.bytes, &source.name, &mut p)
1049 })
1050 .map_err(|e| ConversionError::with_source("mets-gbs", e))?,
1051 // Audio → Whisper ASR (symphonia decode + ONNX inference); each
1052 // transcribed segment becomes a `[time: start-end] text` paragraph.
1053 #[cfg(feature = "asr")]
1054 InputFormat::Audio => docling_asr::convert_audio_with_options(
1055 &source.bytes,
1056 &source.name,
1057 self.asr_model.as_deref(),
1058 self.asr_lang.as_deref(),
1059 )
1060 .map_err(|e| ConversionError::with_source(source.format.as_str(), e))?,
1061 // Video (#138): the audio track transcribes through the same ASR
1062 // path (Phase 1), and — when the ffmpeg binary is available —
1063 // sampled frames interleave with the transcript as timestamped
1064 // pictures (Phase 2). Without ffmpeg: transcript only.
1065 #[cfg(feature = "asr")]
1066 InputFormat::Video => crate::video::convert_video(
1067 &source.bytes,
1068 &source.name,
1069 self.asr_model.as_deref(),
1070 self.asr_lang.as_deref(),
1071 self.video_frames.unwrap_or(DEFAULT_VIDEO_FRAMES),
1072 )
1073 .map_err(|e| ConversionError::with_source(source.format.as_str(), e))?,
1074 // Without the full ML pipeline, `pdf-text` still converts a PDF's
1075 // embedded text layer (pure Rust — the wasm32 path), equivalent to
1076 // `--no-ocr`: flat paragraphs, no headings/tables/pictures. A
1077 // scanned PDF has no text layer, so an empty document means "this
1078 // needs OCR" — say so instead of returning nothing.
1079 #[cfg(all(feature = "pdf-text", not(feature = "pdf")))]
1080 InputFormat::Pdf => {
1081 let doc = docling_pdf::convert_text_layer_pages(
1082 &source.bytes,
1083 &source.name,
1084 self.page_range,
1085 )
1086 .map_err(|e| ConversionError::with_source("pdf", e))?;
1087 if doc.nodes.is_empty() {
1088 return Err(ConversionError::Parse(
1089 "PDF has no embedded text layer (scanned/image-only?); OCR needs a \
1090 build with the `pdf` feature"
1091 .into(),
1092 ));
1093 }
1094 doc
1095 }
1096 // Compiled without the ML pipelines: the formats stay detectable,
1097 // but converting them needs a build with the matching feature.
1098 #[cfg(not(any(feature = "pdf", feature = "pdf-text")))]
1099 InputFormat::Pdf => {
1100 return Err(ConversionError::Parse(
1101 "Pdf conversion is not compiled in (rebuild with the `pdf` feature, or \
1102 `pdf-text` for text-layer-only extraction)"
1103 .into(),
1104 ))
1105 }
1106 #[cfg(not(feature = "pdf"))]
1107 InputFormat::Image | InputFormat::MetsGbs => {
1108 return Err(ConversionError::Parse(format!(
1109 "{:?} conversion is not compiled in (rebuild with the `pdf` feature)",
1110 source.format
1111 )))
1112 }
1113 #[cfg(not(feature = "asr"))]
1114 InputFormat::Audio | InputFormat::Video => {
1115 return Err(ConversionError::Parse(format!(
1116 "{} conversion is not compiled in (rebuild with the `asr` feature)",
1117 source.format.as_str()
1118 )))
1119 }
1120 };
1121 // Carry the mode so `result.document.export_to_markdown()` reflects it.
1122 document.strict_markdown = self.strict;
1123 // Compact tables (#271) is additive: the PDF backend already turns it
1124 // on for its own corpus; never turn it back off here.
1125 if self.compact_tables {
1126 document.compact_tables = true;
1127 }
1128 // Page-break placeholder: a serializer knob like `strict`, carried on
1129 // the document so every Markdown export of it agrees.
1130 if self.page_break_placeholder.is_some() {
1131 document.page_break_placeholder = self.page_break_placeholder.clone();
1132 }
1133 // First-class cells for every table (#240): backends with page
1134 // geometry (the PDF TableFormer paths) set them; everything else —
1135 // declarative tables included — derives them from the grid plus the
1136 // structure overlay (real spans for DOCX/XLSX merges, HTML `th`
1137 // headers, ODF covered cells; 1×1 records otherwise), so the repair
1138 // API and the JSON `table_cells` are populated uniformly.
1139 for table in document.tables_mut() {
1140 if table.cells.is_none() {
1141 table.cells = Some(table.derive_cells());
1142 }
1143 }
1144
1145 Ok(ConversionResult {
1146 document,
1147 status: ConversionStatus::Success,
1148 input_name: source.name,
1149 format: source.format,
1150 })
1151 }
1152}
1153
1154#[cfg(test)]
1155mod tests {
1156 use super::*;
1157
1158 /// docling reads an XML-looking `.txt` as `application/xml` and converts
1159 /// it with the JATS backend when its DOCTYPE names a JATS DTD; other XML
1160 /// under `.txt` stays the generic element-by-element reconstruction.
1161 #[test]
1162 fn jats_doctype_text_file_uses_the_jats_backend() {
1163 let jats = "<!DOCTYPE article PUBLIC \"-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.2 20190208//EN\" \"JATS-archivearticle1.dtd\">\n<article><front><article-meta><title-group><article-title>T</article-title></title-group></article-meta></front><body><sec><title>S</title><p>Body.</p></sec></body></article>";
1164 let src = SourceDocument::from_bytes("a.txt", InputFormat::Md, jats.as_bytes().to_vec());
1165 let doc = DocumentConverter::new().convert(src).unwrap().document;
1166 assert!(doc.tree.is_some(), "JATS tree expected");
1167 assert_eq!(doc.export_to_markdown().trim(), "# T\n\n## S\n\nBody.");
1168 let other = "<?xml version=\"1.0\"?>\n<article><body><sec><title>S</title><p>Body.</p></sec></body></article>";
1169 let src = SourceDocument::from_bytes("b.txt", InputFormat::Md, other.as_bytes().to_vec());
1170 let doc = DocumentConverter::new().convert(src).unwrap().document;
1171 assert!(doc.tree.is_none(), "generic XML path expected");
1172 }
1173
1174 /// docling's `TextBackendOptions.encoding`: the converter's `encoding`
1175 /// decodes text inputs as named (a source's own setting wins), and an
1176 /// undecodable byte or unknown label is an error, not a guess.
1177 #[test]
1178 fn encoding_option_decodes_text_inputs() {
1179 let sjis = b"# \x93\xfa\x96\x7b\n".to_vec();
1180 let md = |c: &DocumentConverter, s: SourceDocument| {
1181 c.convert(s).map(|r| r.document.export_to_markdown())
1182 };
1183 let conv = DocumentConverter::new().encoding(Some("shift_jis".into()));
1184 let src = || SourceDocument::from_bytes("doc", InputFormat::Md, sjis.clone());
1185 assert_eq!(md(&conv, src()).unwrap().trim(), "# 日本");
1186 // Detection reads the same bytes as windows-1252.
1187 assert_ne!(
1188 md(&DocumentConverter::new(), src()).unwrap().trim(),
1189 "# 日本"
1190 );
1191 // The source's own encoding takes precedence over the converter's.
1192 let own = src().with_encoding(Some("shift_jis".into()));
1193 let latin = DocumentConverter::new().encoding(Some("latin1".into()));
1194 assert_eq!(md(&latin, own).unwrap().trim(), "# 日本");
1195 assert!(md(
1196 &DocumentConverter::new().encoding(Some("utf-8".into())),
1197 src()
1198 )
1199 .is_err());
1200 assert!(md(
1201 &DocumentConverter::new().encoding(Some("nope-1".into())),
1202 src()
1203 )
1204 .is_err());
1205 }
1206
1207 #[test]
1208 fn end_to_end_markdown() {
1209 let src =
1210 SourceDocument::from_bytes("doc", InputFormat::Md, b"# Hello\n\nWorld.\n".to_vec());
1211 let result = DocumentConverter::new().convert(src).unwrap();
1212 assert_eq!(result.status, ConversionStatus::Success);
1213 assert_eq!(result.document.export_to_markdown(), "# Hello\n\nWorld.\n");
1214 }
1215
1216 #[test]
1217 fn doctags_input_converts() {
1218 // Raw DocTags markup (#152) — the VLM token stream — as a first-class
1219 // input format (.doctags/.dt), through the tolerant docling-core
1220 // parser.
1221 let markup = b"<doctag><section_header_level_1><loc_1><loc_2><loc_3><loc_4>Intro</section_header_level_1><text>Body.</text></doctag>"
1222 .to_vec();
1223 let src = SourceDocument::from_bytes("page.doctags", InputFormat::DocTags, markup);
1224 let result = DocumentConverter::new().convert(src).unwrap();
1225 let md = result.document.export_to_markdown();
1226 assert!(md.contains("## Intro"), "{md}");
1227 assert!(md.contains("Body."), "{md}");
1228 }
1229
1230 #[test]
1231 fn doclang_xml_round_trips() {
1232 // Every input format now has a backend; DocLang XML reads back in and
1233 // re-exports as Markdown.
1234 let xml = b"<doclang version=\"0.7\">\n <heading>Title</heading>\n \
1235 <text>Hello <bold>world</bold></text>\n</doclang>"
1236 .to_vec();
1237 let src = SourceDocument::from_bytes("doc.dclg", InputFormat::XmlDoclang, xml);
1238 let result = DocumentConverter::new().convert(src).unwrap();
1239 let md = result.document.export_to_markdown();
1240 assert!(md.contains("# Title"), "{md}");
1241 assert!(md.contains("**world**"), "{md}");
1242 }
1243
1244 #[test]
1245 fn sniffs_uspto_doctype_case_insensitively() {
1246 // docling PR #3801: Grant Full Text v2.5 files were missed when the
1247 // DOCTYPE casing differed.
1248 for head in [
1249 "<?xml version=\"1.0\"?><!DOCTYPE PATDOC SYSTEM \"ST32-US-Grant-025xml.dtd\"><PATDOC/>",
1250 "<?xml version=\"1.0\"?><!DOCTYPE patdoc SYSTEM \"st32-us-grant-025xml.dtd\"><patdoc/>",
1251 "<?xml version=\"1.0\"?><US-PATENT-GRANT-V4/>",
1252 ] {
1253 assert_eq!(
1254 super::sniff_xml(head.as_bytes()),
1255 InputFormat::XmlUspto,
1256 "head: {head}"
1257 );
1258 }
1259 }
1260}