Skip to main content

docling_pdf/
ocr.rs

1//! OCR for scanned pages, via the PP-OCRv3 recognition model (CRNN/SVTR) run
2//! with `ort`. The layout model already locates text regions on the page image
3//! (it works without a text layer), so OCR only needs *recognition*: each text
4//! region is cropped, split into lines by horizontal projection, and each line
5//! is recognised and decoded with CTC — producing [`TextCell`]s the normal
6//! layout assembly then consumes. This avoids a separate text-detection model.
7
8use image::RgbImage;
9use ort::session::Session;
10use ort::value::Tensor;
11
12use crate::layout::Region;
13// The ONNX-free half (line prep, batching, CTC decode) lives in `ocr_prep`
14// so the wasm build shares it verbatim (#79 phase 2).
15use crate::ocr_prep::{
16    batch_input, decode_row_scored, dict_chars, prep_region_lines, prep_table_words, width_batches,
17    PrepLine, REC_HEIGHT,
18};
19use crate::pdfium_backend::TextCell;
20
21pub struct OcrModel {
22    /// Single-threaded recognition sessions, one per parallel lane (see
23    /// [`Self::load_with`]); lines are dealt across them by batch index.
24    recs: Vec<Session>,
25    /// CTC classes: index 0 = blank, 1..=6623 = dictionary, 6624 = space.
26    chars: Vec<String>,
27}
28
29/// OCR recognition language: which PP-OCRv3 model + dictionary pair runs.
30///
31/// The default is **English** (`.models/ocr_rec_en.onnx` + `.models/en_dict.txt`):
32/// the multilingual `ch_` model reads Latin scripts with badly degraded word
33/// spacing (glued words on ordinary English scans), which is the common
34/// real-world case. `Ch` selects the `ch_` pair (`.models/ocr_rec.onnx` +
35/// `.models/ppocr_keys_v1.txt`) — that is what upstream docling conformance is
36/// measured with, and `scripts/conformance/pdf_*.sh` pin it explicitly (by
37/// path, which wins over this selector).
38#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
39pub enum OcrLang {
40    /// en_PP-OCRv3 — English-only, proper Latin word spacing.
41    #[default]
42    En,
43    /// ch_PP-OCRv3 — multilingual; the docling-conformance model.
44    Ch,
45}
46
47impl OcrLang {
48    /// Parse a user-supplied language id (#388): the engine's own codes
49    /// (`en`, `ch`) and BCP-47 tags naming a language one of the two
50    /// recognizers reads, trimmed and case-insensitive. `None` for anything
51    /// else — callers surface their own error/warning.
52    ///
53    /// docling canonicalizes OCR languages across its engines (docling#4075):
54    /// a bare value is the engine's native code, an `iso:`-prefixed value a
55    /// BCP-47 tag reduced to a language-script pair with the region dropped
56    /// (`zh-CN` and `zh-Hans` are the same recognizer, `en-GB` is `en`), and
57    /// the RapidOCR adapter maps `en` → its `en` model and `zh-Hans` → `ch`.
58    /// With only those two PP-OCRv3 pairs on board there is no ambiguity, so
59    /// the prefix is optional here: `en-US`, `eng`, `zh`, `zh-Hans`, `iso:zh-CN`
60    /// all resolve without a warning. Accepted primary subtags: English as
61    /// `en` / ISO 639-2/3 `eng` / docling's legacy `english`; Chinese as the
62    /// engine code `ch` (and RapidOCR's `chinese_cht`), `zh` / `zho` / `chi` /
63    /// `cmn` / legacy `chinese` / EasyOCR's `ch_sim` / `ch_tra`. Script,
64    /// region and variant subtags (`-Hans`, `-Hant`, `-CN`, `-TW`, `_US`) are
65    /// ignored: a traditional-script request (`zh-Hant`, `zh-TW`) gets the
66    /// multilingual `ch` recognizer too, the closest model shipped — upstream
67    /// would pick RapidOCR's separate `chinese_cht`, which this engine does
68    /// not carry. Genuinely unsupported languages (`de`, `fr`, `ja`, …) parse
69    /// to `None` and keep warning.
70    pub fn parse(s: &str) -> Option<Self> {
71        let token = s.trim().to_ascii_lowercase();
72        let tag = token.strip_prefix("iso:").unwrap_or(&token).trim();
73        let primary = tag.split(['-', '_']).next().unwrap_or_default();
74        match primary {
75            "en" | "eng" | "english" => Some(Self::En),
76            "ch" | "chinese_cht" | "zh" | "zho" | "chi" | "cmn" | "chinese" | "ch_sim"
77            | "ch_tra" => Some(Self::Ch),
78            _ => None,
79        }
80    }
81
82    /// The process-level choice from `DOCLING_RS_OCR_LANG` (empty/unset → the
83    /// English default; unknown values warn and use English).
84    pub fn from_env() -> Self {
85        let Some(raw) = docling_core::env::nonempty("DOCLING_RS_OCR_LANG") else {
86            return Self::default();
87        };
88        Self::parse(&raw).unwrap_or_else(|| {
89            eprintln!(
90                "docling-pdf: DOCLING_RS_OCR_LANG={raw:?} names no language the en/ch \
91                 recognizers read ({}); using en",
92                Self::ACCEPTED
93            );
94            Self::default()
95        })
96    }
97
98    /// The accepted spellings, for error messages and docs.
99    pub const ACCEPTED: &'static str =
100        "en | ch, or a BCP-47 tag for English or Chinese such as en-US, eng, zh, zh-Hans, zh-TW";
101}
102
103/// Which document regions feed the OCR — docling 2.116's `OcrMode` (#254,
104/// upstream docling#3710). Upstream restructured its pipeline so OCR runs
105/// *after* layout, on layout regions filtered by the PDF text layer — the
106/// architecture this port has always had — and named the strategies:
107///
108/// - `PdfAwareLayoutRegions` (upstream's **default**): OCR only layout regions
109///   the embedded text layer can't cover. Exactly the standard path here —
110///   scanned pages OCR their regions, digital pages OCR only text-less bitmap
111///   areas.
112/// - `FullPage` / `LayoutRegions`: ignore the PDF text layer and OCR
113///   everything. Both map onto the [`force_full_page_ocr`] machinery (discard
114///   the text layer, OCR every layout region): the upstream distinction —
115///   whole-page vs per-region *detector* input — has no analogue in this
116///   engine, whose PP-OCR recognizer always consumes per-region line crops.
117///
118/// [`force_full_page_ocr`]: crate::Pipeline::force_full_page_ocr
119#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
120pub enum OcrMode {
121    /// Upstream's `default`: currently wired to `PdfAwareLayoutRegions`.
122    #[default]
123    Default,
124    /// OCR the full page, text layer ignored (docling's `full_page`; the
125    /// mode-shaped spelling of `force_full_page_ocr`).
126    FullPage,
127    /// OCR every layout region, text layer ignored (docling's
128    /// `layout_regions`).
129    LayoutRegions,
130    /// OCR layout regions the text layer can't cover (docling's
131    /// `pdf_aware_layout_regions` — the default behavior).
132    PdfAwareLayoutRegions,
133}
134
135impl OcrMode {
136    /// Parse docling's mode ids. `None` for anything else — callers surface
137    /// their own error/warning.
138    pub fn parse(s: &str) -> Option<Self> {
139        match s.trim().to_ascii_lowercase().as_str() {
140            "default" => Some(Self::Default),
141            "full_page" => Some(Self::FullPage),
142            "layout_regions" => Some(Self::LayoutRegions),
143            "pdf_aware_layout_regions" => Some(Self::PdfAwareLayoutRegions),
144            _ => None,
145        }
146    }
147
148    /// The process-level choice from `DOCLING_RS_OCR_MODE` (empty/unset → the
149    /// default; unknown values warn and use the default).
150    pub fn from_env() -> Self {
151        let Some(raw) = docling_core::env::nonempty("DOCLING_RS_OCR_MODE") else {
152            return Self::default();
153        };
154        Self::parse(&raw).unwrap_or_else(|| {
155            eprintln!(
156                "docling-pdf: DOCLING_RS_OCR_MODE={raw:?} is not \
157                 default|full_page|layout_regions|pdf_aware_layout_regions; using default"
158            );
159            Self::default()
160        })
161    }
162
163    /// Whether this mode discards the embedded text layer — the engine truth
164    /// both non-default modes reduce to.
165    pub fn forces_full_page(self) -> bool {
166        matches!(self, Self::FullPage | Self::LayoutRegions)
167    }
168}
169
170/// Which OCR engine recognizes text (#460): the built-in PP-OCRv3 recognizer
171/// (+ the RapidOCR text detector) — the default and the engine every
172/// conformance baseline is pinned against — or the system `tesseract` binary
173/// (see [`crate::tesseract`]), docling's `TesseractCliOcrOptions`
174/// counterpart. Both consume the same layout-region crops and produce the
175/// same cells; everything downstream is engine-agnostic.
176#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
177pub enum OcrEngine {
178    /// PP-OCRv3 recognition via ONNX Runtime (docling's `rapidocr` kind).
179    #[default]
180    PpOcr,
181    /// The `tesseract` CLI (docling's `tesseract` kind).
182    Tesseract,
183}
184
185impl OcrEngine {
186    /// Parse an engine id: `ppocr` (also `pp-ocr`, `rapidocr`, docling's
187    /// kind name) or `tesseract` (also `tesseract_cli`, `tesserocr`),
188    /// trimmed and case-insensitive. `None` for anything else.
189    pub fn parse(s: &str) -> Option<Self> {
190        match s.trim().to_ascii_lowercase().as_str() {
191            "ppocr" | "pp-ocr" | "pp_ocr" | "rapidocr" | "onnx" | "default" => Some(Self::PpOcr),
192            "tesseract" | "tesseract_cli" | "tesseract-cli" | "tesserocr" => Some(Self::Tesseract),
193            _ => None,
194        }
195    }
196
197    /// The process-level choice from `DOCLING_RS_OCR_ENGINE` (empty/unset →
198    /// PP-OCR; unknown values warn and use PP-OCR).
199    pub fn from_env() -> Self {
200        let Some(raw) = docling_core::env::nonempty("DOCLING_RS_OCR_ENGINE") else {
201            return Self::default();
202        };
203        Self::parse(&raw).unwrap_or_else(|| {
204            eprintln!(
205                "docling-pdf: DOCLING_RS_OCR_ENGINE={raw:?} is not ppocr|tesseract; using ppocr"
206            );
207            Self::default()
208        })
209    }
210
211    /// The accepted spellings, for error messages and docs.
212    pub const ACCEPTED: &'static str = "ppocr | tesseract";
213
214    /// Whether `raw` is an `ocr_lang` this engine can act on — what the
215    /// option surfaces validate up front: the en/ch model switch (or a
216    /// BCP-47 tag for either, [`OcrLang::parse`]) under PP-OCR; tessdata
217    /// stems and BCP-47 tags ([`crate::tesseract::lang_arg`]) under
218    /// Tesseract. The `Err` says what is accepted.
219    pub fn validate_lang(self, raw: &str) -> Result<(), String> {
220        match self {
221            Self::PpOcr => OcrLang::parse(raw).map(|_| ()).ok_or_else(|| {
222                format!(
223                    "ocr_lang {raw:?} names no language the OCR models read ({})",
224                    OcrLang::ACCEPTED
225                )
226            }),
227            Self::Tesseract => crate::tesseract::lang_arg(raw).map(|_| ()),
228        }
229    }
230}
231
232/// The process-level OCR render scale from `DOCLING_RS_OCR_SCALE` (#254,
233/// upstream docling#3877's `OcrOptions.scale`): pixels per PDF point fed to
234/// the recognizer. Unset/empty → `None` (OCR reads the pipeline's own page
235/// render, 2.0 px/pt); non-positive or unparsable values warn and are ignored.
236pub fn scale_from_env() -> Option<f32> {
237    let raw = docling_core::env::nonempty("DOCLING_RS_OCR_SCALE")?;
238    match raw.parse::<f32>() {
239        Ok(s) if s > 0.0 && s.is_finite() => Some(s),
240        _ => {
241            eprintln!(
242                "docling-pdf: DOCLING_RS_OCR_SCALE={raw:?} is not a positive number; ignored"
243            );
244            None
245        }
246    }
247}
248
249/// Resolve the recognition model + dictionary pair for `lang`. An English
250/// default that isn't on disk (older model checkouts) degrades to the `ch_`
251/// pair with a warning rather than failing — the usual missing-optional-asset
252/// convention. Explicit `DOCLING_OCR_REC_ONNX` / `DOCLING_OCR_DICT` paths win
253/// over all of this; they are a pair, so set both together.
254pub(crate) fn resolve_rec_pair(lang: OcrLang) -> (String, String) {
255    const CH: (&str, &str) = (".models/ocr_rec.onnx", ".models/ppocr_keys_v1.txt");
256    const EN: (&str, &str) = (".models/ocr_rec_en.onnx", ".models/en_dict.txt");
257    let want_ch = lang == OcrLang::Ch;
258    let pick = if want_ch { CH } else { EN };
259    let (mut rec, mut dict) = (crate::resolve_asset(pick.0), crate::resolve_asset(pick.1));
260    if !want_ch && (!std::path::Path::new(&rec).exists() || !std::path::Path::new(&dict).exists()) {
261        let (ch_rec, ch_dict) = (crate::resolve_asset(CH.0), crate::resolve_asset(CH.1));
262        if std::path::Path::new(&ch_rec).exists() && std::path::Path::new(&ch_dict).exists() {
263            eprintln!(
264                "docling-pdf: English OCR model not found ({rec}); falling back to the \
265                 multilingual ch_ model — expect weak Latin word spacing. Fetch it with \
266                 scripts/install/download_dependencies.sh"
267            );
268            (rec, dict) = (ch_rec, ch_dict);
269        }
270    }
271    (
272        docling_core::env::nonempty("DOCLING_OCR_REC_ONNX").unwrap_or(rec),
273        docling_core::env::nonempty("DOCLING_OCR_DICT").unwrap_or(dict),
274    )
275}
276
277/// One recognised line: its text and mean emitted-character confidence.
278type Recognized = (String, f32);
279
280impl OcrModel {
281    /// Load the recognition model and its character dictionary for `lang` —
282    /// see [`resolve_rec_pair`] for the selection rules (explicit
283    /// `DOCLING_OCR_REC_ONNX`/`DOCLING_OCR_DICT` paths win) — with `lanes`
284    /// recognition sessions.
285    ///
286    /// Each session is pinned to one intra-op thread: ORT's multi-threaded
287    /// float-reduction order varies across runs, which flips the CTC argmax on
288    /// low-confidence characters (e.g. noisy faxes) and makes the snapshot
289    /// output non-deterministic. Recognition is linear in line width (~0.17 ms
290    /// per pixel column on one core) and on a scanned page it, plus the
291    /// orientation probe that reads the six widest lines, is ~35% of the wall
292    /// time while the other cores idle. Lines are independent, so `lanes`
293    /// sessions recognise disjoint same-width batches concurrently — each
294    /// line still sees exactly the single-thread kernel path, results are
295    /// placed by index, and the output is byte-identical to one lane.
296    /// `DOCLING_RS_OCR_SESSIONS` overrides the caller's lane count.
297    pub fn load_with(lang: OcrLang, lanes: usize) -> Result<Self, String> {
298        let (rec_path, dict_path) = resolve_rec_pair(lang);
299        let lanes = docling_core::env::parse::<usize>("DOCLING_RS_OCR_SESSIONS")
300            .filter(|&n| n > 0)
301            .unwrap_or(lanes)
302            .clamp(1, 8);
303        let open = || -> Result<Session, String> {
304            let builder = Session::builder()
305                .map_err(|e| format!("ocr: builder: {e}"))?
306                .with_intra_threads(1)
307                .map_err(|e| format!("ocr: intra_threads: {e}"))?;
308            let builder = docling_onnx::apply(builder).map_err(|e| format!("ocr: {e}"))?;
309            docling_onnx::commit(builder, &rec_path, "rec")
310                .map_err(|e| format!("ocr: load {rec_path}: {e}"))
311        };
312        // The lanes are independent sessions over the same file — open them
313        // concurrently so extra lanes cost no extra start-up latency.
314        let recs: Vec<Session> = std::thread::scope(|s| {
315            let handles: Vec<_> = (0..lanes).map(|_| s.spawn(open)).collect();
316            handles
317                .into_iter()
318                .map(|h| {
319                    h.join()
320                        .map_err(|_| "ocr: session thread panicked".to_string())?
321                })
322                .collect::<Result<Vec<_>, String>>()
323        })?;
324        let dict = std::fs::read_to_string(&dict_path)
325            .map_err(|e| format!("ocr: read dict {dict_path}: {e}"))?;
326        Ok(Self {
327            recs,
328            chars: dict_chars(&dict),
329        })
330    }
331
332    /// Recognise every width batch of `lines`, dealt round-robin across the
333    /// lanes, and return `(line index, (text, confidence))` in batch order —
334    /// the same order the sequential loop produced, whatever the scheduling.
335    fn recognize_all(&mut self, lines: &[PrepLine]) -> Result<Vec<(usize, Recognized)>, String> {
336        let batches = width_batches(lines);
337        let lanes = self.recs.len().min(batches.len()).max(1);
338        let chars = &self.chars;
339        // One result slot per batch keeps the merge order independent of
340        // which lane finished first.
341        let mut per_batch: Vec<Option<Result<Vec<Recognized>, String>>> =
342            (0..batches.len()).map(|_| None).collect();
343        if lanes <= 1 {
344            for (slot, (w, chunk)) in per_batch.iter_mut().zip(&batches) {
345                *slot = Some(recognize_batch(&mut self.recs[0], chars, *w, chunk, lines));
346            }
347        } else {
348            std::thread::scope(|s| {
349                let handles: Vec<_> = self
350                    .recs
351                    .iter_mut()
352                    .take(lanes)
353                    .enumerate()
354                    .map(|(lane, rec)| {
355                        let batches = &batches;
356                        s.spawn(move || {
357                            batches
358                                .iter()
359                                .enumerate()
360                                .filter(|(k, _)| k % lanes == lane)
361                                .map(|(k, (w, chunk))| {
362                                    (k, recognize_batch(rec, chars, *w, chunk, lines))
363                                })
364                                .collect::<Vec<_>>()
365                        })
366                    })
367                    .collect();
368                for h in handles {
369                    for (k, r) in h.join().expect("ocr lane panicked") {
370                        per_batch[k] = Some(r);
371                    }
372                }
373            });
374        }
375        let mut out = Vec::with_capacity(lines.len());
376        for ((_, chunk), slot) in batches.iter().zip(per_batch) {
377            let texts = slot.expect("every batch is assigned a lane")?;
378            out.extend(chunk.iter().copied().zip(texts));
379        }
380        Ok(out)
381    }
382
383    /// Recognise a batch of prepared *same-width* lines in one session run.
384    ///
385    /// Only equal widths ever share a run: same-width batching is
386    /// bit-identical to one-at-a-time recognition (each sample keeps its own
387    /// data and per-sample kernel reduction order — verified empirically on
388    /// the scanned corpus), whereas width-padding leaks into the real
389    /// timesteps through the model's global-attention blocks and measurably
390    /// changes low-confidence characters.
391    /// Recognize `lines` and reduce to orientation-probe evidence: the
392    /// confidence-weighted character count `Σ(conf × chars)` plus the raw
393    /// character total (#225). Same deterministic width-batching as page OCR.
394    pub(crate) fn score_lines(&mut self, lines: &[PrepLine]) -> Result<(f32, usize), String> {
395        // Same accumulation order as the sequential loop (batch order), so
396        // the f32 sum is bit-identical regardless of lane scheduling.
397        let mut weighted = 0.0f32;
398        let mut chars = 0usize;
399        for (_, (text, conf)) in self.recognize_all(lines)? {
400            let n = text.trim().chars().count();
401            weighted += conf * n as f32;
402            chars += n;
403        }
404        Ok((weighted, chars))
405    }
406
407    /// OCR a page: produce text cells (page points) for every line found inside
408    /// the text regions, each paired with its recognition confidence (mean
409    /// emitted-character probability — feeds the page `ocr_score`, #183).
410    /// `scale` is image-px per page-point.
411    pub fn ocr_page(
412        &mut self,
413        img: &RgbImage,
414        regions: &[Region],
415        scale: f32,
416    ) -> Result<Vec<(TextCell, f32)>, String> {
417        // Gather every line crop on the page first (shared with the browser
418        // path), so equal-width lines can share a recognition run regardless
419        // of which region they came from.
420        let (bboxes, lines) =
421            crate::timing::timed("ocr.prep", || prep_region_lines(img, regions, scale));
422
423        // Deterministic width-batching (shared with the wasm path), dealt
424        // across the recognition lanes.
425        let mut texts = vec![(String::new(), 0.0f32); lines.len()];
426        crate::timing::timed("ocr.rec", || -> Result<(), String> {
427            for (i, text) in self.recognize_all(&lines)? {
428                texts[i] = text;
429            }
430            Ok(())
431        })?;
432
433        // Emit cells in page order, exactly as the sequential walk did.
434        let mut cells = Vec::new();
435        for ((l, t, r, b), (text, conf)) in bboxes.into_iter().zip(texts) {
436            let text = text.trim().to_string();
437            if text.is_empty() {
438                continue;
439            }
440            cells.push((TextCell { text, l, t, r, b }, conf));
441        }
442        Ok(cells)
443    }
444
445    /// Recognize the *word* crops inside the page's table regions (mirroring
446    /// the browser scanned path): [`ocr_page`](Self::ocr_page) deliberately
447    /// skips table labels, so a scanned table would otherwise reach the cell
448    /// matcher with no words at all and dissolve (#173). Returns word-level
449    /// [`TextCell`]s in page points.
450    pub fn ocr_table_words(
451        &mut self,
452        img: &RgbImage,
453        regions: &[Region],
454        scale: f32,
455    ) -> Result<Vec<(TextCell, f32)>, String> {
456        let (bboxes, lines) = prep_table_words(img, regions, scale);
457        let mut texts = vec![(String::new(), 0.0f32); lines.len()];
458        for (i, text) in self.recognize_all(&lines)? {
459            texts[i] = text;
460        }
461        let mut cells = Vec::new();
462        for ((l, t, r, b), (text, conf)) in bboxes.into_iter().zip(texts) {
463            let text = text.trim().to_string();
464            if text.is_empty() {
465                continue;
466            }
467            cells.push((TextCell { text, l, t, r, b }, conf));
468        }
469        Ok(cells)
470    }
471}
472
473/// Recognise a batch of prepared *same-width* lines in one run of `rec`.
474///
475/// Only equal widths ever share a run: same-width batching is bit-identical
476/// to one-at-a-time recognition (each sample keeps its own data and per-sample
477/// kernel reduction order — verified empirically on the scanned corpus),
478/// whereas width-padding leaks into the real timesteps through the model's
479/// global-attention blocks and measurably changes low-confidence characters.
480fn recognize_batch(
481    rec: &mut Session,
482    chars: &[String],
483    w: usize,
484    chunk: &[usize],
485    lines: &[PrepLine],
486) -> Result<Vec<(String, f32)>, String> {
487    let n = chunk.len();
488    let data = batch_input(w, chunk, lines);
489    let input = Tensor::from_array(([n, 3, REC_HEIGHT as usize, w], data))
490        .map_err(|e| format!("ocr: input tensor: {e}"))?;
491    let outputs = rec
492        .run(ort::inputs!["x" => input])
493        .map_err(|e| format!("ocr: rec inference: {e}"))?;
494    let (shape, probs) = outputs[0]
495        .try_extract_tensor::<f32>()
496        .map_err(|e| format!("ocr: extract rec: {e}"))?;
497    let t_len = shape[1] as usize;
498    let nc = shape[2] as usize;
499    Ok((0..n)
500        .map(|i| decode_row_scored(chars, &probs[i * t_len * nc..(i + 1) * t_len * nc], nc))
501        .collect())
502}
503
504#[cfg(test)]
505mod tests {
506    use super::*;
507
508    /// #388: BCP-47 tags and the ISO 639-2/3 codes for English and Chinese
509    /// resolve to the two recognizers, with or without docling's `iso:`
510    /// prefix and whatever the script/region subtags; other languages and
511    /// nonsense stay `None`.
512    #[test]
513    fn ocr_lang_accepts_bcp47_tags_for_the_two_recognizers() {
514        for id in [
515            "en",
516            "EN",
517            " en ",
518            "en-US",
519            "en_GB",
520            "eng",
521            "english",
522            "iso:en",
523            "ISO:en-GB",
524            "en-Latn-US",
525        ] {
526            assert_eq!(OcrLang::parse(id), Some(OcrLang::En), "{id:?}");
527        }
528        for id in [
529            "ch",
530            "zh",
531            "zho",
532            "chi",
533            "cmn",
534            "chinese",
535            "ch_sim",
536            "ch_tra",
537            "chinese_cht",
538            "zh-Hans",
539            "zh-Hant",
540            "zh-CN",
541            "zh-TW",
542            "zh-Hant-HK",
543            "zh_SG",
544            "iso:zh-Hans",
545        ] {
546            assert_eq!(OcrLang::parse(id), Some(OcrLang::Ch), "{id:?}");
547        }
548        for id in [
549            "", "de", "fr-FR", "ja", "deu", "cn", "latin", "iso:", "iso:und", "e",
550        ] {
551            assert_eq!(OcrLang::parse(id), None, "{id:?}");
552        }
553    }
554
555    /// #254: docling's four `OcrMode` ids parse; `full_page`/`layout_regions`
556    /// reduce to the force-full-page machinery, the default/pdf-aware pair to
557    /// the standard text-layer-aware path. Unknown ids parse to nothing.
558    #[test]
559    fn ocr_mode_ids_parse_and_map_to_forcing() {
560        for (id, mode, forces) in [
561            ("default", OcrMode::Default, false),
562            ("full_page", OcrMode::FullPage, true),
563            ("layout_regions", OcrMode::LayoutRegions, true),
564            (
565                "pdf_aware_layout_regions",
566                OcrMode::PdfAwareLayoutRegions,
567                false,
568            ),
569        ] {
570            assert_eq!(OcrMode::parse(id), Some(mode));
571            assert_eq!(mode.forces_full_page(), forces, "{id}");
572        }
573        assert_eq!(OcrMode::parse(" Full_Page "), Some(OcrMode::FullPage));
574        assert_eq!(OcrMode::parse("easyocr"), None);
575        assert_eq!(OcrMode::parse(""), None);
576    }
577
578    /// #460: the engine ids, and engine-aware `ocr_lang` validation — `deu`
579    /// is a Tesseract stem, not a PP-OCR model; `en` works under both.
580    #[test]
581    fn ocr_engine_ids_and_lang_validation() {
582        for id in ["ppocr", "PP-OCR", " rapidocr ", "default"] {
583            assert_eq!(OcrEngine::parse(id), Some(OcrEngine::PpOcr), "{id:?}");
584        }
585        for id in ["tesseract", "Tesseract_CLI", "tesserocr"] {
586            assert_eq!(OcrEngine::parse(id), Some(OcrEngine::Tesseract), "{id:?}");
587        }
588        assert_eq!(OcrEngine::parse("easyocr"), None);
589        assert!(OcrEngine::PpOcr.validate_lang("en").is_ok());
590        assert!(OcrEngine::PpOcr.validate_lang("deu").is_err());
591        assert!(OcrEngine::Tesseract.validate_lang("en").is_ok());
592        assert!(OcrEngine::Tesseract.validate_lang("deu+fra").is_ok());
593        assert!(OcrEngine::Tesseract.validate_lang("xx").is_err());
594    }
595}