Skip to main content

docling_pdf/
ocr.rs

1//! OCR for scanned pages, via the PP-OCRv3 recognition model (CRNN/SVTR) run
2//! with `ort`. The layout model locates the text regions on the page image
3//! (it works without a text layer); inside them the recognizer reads the
4//! lines the PP-OCR text detector found (`ocr_det`, #429/#570 — RapidOCR's
5//! own crops, each one text run with the detector's margin), falling back to
6//! a horizontal-projection split of the region crop where the detector saw
7//! nothing or is not installed; each line is recognised and decoded with CTC
8//! — producing [`TextCell`]s the normal layout assembly then consumes. A line
9//! under RapidOCR's `text_score` confidence is dropped (`ocr_prep::text_score`).
10
11use image::RgbImage;
12use ort::session::Session;
13use ort::value::Tensor;
14
15use crate::layout::Region;
16// The ONNX-free half (line prep, batching, CTC decode) lives in `ocr_prep`
17// so the wasm build shares it verbatim (#79 phase 2).
18use crate::ocr_prep::{
19    batch_input, decode_row_scored, dict_chars, prep_region_lines, prep_region_lines_det,
20    prep_table_words, prep_table_words_det, width_batches, PrepLine, REC_HEIGHT,
21};
22use crate::pdfium_backend::TextCell;
23
24pub struct OcrModel {
25    /// Single-threaded recognition sessions, one per parallel lane (see
26    /// [`Self::load_with`]); lines are dealt across them by batch index.
27    recs: Vec<Session>,
28    /// CTC classes: index 0 = blank, then the dictionary, then space.
29    chars: Vec<String>,
30}
31
32/// OCR recognition language: which PP-OCRv3 model + dictionary pair runs
33/// when the PP-OCRv6 recognizer is not installed.
34///
35/// With `.models/ocr_rec_v6.onnx` + `.models/ocr_rec_v6_dict.txt` on disk
36/// (#570; `download_dependencies.sh` fetches them) both languages run that
37/// one multilingual model — RapidOCR's `PP-OCRv6_rec_small`, the recognizer
38/// docling runs for English and Chinese alike, and the single largest factor
39/// in the FUNSD word-recall gap once the detector's lines are the crops
40/// (0.70 → 0.77 on 30 forms). Without it, the PP-OCRv3 pairs: the default is
41/// **English** (`.models/ocr_rec_en.onnx` + `.models/en_dict.txt`) — the
42/// multilingual `ch_` v3 model reads Latin scripts with badly degraded word
43/// spacing (glued words on ordinary English scans) — and `Ch` selects the
44/// `ch_` pair (`.models/ocr_rec.onnx` + `.models/ppocr_keys_v1.txt`), what
45/// the PDF conformance baselines were pinned against;
46/// `scripts/conformance/pdf_*.sh` pin it explicitly by path, which wins over
47/// this selector and over the v6 preference.
48#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
49pub enum OcrLang {
50    /// en_PP-OCRv3 — English-only, proper Latin word spacing.
51    #[default]
52    En,
53    /// ch_PP-OCRv3 — multilingual; the docling-conformance model.
54    Ch,
55}
56
57impl OcrLang {
58    /// Parse a user-supplied language id (#388): the engine's own codes
59    /// (`en`, `ch`) and BCP-47 tags naming a language one of the two
60    /// recognizers reads, trimmed and case-insensitive. `None` for anything
61    /// else — callers surface their own error/warning.
62    ///
63    /// docling canonicalizes OCR languages across its engines (docling#4075):
64    /// a bare value is the engine's native code, an `iso:`-prefixed value a
65    /// BCP-47 tag reduced to a language-script pair with the region dropped
66    /// (`zh-CN` and `zh-Hans` are the same recognizer, `en-GB` is `en`), and
67    /// the RapidOCR adapter maps `en` → its `en` model and `zh-Hans` → `ch`.
68    /// With only those two PP-OCRv3 pairs on board there is no ambiguity, so
69    /// the prefix is optional here: `en-US`, `eng`, `zh`, `zh-Hans`, `iso:zh-CN`
70    /// all resolve without a warning. Accepted primary subtags: English as
71    /// `en` / ISO 639-2/3 `eng` / docling's legacy `english`; Chinese as the
72    /// engine code `ch` (and RapidOCR's `chinese_cht`), `zh` / `zho` / `chi` /
73    /// `cmn` / legacy `chinese` / EasyOCR's `ch_sim` / `ch_tra`. Script,
74    /// region and variant subtags (`-Hans`, `-Hant`, `-CN`, `-TW`, `_US`) are
75    /// ignored: a traditional-script request (`zh-Hant`, `zh-TW`) gets the
76    /// multilingual `ch` recognizer too, the closest model shipped — upstream
77    /// would pick RapidOCR's separate `chinese_cht`, which this engine does
78    /// not carry. Genuinely unsupported languages (`de`, `fr`, `ja`, …) parse
79    /// to `None` and keep warning.
80    pub fn parse(s: &str) -> Option<Self> {
81        let token = s.trim().to_ascii_lowercase();
82        let tag = token.strip_prefix("iso:").unwrap_or(&token).trim();
83        let primary = tag.split(['-', '_']).next().unwrap_or_default();
84        match primary {
85            "en" | "eng" | "english" => Some(Self::En),
86            "ch" | "chinese_cht" | "zh" | "zho" | "chi" | "cmn" | "chinese" | "ch_sim"
87            | "ch_tra" => Some(Self::Ch),
88            _ => None,
89        }
90    }
91
92    /// The process-level choice from `DOCLING_RS_OCR_LANG` (empty/unset → the
93    /// English default; unknown values warn and use English).
94    pub fn from_env() -> Self {
95        let Some(raw) = docling_core::env::nonempty("DOCLING_RS_OCR_LANG") else {
96            return Self::default();
97        };
98        Self::parse(&raw).unwrap_or_else(|| {
99            eprintln!(
100                "docling-pdf: DOCLING_RS_OCR_LANG={raw:?} names no language the en/ch \
101                 recognizers read ({}); using en",
102                Self::ACCEPTED
103            );
104            Self::default()
105        })
106    }
107
108    /// The accepted spellings, for error messages and docs.
109    pub const ACCEPTED: &'static str =
110        "en | ch, or a BCP-47 tag for English or Chinese such as en-US, eng, zh, zh-Hans, zh-TW";
111}
112
113/// Which document regions feed the OCR — docling 2.116's `OcrMode` (#254,
114/// upstream docling#3710). Upstream restructured its pipeline so OCR runs
115/// *after* layout, on layout regions filtered by the PDF text layer — the
116/// architecture this port has always had — and named the strategies:
117///
118/// - `PdfAwareLayoutRegions` (upstream's **default**): OCR only layout regions
119///   the embedded text layer can't cover. Exactly the standard path here —
120///   scanned pages OCR their regions, digital pages OCR only text-less bitmap
121///   areas.
122/// - `FullPage` / `LayoutRegions`: ignore the PDF text layer and OCR
123///   everything. Both map onto the [`force_full_page_ocr`] machinery (discard
124///   the text layer, OCR every layout region): the upstream distinction —
125///   whole-page vs per-region *detector* input — has no analogue in this
126///   engine, whose PP-OCR recognizer always consumes per-region line crops.
127///
128/// [`force_full_page_ocr`]: crate::Pipeline::force_full_page_ocr
129#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
130pub enum OcrMode {
131    /// Upstream's `default`: currently wired to `PdfAwareLayoutRegions`.
132    #[default]
133    Default,
134    /// OCR the full page, text layer ignored (docling's `full_page`; the
135    /// mode-shaped spelling of `force_full_page_ocr`).
136    FullPage,
137    /// OCR every layout region, text layer ignored (docling's
138    /// `layout_regions`).
139    LayoutRegions,
140    /// OCR layout regions the text layer can't cover (docling's
141    /// `pdf_aware_layout_regions` — the default behavior).
142    PdfAwareLayoutRegions,
143}
144
145impl OcrMode {
146    /// Parse docling's mode ids. `None` for anything else — callers surface
147    /// their own error/warning.
148    pub fn parse(s: &str) -> Option<Self> {
149        match s.trim().to_ascii_lowercase().as_str() {
150            "default" => Some(Self::Default),
151            "full_page" => Some(Self::FullPage),
152            "layout_regions" => Some(Self::LayoutRegions),
153            "pdf_aware_layout_regions" => Some(Self::PdfAwareLayoutRegions),
154            _ => None,
155        }
156    }
157
158    /// The process-level choice from `DOCLING_RS_OCR_MODE` (empty/unset → the
159    /// default; unknown values warn and use the default).
160    pub fn from_env() -> Self {
161        let Some(raw) = docling_core::env::nonempty("DOCLING_RS_OCR_MODE") else {
162            return Self::default();
163        };
164        Self::parse(&raw).unwrap_or_else(|| {
165            eprintln!(
166                "docling-pdf: DOCLING_RS_OCR_MODE={raw:?} is not \
167                 default|full_page|layout_regions|pdf_aware_layout_regions; using default"
168            );
169            Self::default()
170        })
171    }
172
173    /// Whether this mode discards the embedded text layer — the engine truth
174    /// both non-default modes reduce to.
175    pub fn forces_full_page(self) -> bool {
176        matches!(self, Self::FullPage | Self::LayoutRegions)
177    }
178}
179
180/// Which OCR engine recognizes text (#460): the built-in PP-OCRv3 recognizer
181/// (+ the RapidOCR text detector) — the default and the engine every
182/// conformance baseline is pinned against — or the system `tesseract` binary
183/// (see [`crate::tesseract`]), docling's `TesseractCliOcrOptions`
184/// counterpart. Both consume the same layout-region crops and produce the
185/// same cells; everything downstream is engine-agnostic.
186#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
187pub enum OcrEngine {
188    /// PP-OCRv3 recognition via ONNX Runtime (docling's `rapidocr` kind).
189    #[default]
190    PpOcr,
191    /// The `tesseract` CLI (docling's `tesseract` kind).
192    Tesseract,
193}
194
195impl OcrEngine {
196    /// Parse an engine id: `ppocr` (also `pp-ocr`, `rapidocr`, docling's
197    /// kind name) or `tesseract` (also `tesseract_cli`, `tesserocr`),
198    /// trimmed and case-insensitive. `None` for anything else.
199    pub fn parse(s: &str) -> Option<Self> {
200        match s.trim().to_ascii_lowercase().as_str() {
201            "ppocr" | "pp-ocr" | "pp_ocr" | "rapidocr" | "onnx" | "default" => Some(Self::PpOcr),
202            "tesseract" | "tesseract_cli" | "tesseract-cli" | "tesserocr" => Some(Self::Tesseract),
203            _ => None,
204        }
205    }
206
207    /// The process-level choice from `DOCLING_RS_OCR_ENGINE` (empty/unset →
208    /// PP-OCR; unknown values warn and use PP-OCR).
209    pub fn from_env() -> Self {
210        let Some(raw) = docling_core::env::nonempty("DOCLING_RS_OCR_ENGINE") else {
211            return Self::default();
212        };
213        Self::parse(&raw).unwrap_or_else(|| {
214            eprintln!(
215                "docling-pdf: DOCLING_RS_OCR_ENGINE={raw:?} is not ppocr|tesseract; using ppocr"
216            );
217            Self::default()
218        })
219    }
220
221    /// The accepted spellings, for error messages and docs.
222    pub const ACCEPTED: &'static str = "ppocr | tesseract";
223
224    /// Whether `raw` is an `ocr_lang` this engine can act on — what the
225    /// option surfaces validate up front: the en/ch model switch (or a
226    /// BCP-47 tag for either, [`OcrLang::parse`]) under PP-OCR; tessdata
227    /// stems and BCP-47 tags ([`crate::tesseract::lang_arg`]) under
228    /// Tesseract. The `Err` says what is accepted.
229    pub fn validate_lang(self, raw: &str) -> Result<(), String> {
230        match self {
231            Self::PpOcr => OcrLang::parse(raw).map(|_| ()).ok_or_else(|| {
232                format!(
233                    "ocr_lang {raw:?} names no language the OCR models read ({})",
234                    OcrLang::ACCEPTED
235                )
236            }),
237            Self::Tesseract => crate::tesseract::lang_arg(raw).map(|_| ()),
238        }
239    }
240}
241
242/// The process-level OCR render scale from `DOCLING_RS_OCR_SCALE` (#254,
243/// upstream docling#3877's `OcrOptions.scale`): pixels per PDF point fed to
244/// the recognizer. Unset/empty → `None` (OCR reads the pipeline's own page
245/// render, 2.0 px/pt); non-positive or unparsable values warn and are ignored.
246pub fn scale_from_env() -> Option<f32> {
247    let raw = docling_core::env::nonempty("DOCLING_RS_OCR_SCALE")?;
248    match raw.parse::<f32>() {
249        Ok(s) if s > 0.0 && s.is_finite() => Some(s),
250        _ => {
251            eprintln!(
252                "docling-pdf: DOCLING_RS_OCR_SCALE={raw:?} is not a positive number; ignored"
253            );
254            None
255        }
256    }
257}
258
259/// Whether the text detector's boxes are the recognizer's line source inside
260/// layout regions (#570; `DOCLING_RS_OCR_LINES`: `det` default, `projection`
261/// = the ink-projection strips alone, the pre-#570 behavior). Cached.
262pub fn det_lines() -> bool {
263    static MODE: std::sync::OnceLock<bool> = std::sync::OnceLock::new();
264    *MODE.get_or_init(|| {
265        let raw = docling_core::env::nonempty("DOCLING_RS_OCR_LINES").unwrap_or_default();
266        match raw.trim().to_ascii_lowercase().as_str() {
267            "" | "det" | "detector" | "auto" => true,
268            "projection" | "proj" | "off" => false,
269            _ => {
270                eprintln!(
271                    "docling-pdf: DOCLING_RS_OCR_LINES={raw:?} is not det|projection; using det"
272                );
273                true
274            }
275        }
276    })
277}
278
279/// Resolve the recognition model + dictionary pair for `lang`. An English
280/// default that isn't on disk (older model checkouts) degrades to the `ch_`
281/// pair with a warning rather than failing — the usual missing-optional-asset
282/// convention. Explicit `DOCLING_OCR_REC_ONNX` / `DOCLING_OCR_DICT` paths win
283/// over all of this; they are a pair, so set both together.
284pub(crate) fn resolve_rec_pair(lang: OcrLang) -> (String, String) {
285    const CH: (&str, &str) = (".models/ocr_rec.onnx", ".models/ppocr_keys_v1.txt");
286    const EN: (&str, &str) = (".models/ocr_rec_en.onnx", ".models/en_dict.txt");
287    const V6: (&str, &str) = (".models/ocr_rec_v6.onnx", ".models/ocr_rec_v6_dict.txt");
288    let exists = |p: &str| std::path::Path::new(p).exists();
289    // The multilingual PP-OCRv6 recognizer, when installed, serves both
290    // languages (see `OcrLang`); explicit paths below still win.
291    let (v6_rec, v6_dict) = (crate::resolve_asset(V6.0), crate::resolve_asset(V6.1));
292    let (mut rec, mut dict) = if exists(&v6_rec) && exists(&v6_dict) {
293        (v6_rec, v6_dict)
294    } else {
295        let pick = if lang == OcrLang::Ch { CH } else { EN };
296        (crate::resolve_asset(pick.0), crate::resolve_asset(pick.1))
297    };
298    let want_ch = lang == OcrLang::Ch;
299    if !want_ch && (!exists(&rec) || !exists(&dict)) {
300        let (ch_rec, ch_dict) = (crate::resolve_asset(CH.0), crate::resolve_asset(CH.1));
301        if std::path::Path::new(&ch_rec).exists() && std::path::Path::new(&ch_dict).exists() {
302            eprintln!(
303                "docling-pdf: English OCR model not found ({rec}); falling back to the \
304                 multilingual ch_ model — expect weak Latin word spacing. Fetch it with \
305                 scripts/install/download_dependencies.sh"
306            );
307            (rec, dict) = (ch_rec, ch_dict);
308        }
309    }
310    (
311        docling_core::env::nonempty("DOCLING_OCR_REC_ONNX").unwrap_or(rec),
312        docling_core::env::nonempty("DOCLING_OCR_DICT").unwrap_or(dict),
313    )
314}
315
316/// One recognised line: its text and mean emitted-character confidence.
317type Recognized = (String, f32);
318
319impl OcrModel {
320    /// Load the recognition model and its character dictionary for `lang` —
321    /// see [`resolve_rec_pair`] for the selection rules (explicit
322    /// `DOCLING_OCR_REC_ONNX`/`DOCLING_OCR_DICT` paths win) — with `lanes`
323    /// recognition sessions.
324    ///
325    /// Each session is pinned to one intra-op thread: ORT's multi-threaded
326    /// float-reduction order varies across runs, which flips the CTC argmax on
327    /// low-confidence characters (e.g. noisy faxes) and makes the snapshot
328    /// output non-deterministic. Recognition is linear in line width (~0.17 ms
329    /// per pixel column on one core) and on a scanned page it, plus the
330    /// orientation probe that reads the six widest lines, is ~35% of the wall
331    /// time while the other cores idle. Lines are independent, so `lanes`
332    /// sessions recognise disjoint same-width batches concurrently — each
333    /// line still sees exactly the single-thread kernel path, results are
334    /// placed by index, and the output is byte-identical to one lane.
335    /// `DOCLING_RS_OCR_SESSIONS` overrides the caller's lane count.
336    pub fn load_with(lang: OcrLang, lanes: usize) -> Result<Self, String> {
337        let (rec_path, dict_path) = resolve_rec_pair(lang);
338        let lanes = docling_core::env::parse::<usize>("DOCLING_RS_OCR_SESSIONS")
339            .filter(|&n| n > 0)
340            .unwrap_or(lanes)
341            .clamp(1, 8);
342        let open = || -> Result<Session, String> {
343            let builder = docling_onnx::session_builder()
344                .map_err(|e| format!("ocr: builder: {e}"))?
345                .with_intra_threads(1)
346                .map_err(|e| format!("ocr: intra_threads: {e}"))?;
347            let builder = docling_onnx::apply(builder).map_err(|e| format!("ocr: {e}"))?;
348            docling_onnx::commit(builder, &rec_path, "rec")
349                .map_err(|e| format!("ocr: load {rec_path}: {e}"))
350        };
351        // The lanes are independent sessions over the same file — open them
352        // concurrently so extra lanes cost no extra start-up latency.
353        let recs: Vec<Session> = std::thread::scope(|s| {
354            let handles: Vec<_> = (0..lanes).map(|_| s.spawn(open)).collect();
355            handles
356                .into_iter()
357                .map(|h| {
358                    h.join()
359                        .map_err(|_| "ocr: session thread panicked".to_string())?
360                })
361                .collect::<Result<Vec<_>, String>>()
362        })?;
363        let dict = std::fs::read_to_string(&dict_path)
364            .map_err(|e| format!("ocr: read dict {dict_path}: {e}"))?;
365        Ok(Self {
366            recs,
367            chars: dict_chars(&dict),
368        })
369    }
370
371    /// Recognise every width batch of `lines`, dealt round-robin across the
372    /// lanes, and return `(line index, (text, confidence))` in batch order —
373    /// the same order the sequential loop produced, whatever the scheduling.
374    fn recognize_all(&mut self, lines: &[PrepLine]) -> Result<Vec<(usize, Recognized)>, String> {
375        let batches = width_batches(lines);
376        let lanes = self.recs.len().min(batches.len()).max(1);
377        let chars = &self.chars;
378        // One result slot per batch keeps the merge order independent of
379        // which lane finished first.
380        let mut per_batch: Vec<Option<Result<Vec<Recognized>, String>>> =
381            (0..batches.len()).map(|_| None).collect();
382        if lanes <= 1 {
383            for (slot, (w, chunk)) in per_batch.iter_mut().zip(&batches) {
384                *slot = Some(recognize_batch(&mut self.recs[0], chars, *w, chunk, lines));
385            }
386        } else {
387            std::thread::scope(|s| {
388                let handles: Vec<_> = self
389                    .recs
390                    .iter_mut()
391                    .take(lanes)
392                    .enumerate()
393                    .map(|(lane, rec)| {
394                        let batches = &batches;
395                        s.spawn(move || {
396                            batches
397                                .iter()
398                                .enumerate()
399                                .filter(|(k, _)| k % lanes == lane)
400                                .map(|(k, (w, chunk))| {
401                                    (k, recognize_batch(rec, chars, *w, chunk, lines))
402                                })
403                                .collect::<Vec<_>>()
404                        })
405                    })
406                    .collect();
407                for h in handles {
408                    for (k, r) in h.join().expect("ocr lane panicked") {
409                        per_batch[k] = Some(r);
410                    }
411                }
412            });
413        }
414        let mut out = Vec::with_capacity(lines.len());
415        for ((_, chunk), slot) in batches.iter().zip(per_batch) {
416            let texts = slot.expect("every batch is assigned a lane")?;
417            out.extend(chunk.iter().copied().zip(texts));
418        }
419        Ok(out)
420    }
421
422    /// Recognise a batch of prepared *same-width* lines in one session run.
423    ///
424    /// Only equal widths ever share a run: same-width batching is
425    /// bit-identical to one-at-a-time recognition (each sample keeps its own
426    /// data and per-sample kernel reduction order — verified empirically on
427    /// the scanned corpus), whereas width-padding leaks into the real
428    /// timesteps through the model's global-attention blocks and measurably
429    /// changes low-confidence characters.
430    /// Recognize `lines` and reduce to orientation-probe evidence: the
431    /// confidence-weighted character count `Σ(conf × chars)` plus the raw
432    /// character total (#225). Same deterministic width-batching as page OCR.
433    pub(crate) fn score_lines(&mut self, lines: &[PrepLine]) -> Result<(f32, usize), String> {
434        // Same accumulation order as the sequential loop (batch order), so
435        // the f32 sum is bit-identical regardless of lane scheduling.
436        let mut weighted = 0.0f32;
437        let mut chars = 0usize;
438        for (_, (text, conf)) in self.recognize_all(lines)? {
439            let n = text.trim().chars().count();
440            weighted += conf * n as f32;
441            chars += n;
442        }
443        Ok((weighted, chars))
444    }
445
446    /// OCR a page: produce text cells (page points) for every line found inside
447    /// the text regions, each paired with its recognition confidence (mean
448    /// emitted-character probability — feeds the page `ocr_score`, #183).
449    /// `scale` is image-px per page-point. `detected` — the text detector's
450    /// boxes, in image pixels of `img` — makes them the line source inside
451    /// the regions (#570, see [`prep_region_lines_det`]); `None` keeps the
452    /// projection segmentation.
453    pub fn ocr_page_with(
454        &mut self,
455        img: &RgbImage,
456        regions: &[Region],
457        scale: f32,
458        detected: Option<&[crate::ocr_det::DetBox]>,
459    ) -> Result<Vec<(TextCell, f32)>, String> {
460        // Gather every line crop on the page first (shared with the browser
461        // path), so equal-width lines can share a recognition run regardless
462        // of which region they came from.
463        let (bboxes, lines) = crate::timing::timed("ocr.prep", || match detected {
464            Some(det) => prep_region_lines_det(img, regions, scale, det),
465            None => prep_region_lines(img, regions, scale),
466        });
467
468        // Deterministic width-batching (shared with the wasm path), dealt
469        // across the recognition lanes.
470        let mut texts = vec![(String::new(), 0.0f32); lines.len()];
471        crate::timing::timed("ocr.rec", || -> Result<(), String> {
472            for (i, text) in self.recognize_all(&lines)? {
473                texts[i] = text;
474            }
475            Ok(())
476        })?;
477
478        // Emit cells in page order, exactly as the sequential walk did.
479        Ok(collect_cells(bboxes, texts))
480    }
481
482    /// Recognize the *word* crops inside the page's table regions (mirroring
483    /// the browser scanned path): [`ocr_page`](Self::ocr_page) deliberately
484    /// skips table labels, so a scanned table would otherwise reach the cell
485    /// matcher with no words at all and dissolve (#173). Returns word-level
486    /// [`TextCell`]s in page points.
487    pub fn ocr_table_words(
488        &mut self,
489        img: &RgbImage,
490        regions: &[Region],
491        scale: f32,
492        detected: Option<&[crate::ocr_det::DetBox]>,
493    ) -> Result<Vec<(TextCell, f32)>, String> {
494        let (bboxes, lines) = match detected {
495            Some(det) => prep_table_words_det(img, regions, scale, det),
496            None => prep_table_words(img, regions, scale),
497        };
498        let mut texts = vec![(String::new(), 0.0f32); lines.len()];
499        for (i, text) in self.recognize_all(&lines)? {
500            texts[i] = text;
501        }
502        Ok(collect_cells(bboxes, texts))
503    }
504}
505
506/// Pair recognized texts with their line boxes into page-point cells, in page
507/// order, dropping empty lines and those under [`text_score`].
508fn collect_cells(
509    bboxes: Vec<crate::ocr_prep::LineBox>,
510    texts: Vec<Recognized>,
511) -> Vec<(TextCell, f32)> {
512    let min_conf = crate::ocr_prep::text_score();
513    let mut cells = Vec::new();
514    for ((l, t, r, b), (text, conf)) in bboxes.into_iter().zip(texts) {
515        let text = text.trim().to_string();
516        if text.is_empty() || conf < min_conf {
517            continue;
518        }
519        cells.push((TextCell { text, l, t, r, b }, conf));
520    }
521    cells
522}
523
524/// Recognise a batch of prepared *same-width* lines in one run of `rec`.
525///
526/// Only equal widths ever share a run: same-width batching is bit-identical
527/// to one-at-a-time recognition (each sample keeps its own data and per-sample
528/// kernel reduction order — verified empirically on the scanned corpus),
529/// whereas width-padding leaks into the real timesteps through the model's
530/// global-attention blocks and measurably changes low-confidence characters.
531fn recognize_batch(
532    rec: &mut Session,
533    chars: &[String],
534    w: usize,
535    chunk: &[usize],
536    lines: &[PrepLine],
537) -> Result<Vec<(String, f32)>, String> {
538    let n = chunk.len();
539    let data = batch_input(w, chunk, lines);
540    let input = Tensor::from_array(([n, 3, REC_HEIGHT as usize, w], data))
541        .map_err(|e| format!("ocr: input tensor: {e}"))?;
542    let outputs = rec
543        .run(ort::inputs!["x" => input])
544        .map_err(|e| format!("ocr: rec inference: {e}"))?;
545    let (shape, probs) = outputs[0]
546        .try_extract_tensor::<f32>()
547        .map_err(|e| format!("ocr: extract rec: {e}"))?;
548    let t_len = shape[1] as usize;
549    let nc = shape[2] as usize;
550    Ok((0..n)
551        .map(|i| decode_row_scored(chars, &probs[i * t_len * nc..(i + 1) * t_len * nc], nc))
552        .collect())
553}
554
555#[cfg(test)]
556mod tests {
557    use super::*;
558
559    /// #388: BCP-47 tags and the ISO 639-2/3 codes for English and Chinese
560    /// resolve to the two recognizers, with or without docling's `iso:`
561    /// prefix and whatever the script/region subtags; other languages and
562    /// nonsense stay `None`.
563    #[test]
564    fn ocr_lang_accepts_bcp47_tags_for_the_two_recognizers() {
565        for id in [
566            "en",
567            "EN",
568            " en ",
569            "en-US",
570            "en_GB",
571            "eng",
572            "english",
573            "iso:en",
574            "ISO:en-GB",
575            "en-Latn-US",
576        ] {
577            assert_eq!(OcrLang::parse(id), Some(OcrLang::En), "{id:?}");
578        }
579        for id in [
580            "ch",
581            "zh",
582            "zho",
583            "chi",
584            "cmn",
585            "chinese",
586            "ch_sim",
587            "ch_tra",
588            "chinese_cht",
589            "zh-Hans",
590            "zh-Hant",
591            "zh-CN",
592            "zh-TW",
593            "zh-Hant-HK",
594            "zh_SG",
595            "iso:zh-Hans",
596        ] {
597            assert_eq!(OcrLang::parse(id), Some(OcrLang::Ch), "{id:?}");
598        }
599        for id in [
600            "", "de", "fr-FR", "ja", "deu", "cn", "latin", "iso:", "iso:und", "e",
601        ] {
602            assert_eq!(OcrLang::parse(id), None, "{id:?}");
603        }
604    }
605
606    /// #254: docling's four `OcrMode` ids parse; `full_page`/`layout_regions`
607    /// reduce to the force-full-page machinery, the default/pdf-aware pair to
608    /// the standard text-layer-aware path. Unknown ids parse to nothing.
609    #[test]
610    fn ocr_mode_ids_parse_and_map_to_forcing() {
611        for (id, mode, forces) in [
612            ("default", OcrMode::Default, false),
613            ("full_page", OcrMode::FullPage, true),
614            ("layout_regions", OcrMode::LayoutRegions, true),
615            (
616                "pdf_aware_layout_regions",
617                OcrMode::PdfAwareLayoutRegions,
618                false,
619            ),
620        ] {
621            assert_eq!(OcrMode::parse(id), Some(mode));
622            assert_eq!(mode.forces_full_page(), forces, "{id}");
623        }
624        assert_eq!(OcrMode::parse(" Full_Page "), Some(OcrMode::FullPage));
625        assert_eq!(OcrMode::parse("easyocr"), None);
626        assert_eq!(OcrMode::parse(""), None);
627    }
628
629    /// #460: the engine ids, and engine-aware `ocr_lang` validation — `deu`
630    /// is a Tesseract stem, not a PP-OCR model; `en` works under both.
631    #[test]
632    fn ocr_engine_ids_and_lang_validation() {
633        for id in ["ppocr", "PP-OCR", " rapidocr ", "default"] {
634            assert_eq!(OcrEngine::parse(id), Some(OcrEngine::PpOcr), "{id:?}");
635        }
636        for id in ["tesseract", "Tesseract_CLI", "tesserocr"] {
637            assert_eq!(OcrEngine::parse(id), Some(OcrEngine::Tesseract), "{id:?}");
638        }
639        assert_eq!(OcrEngine::parse("easyocr"), None);
640        assert!(OcrEngine::PpOcr.validate_lang("en").is_ok());
641        assert!(OcrEngine::PpOcr.validate_lang("deu").is_err());
642        assert!(OcrEngine::Tesseract.validate_lang("en").is_ok());
643        assert!(OcrEngine::Tesseract.validate_lang("deu+fra").is_ok());
644        assert!(OcrEngine::Tesseract.validate_lang("xx").is_err());
645    }
646}