Skip to main content

docling_pdf/
tesseract.rs

1//! Tesseract as an alternative OCR engine (#460) — the system `tesseract`
2//! binary driven like docling's `TesseractOcrCliModel`: a subprocess per
3//! crop, `tsv` output on stdout, no bindings and no build-time dependency.
4//! Selected with `DOCLING_RS_OCR_ENGINE=tesseract` / `--ocr-engine tesseract`;
5//! PP-OCRv3 stays the default and the conformance engine.
6//!
7//! It plugs into the pipeline where the PP-OCR recognizer does — the same
8//! layout-region crops, the same [`TextCell`] output in page points — so
9//! everything downstream (orphan recovery, TableFormer word matching, the page
10//! `ocr_score`) is engine-agnostic. Differences that follow from the engine:
11//!
12//! - Tesseract segments its own lines and words inside a region crop, so the
13//!   projection-profile line splitter (`ocr_prep`) is not used; a region's
14//!   cells are its `tsv` lines (words joined by single spaces), a table's
15//!   cells its words — the granularity each consumer expects.
16//! - Orientation (#225) comes from Tesseract's own OSD (`--psm 0 -l osd`,
17//!   needs the `osd` traineddata) instead of the recognize-four-ways probe.
18//! - Languages are tessdata stems (`eng`, `deu+fra`, `script/Cyrillic`), not
19//!   the en/ch model switch — see [`lang_arg`] for what `ocr_lang` accepts
20//!   under this engine.
21//!
22//! Every crop runs as its own `tesseract` process, dealt across the worker's
23//! OCR lanes (`intra` threads, `DOCLING_RS_OCR_SESSIONS` overrides); the
24//! child is pinned to one OpenMP thread so the lanes, not libgomp, own the
25//! parallelism. Environment: `DOCLING_TESSERACT` (the binary; default
26//! `tesseract` on `PATH`), `DOCLING_RS_TESSERACT_PSM` (page segmentation
27//! mode 0–13, unset = Tesseract's default), `DOCLING_RS_TESSDATA_DIR`
28//! (`--tessdata-dir`; `TESSDATA_PREFIX` is honored by Tesseract itself).
29//! Runs are deterministic (Tesseract is; the results are placed by crop
30//! index), so pinned outputs stay stable.
31
32use std::io::Write;
33use std::process::{Command, Stdio};
34
35use image::{imageops, RgbImage};
36
37use crate::layout::Region;
38use crate::ocr_prep::is_text_label;
39use crate::pdfium_backend::TextCell;
40use docling_core::debug_log;
41
42/// How the `tesseract` binary is invoked.
43#[derive(Debug, Clone, PartialEq, Eq)]
44pub struct TesseractOptions {
45    /// Command or path of the executable (`DOCLING_TESSERACT`, default
46    /// `tesseract`).
47    pub cmd: String,
48    /// The `-l` argument — tessdata stems joined with `+`, already mapped by
49    /// [`lang_arg`]. `None` runs Tesseract's own default (`eng`).
50    pub lang: Option<String>,
51    /// Page segmentation mode (`--psm`, `DOCLING_RS_TESSERACT_PSM`, 0–13).
52    /// `None` keeps Tesseract's default (3, fully automatic) — the right
53    /// choice for layout-region crops; 6 (one uniform block) or 11 (sparse
54    /// text) help on odd inputs.
55    pub psm: Option<u8>,
56    /// `--tessdata-dir` (`DOCLING_RS_TESSDATA_DIR`); `None` leaves the lookup
57    /// to Tesseract (`TESSDATA_PREFIX`, then its build-time default).
58    pub tessdata_dir: Option<String>,
59}
60
61impl TesseractOptions {
62    /// The process-level options: `lang` from the caller (the mapped
63    /// `ocr_lang`), the rest from the environment. A `DOCLING_RS_TESSERACT_PSM`
64    /// outside 0–13 warns and is ignored.
65    pub fn from_env(lang: Option<String>) -> Self {
66        let psm = docling_core::env::nonempty("DOCLING_RS_TESSERACT_PSM").and_then(|raw| {
67            let psm = parse_psm(&raw);
68            if psm.is_none() {
69                eprintln!(
70                    "docling-pdf: DOCLING_RS_TESSERACT_PSM={raw:?} is not a page \
71                         segmentation mode 0-13; using Tesseract's default"
72                );
73            }
74            psm
75        });
76        Self {
77            cmd: docling_core::env::nonempty("DOCLING_TESSERACT")
78                .unwrap_or_else(|| "tesseract".to_string()),
79            lang,
80            psm,
81            tessdata_dir: docling_core::env::nonempty("DOCLING_RS_TESSDATA_DIR"),
82        }
83    }
84}
85
86/// A `DOCLING_RS_TESSERACT_PSM` value: a page segmentation mode 0–13, or
87/// `None` for anything else.
88fn parse_psm(raw: &str) -> Option<u8> {
89    raw.trim().parse::<u8>().ok().filter(|&n| n <= 13)
90}
91
92/// Tesseract's `-l` argument for an `ocr_lang` value under this engine.
93///
94/// `+`-separated parts, each either a tessdata stem handed over verbatim —
95/// `eng`, `chi_sim`, `srp_latn`, `script/Cyrillic`, a traineddata file of your
96/// own (`[A-Za-z0-9_/][A-Za-z0-9_/-]*`, the identifier grammar docling's CLI
97/// model sanitizes with) — or a BCP-47 tag mapped onto the stem Tesseract
98/// names it by (its vocabulary is ISO 639-2/T: `de` → `deu`, `fr` → `fra`,
99/// `zh-Hans` → `chi_sim`, `zh-TW` → `chi_tra`, `sr-Latn` → `srp_latn`).
100/// A part is a tag when its primary subtag is two letters (docling's `iso:`
101/// prefix is accepted and optional, as it is for the PP-OCR engine, #388);
102/// region subtags are dropped, script subtags pick the variant. The PP-OCR
103/// codes map too (`en` → `eng`, `ch` → `chi_sim`, docling's legacy
104/// `english`/`chinese` and EasyOCR's `ch_sim`/`ch_tra`), so switching the
105/// engine never invalidates an `ocr_lang` that worked. Multiple languages
106/// are Tesseract's preference order. Unknown two-letter tags and malformed
107/// stems are errors (the message says what to write instead); whether a stem
108/// is *installed* is checked when the engine loads.
109pub fn lang_arg(raw: &str) -> Result<String, String> {
110    let mut stems = Vec::new();
111    for part in raw.split('+') {
112        let part = part.trim();
113        let part = part
114            .strip_prefix("iso:")
115            .or_else(|| part.strip_prefix("ISO:"))
116            .map_or(part, |tag| tag.trim());
117        if part.is_empty() {
118            return Err(format!("ocr_lang {raw:?} has an empty language entry"));
119        }
120        let mut subtags = part.split(['-', '_']);
121        let primary = subtags.next().unwrap_or_default();
122        let stem = if primary.len() == 2 && primary.bytes().all(|b| b.is_ascii_alphabetic()) {
123            let rest: Vec<String> = subtags.map(str::to_ascii_lowercase).collect();
124            bcp47_stem(&primary.to_ascii_lowercase(), &rest).ok_or_else(|| {
125                format!(
126                    "ocr_lang {part:?}: no Tesseract traineddata name is known for that \
127                     BCP-47 tag; give the tessdata stem directly (e.g. `deu`, `chi_sim`, \
128                     `script/Latin` — see `tesseract --list-langs`)"
129                )
130            })?
131        } else {
132            match part.to_ascii_lowercase().as_str() {
133                "english" => "eng".to_string(),
134                "chinese" => "chi_sim".to_string(),
135                "chinese_cht" => "chi_tra".to_string(),
136                _ => {
137                    let mut chars = part.chars();
138                    let head_ok = chars
139                        .next()
140                        .is_some_and(|c| c.is_ascii_alphanumeric() || c == '_' || c == '/');
141                    let tail_ok =
142                        chars.all(|c| c.is_ascii_alphanumeric() || matches!(c, '_' | '/' | '-'));
143                    if !(head_ok && tail_ok) {
144                        return Err(format!(
145                            "ocr_lang {part:?} is not a Tesseract language identifier \
146                             (letters, digits, `_`, `/`, `-`; e.g. `eng`, `deu+fra`, \
147                             `script/Cyrillic`)"
148                        ));
149                    }
150                    part.to_string()
151                }
152            }
153        };
154        if !stems.contains(&stem) {
155            stems.push(stem);
156        }
157    }
158    Ok(stems.join("+"))
159}
160
161/// The tessdata stem for a two-letter primary subtag plus its remaining
162/// subtags (lower-cased). Tesseract's names are ISO 639-2/T with a handful of
163/// deviations (docling's `_TESSERACT_CANONICAL_TO_CODE_DEVIATIONS`); the
164/// table covers the languages with a two-letter code among the traineddata
165/// files Tesseract ships.
166fn bcp47_stem(primary: &str, subtags: &[String]) -> Option<String> {
167    let has = |s: &str| subtags.iter().any(|t| t == s);
168    let stem = match primary {
169        // Script-dependent stems.
170        "zh" | "ch" => {
171            if has("hant") || has("tw") || has("hk") || has("mo") || has("tra") {
172                "chi_tra"
173            } else {
174                "chi_sim"
175            }
176        }
177        "sr" => {
178            if has("latn") {
179                "srp_latn"
180            } else {
181                "srp"
182            }
183        }
184        "az" => {
185            if has("cyrl") {
186                "aze_cyrl"
187            } else {
188                "aze"
189            }
190        }
191        "uz" => {
192            if has("cyrl") {
193                "uzb_cyrl"
194            } else {
195                "uzb"
196            }
197        }
198        "de" => {
199            if has("latf") {
200                "deu_latf"
201            } else {
202                "deu"
203            }
204        }
205        "ku" => "kmr",
206        "no" | "nb" | "nn" => "nor",
207        "af" => "afr",
208        "am" => "amh",
209        "ar" => "ara",
210        "as" => "asm",
211        "be" => "bel",
212        "bn" => "ben",
213        "bo" => "bod",
214        "bs" => "bos",
215        "br" => "bre",
216        "bg" => "bul",
217        "ca" => "cat",
218        "cs" => "ces",
219        "co" => "cos",
220        "cy" => "cym",
221        "da" => "dan",
222        "dv" => "div",
223        "dz" => "dzo",
224        "el" => "ell",
225        "en" => "eng",
226        "eo" => "epo",
227        "et" => "est",
228        "eu" => "eus",
229        "fo" => "fao",
230        "fa" => "fas",
231        "fi" => "fin",
232        "fr" => "fra",
233        "fy" => "fry",
234        "gd" => "gla",
235        "ga" => "gle",
236        "gl" => "glg",
237        "gu" => "guj",
238        "ht" => "hat",
239        "he" => "heb",
240        "hi" => "hin",
241        "hr" => "hrv",
242        "hu" => "hun",
243        "hy" => "hye",
244        "iu" => "iku",
245        "id" => "ind",
246        "is" => "isl",
247        "it" => "ita",
248        "jv" => "jav",
249        "ja" => "jpn",
250        "kn" => "kan",
251        "ka" => "kat",
252        "kk" => "kaz",
253        "km" => "khm",
254        "ky" => "kir",
255        "ko" => "kor",
256        "lo" => "lao",
257        "la" => "lat",
258        "lv" => "lav",
259        "lt" => "lit",
260        "lb" => "ltz",
261        "ml" => "mal",
262        "mr" => "mar",
263        "mk" => "mkd",
264        "mt" => "mlt",
265        "mn" => "mon",
266        "mi" => "mri",
267        "ms" => "msa",
268        "my" => "mya",
269        "ne" => "nep",
270        "nl" => "nld",
271        "oc" => "oci",
272        "or" => "ori",
273        "pa" => "pan",
274        "pl" => "pol",
275        "pt" => "por",
276        "ps" => "pus",
277        "qu" => "que",
278        "ro" => "ron",
279        "ru" => "rus",
280        "sa" => "san",
281        "si" => "sin",
282        "sk" => "slk",
283        "sl" => "slv",
284        "sd" => "snd",
285        "es" => "spa",
286        "sq" => "sqi",
287        "su" => "sun",
288        "sw" => "swa",
289        "sv" => "swe",
290        "ta" => "tam",
291        "tt" => "tat",
292        "te" => "tel",
293        "tg" => "tgk",
294        "th" => "tha",
295        "ti" => "tir",
296        "to" => "ton",
297        "tr" => "tur",
298        "ug" => "uig",
299        "uk" => "ukr",
300        "ur" => "urd",
301        "vi" => "vie",
302        "yi" => "yid",
303        "yo" => "yor",
304        _ => return None,
305    };
306    Some(stem.to_string())
307}
308
309/// One `tsv` word row (level 5): its position in Tesseract's block /
310/// paragraph / line hierarchy, its box in crop pixels and its confidence
311/// (0–100).
312#[derive(Debug, Clone, PartialEq)]
313struct Word {
314    block: u32,
315    par: u32,
316    line: u32,
317    l: f32,
318    t: f32,
319    r: f32,
320    b: f32,
321    conf: f32,
322    text: String,
323}
324
325/// Parse Tesseract's `tsv` output: the header names the columns (`level
326/// page_num block_num par_num line_num word_num left top width height conf
327/// text`); only level-5 rows with non-blank text are words — the block /
328/// paragraph / line rows carry no text and `conf -1`.
329fn parse_tsv(tsv: &str) -> Vec<Word> {
330    let mut lines = tsv.lines();
331    let Some(header) = lines.next() else {
332        return Vec::new();
333    };
334    let cols: Vec<&str> = header.split('\t').collect();
335    let col = |name: &str| cols.iter().position(|c| *c == name);
336    let (
337        Some(level),
338        Some(block),
339        Some(par),
340        Some(line),
341        Some(left),
342        Some(top),
343        Some(w),
344        Some(h),
345        Some(conf),
346        Some(text),
347    ) = (
348        col("level"),
349        col("block_num"),
350        col("par_num"),
351        col("line_num"),
352        col("left"),
353        col("top"),
354        col("width"),
355        col("height"),
356        col("conf"),
357        col("text"),
358    )
359    else {
360        return Vec::new();
361    };
362    let mut words = Vec::new();
363    for row in lines {
364        let f: Vec<&str> = row.split('\t').collect();
365        if f.len() <= text || f[level] != "5" {
366            continue;
367        }
368        let text = f[text].trim();
369        if text.is_empty() {
370            continue;
371        }
372        let num = |i: usize| f[i].trim().parse::<f32>().unwrap_or(0.0);
373        let idx = |i: usize| f[i].trim().parse::<u32>().unwrap_or(0);
374        let (l, t) = (num(left), num(top));
375        words.push(Word {
376            block: idx(block),
377            par: idx(par),
378            line: idx(line),
379            l,
380            t,
381            r: l + num(w),
382            b: t + num(h),
383            conf: num(conf).clamp(0.0, 100.0),
384            text: text.to_string(),
385        });
386    }
387    words
388}
389
390/// A recognized text unit in crop pixels: `(l, t, r, b, text, confidence)`,
391/// confidence on the 0–1 scale the page `ocr_score` expects.
392type Unit = (f32, f32, f32, f32, String, f32);
393
394/// Words → lines: consecutive words with the same block / paragraph / line
395/// index join with single spaces (the order `tsv` lists them in is reading
396/// order within the line), the box is their union and the confidence the
397/// mean of theirs.
398fn group_lines(words: &[Word]) -> Vec<Unit> {
399    let mut out: Vec<Unit> = Vec::new();
400    let mut key: Option<(u32, u32, u32)> = None;
401    let mut n = 0.0f32;
402    for w in words {
403        let k = (w.block, w.par, w.line);
404        if key == Some(k) {
405            let last = out.last_mut().expect("a line is open");
406            last.0 = last.0.min(w.l);
407            last.1 = last.1.min(w.t);
408            last.2 = last.2.max(w.r);
409            last.3 = last.3.max(w.b);
410            last.4.push(' ');
411            last.4.push_str(&w.text);
412            last.5 = (last.5 * n + w.conf / 100.0) / (n + 1.0);
413            n += 1.0;
414        } else {
415            key = Some(k);
416            n = 1.0;
417            out.push((w.l, w.t, w.r, w.b, w.text.clone(), w.conf / 100.0));
418        }
419    }
420    out
421}
422
423/// Each word on its own, for the table cell matcher.
424fn word_units(words: &[Word]) -> Vec<Unit> {
425    words
426        .iter()
427        .map(|w| (w.l, w.t, w.r, w.b, w.text.clone(), w.conf / 100.0))
428        .collect()
429}
430
431/// Minimum pixels of white paper added around a region crop: Tesseract's
432/// page segmentation wants text clear of the image border, and a layout box
433/// often sits on the ink.
434const CROP_PAD: u32 = 6;
435/// The pad also grows with the region ([`crop_pad`]): a layout box hugs the
436/// x-height/baseline band, and a diacritic — tilde, acute, cedilla — sits
437/// outside it by a fraction of the glyph size. At the PDF path's 2.0 px/pt
438/// render 6 px is a quarter of a 12 pt line and clears them; a 300 dpi PNG
439/// converts at scale 1.0 with 48 px glyphs, where the same 6 px clipped the
440/// tilde and the cedilla and Tesseract read "Acao" for "Ação" (found with
441/// #471's synthetic Portuguese scan). A quarter of the line (12 px) still
442/// landed on the cedilla's edge — its ink runs 12 px under that layout box —
443/// so 0.3 (14 px); the 12 pt PDF case goes from 6 to 7 px of white paper.
444const CROP_PAD_RATIO: f32 = 0.3;
445/// Cap on the grown pad, so a tall multi-line block does not drag whole
446/// neighbouring lines into its crop — and whatever partial neighbour the pad
447/// still admits is dropped by [`unit_in_region`], never a cell.
448const CROP_PAD_MAX: u32 = 48;
449
450/// The pad in image pixels for a region `h_px` tall: `CROP_PAD_RATIO` of the
451/// height, no less than [`CROP_PAD`], no more than [`CROP_PAD_MAX`].
452fn crop_pad(h_px: f32) -> u32 {
453    // `max` / `min` (not `clamp`) so a NaN height degrades to the minimum.
454    (h_px * CROP_PAD_RATIO)
455        .round()
456        .max(CROP_PAD as f32)
457        .min(CROP_PAD_MAX as f32) as u32
458}
459
460/// One region's crop for Tesseract.
461struct Crop {
462    /// The crop's origin in image pixels (its top-left corner on the page).
463    ox: u32,
464    oy: u32,
465    png: Vec<u8>,
466    /// The unpadded region box in image pixels, `(l, t, r, b)`: only units
467    /// centred inside it (plus [`CROP_PAD`] of slack for layout-box jitter)
468    /// count as this region's text — the pad exists to keep glyph extremities
469    /// whole, not to read the neighbours it lets in.
470    bounds: (f32, f32, f32, f32),
471}
472
473/// A region's padded crop as PNG bytes.
474fn crop_png(img: &RgbImage, region: &Region, scale: f32) -> Option<Crop> {
475    let (iw, ih) = img.dimensions();
476    let bounds = (
477        region.l * scale,
478        region.t * scale,
479        region.r * scale,
480        region.b * scale,
481    );
482    let pad = crop_pad(bounds.3 - bounds.1);
483    let l = (bounds.0.max(0.0) as u32).saturating_sub(pad);
484    let t = (bounds.1.max(0.0) as u32).saturating_sub(pad);
485    let r = ((bounds.2.max(0.0) as u32).saturating_add(pad)).min(iw);
486    let b = ((bounds.3.max(0.0) as u32).saturating_add(pad)).min(ih);
487    if r <= l || b <= t {
488        return None;
489    }
490    let crop = imageops::crop_imm(img, l, t, r - l, b - t).to_image();
491    Some(Crop {
492        ox: l,
493        oy: t,
494        png: encode_png(&crop)?,
495        bounds,
496    })
497}
498
499/// Whether a recognized unit (crop-relative px box) belongs to the crop's
500/// region: its centre lies inside the region box, with [`CROP_PAD`] of slack
501/// on every side. A neighbouring line the pad pulled into the crop has its
502/// centre outside and is discarded; a line the layout box merely sits a few
503/// pixels off from stays.
504fn unit_in_region(crop: &Crop, unit: &Unit) -> bool {
505    let (l, t, r, b, _, _) = unit;
506    let cx = crop.ox as f32 + (l + r) / 2.0;
507    let cy = crop.oy as f32 + (t + b) / 2.0;
508    let slack = CROP_PAD as f32;
509    let (bl, bt, br, bb) = crop.bounds;
510    cx >= bl - slack && cx <= br + slack && cy >= bt - slack && cy <= bb + slack
511}
512
513fn encode_png(img: &RgbImage) -> Option<Vec<u8>> {
514    let mut buf = std::io::Cursor::new(Vec::new());
515    img.write_to(&mut buf, image::ImageFormat::Png).ok()?;
516    Some(buf.into_inner())
517}
518
519/// The loaded engine: the validated options plus what `--list-langs` said.
520pub struct TesseractOcr {
521    opts: TesseractOptions,
522    /// Crops recognized concurrently (one child process each).
523    lanes: usize,
524    /// Whether the `osd` traineddata is installed — orientation detection
525    /// needs it.
526    has_osd: bool,
527}
528
529impl TesseractOcr {
530    /// Probe the binary (`--version`) and its languages (`--list-langs`),
531    /// checking every requested stem is installed. An error here is what the
532    /// pipeline shows as "OCR unavailable" (degrading like a missing model,
533    /// #244) — it names the binary, the missing stems and what is installed.
534    pub fn load(opts: TesseractOptions, lanes: usize) -> Result<Self, String> {
535        let lanes = docling_core::env::parse::<usize>("DOCLING_RS_OCR_SESSIONS")
536            .filter(|&n| n > 0)
537            .unwrap_or(lanes)
538            .clamp(1, 16);
539        let version = Command::new(&opts.cmd)
540            .arg("--version")
541            .stdin(Stdio::null())
542            .output()
543            .map_err(|e| {
544                format!(
545                    "tesseract binary {:?} not runnable ({e}); install tesseract-ocr or point \
546                     DOCLING_TESSERACT at it",
547                    opts.cmd
548                )
549            })?;
550        // Linux prints the version on stdout, older Windows builds on stderr.
551        let banner = [version.stdout, version.stderr]
552            .iter()
553            .map(|b| String::from_utf8_lossy(b).trim().to_string())
554            .find(|s| !s.is_empty())
555            .unwrap_or_default();
556        let banner = banner.lines().next().unwrap_or_default().to_string();
557        let mut list = Command::new(&opts.cmd);
558        list.arg("--list-langs").stdin(Stdio::null());
559        if let Some(dir) = &opts.tessdata_dir {
560            list.arg("--tessdata-dir").arg(dir);
561        }
562        let list = list
563            .output()
564            .map_err(|e| format!("tesseract --list-langs failed to run: {e}"))?;
565        // The header line ("List of available languages in ... (N):") is
566        // followed by one stem per line; Windows spells script packs with a
567        // backslash, `-l` wants the slash everywhere.
568        let installed: Vec<String> = String::from_utf8_lossy(&list.stdout)
569            .lines()
570            .skip(1)
571            .map(|l| l.trim().replace('\\', "/"))
572            .filter(|l| !l.is_empty())
573            .collect();
574        if installed.is_empty() {
575            return Err(format!(
576                "{banner}: no traineddata found (tesseract --list-langs printed nothing); \
577                 install a language pack (e.g. tesseract-ocr-eng) or set \
578                 DOCLING_RS_TESSDATA_DIR / TESSDATA_PREFIX"
579            ));
580        }
581        let wanted: Vec<&str> = match &opts.lang {
582            Some(l) => l.split('+').collect(),
583            None => vec!["eng"],
584        };
585        let missing: Vec<&str> = wanted
586            .iter()
587            .copied()
588            .filter(|w| !installed.iter().any(|i| i == w))
589            .collect();
590        if !missing.is_empty() {
591            return Err(format!(
592                "{banner}: no traineddata for {} (installed: {}); install the language \
593                 pack or pick an installed one with ocr_lang",
594                missing.join(", "),
595                installed.join(", ")
596            ));
597        }
598        let has_osd = installed.iter().any(|i| i == "osd");
599        debug_log!(
600            "docling-pdf: OCR engine {banner} (lang {}, psm {}, {} lane(s), osd {})",
601            opts.lang.as_deref().unwrap_or("eng"),
602            opts.psm.map_or("default".to_string(), |p| p.to_string()),
603            lanes,
604            if has_osd { "yes" } else { "no" }
605        );
606        Ok(Self {
607            opts,
608            lanes,
609            has_osd,
610        })
611    }
612
613    /// A child process with the common arguments: the binary, the language,
614    /// the data directory, and one OpenMP thread (the lanes own the
615    /// parallelism; libgomp's default of one thread per core per process
616    /// would oversubscribe every core `lanes` times — and Tesseract is not
617    /// faster multi-threaded on a small crop anyway).
618    fn command(&self) -> Command {
619        let mut cmd = Command::new(&self.opts.cmd);
620        cmd.env("OMP_THREAD_LIMIT", "1")
621            .stdin(Stdio::piped())
622            .stdout(Stdio::piped())
623            .stderr(Stdio::null());
624        if let Some(dir) = &self.opts.tessdata_dir {
625            cmd.arg("--tessdata-dir").arg(dir);
626        }
627        cmd
628    }
629
630    /// Feed `png` to a child on stdin and return its stdout. A child that
631    /// fails (a crop Tesseract rejects) is an `Err` the caller degrades
632    /// per crop.
633    fn run(&self, mut cmd: Command, png: &[u8]) -> Result<String, String> {
634        let mut child = cmd
635            .spawn()
636            .map_err(|e| format!("tesseract: spawn {:?}: {e}", self.opts.cmd))?;
637        // Write on a thread: a crop bigger than the pipe buffer would
638        // otherwise deadlock against a child that has started printing.
639        let mut stdin = child.stdin.take().expect("piped stdin");
640        let png = png.to_vec();
641        let writer = std::thread::spawn(move || stdin.write_all(&png));
642        let out = child
643            .wait_with_output()
644            .map_err(|e| format!("tesseract: wait: {e}"))?;
645        let _ = writer.join();
646        if !out.status.success() {
647            return Err(format!("tesseract exited with {}", out.status));
648        }
649        Ok(String::from_utf8_lossy(&out.stdout).into_owned())
650    }
651
652    /// Recognize one crop: `tesseract stdin stdout [-l L] [--psm N] --dpi D tsv`.
653    fn recognize(&self, png: &[u8], dpi: u32) -> Result<Vec<Word>, String> {
654        let mut cmd = self.command();
655        if let Some(lang) = &self.opts.lang {
656            cmd.arg("-l").arg(lang);
657        }
658        if let Some(psm) = self.opts.psm {
659            cmd.arg("--psm").arg(psm.to_string());
660        }
661        cmd.arg("--dpi").arg(dpi.to_string());
662        cmd.args(["stdin", "stdout", "tsv"]);
663        Ok(parse_tsv(&self.run(cmd, png)?))
664    }
665
666    /// Recognize every crop of `jobs` across the lanes, and return each
667    /// crop's units mapped to page points (`origin + px` / `scale`), in job
668    /// order; units centred outside the crop's region ([`unit_in_region`])
669    /// are dropped. A crop Tesseract fails on contributes nothing (logged
670    /// under `DOCLING_RS_DEBUG`), the rest of the page still reads out.
671    fn recognize_all(
672        &self,
673        jobs: &[Crop],
674        scale: f32,
675        units: fn(&[Word]) -> Vec<Unit>,
676    ) -> Vec<(TextCell, f32)> {
677        let dpi = (scale * 72.0).round().max(1.0) as u32;
678        let lanes = self.lanes.min(jobs.len()).max(1);
679        let next = std::sync::atomic::AtomicUsize::new(0);
680        let results: Vec<std::sync::Mutex<Option<Vec<Word>>>> =
681            jobs.iter().map(|_| std::sync::Mutex::new(None)).collect();
682        std::thread::scope(|s| {
683            for _ in 0..lanes {
684                s.spawn(|| loop {
685                    let i = next.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
686                    let Some(job) = jobs.get(i) else {
687                        break;
688                    };
689                    let words = match self.recognize(&job.png, dpi) {
690                        Ok(words) => words,
691                        Err(e) => {
692                            debug_log!("docling-pdf: tesseract: crop {i}: {e}; no text");
693                            Vec::new()
694                        }
695                    };
696                    *results[i].lock().expect("crop result slot") = Some(words);
697                });
698            }
699        });
700        let mut cells = Vec::new();
701        for (job, slot) in jobs.iter().zip(results) {
702            let words = slot
703                .into_inner()
704                .expect("crop result slot")
705                .unwrap_or_default();
706            let (ox, oy) = (job.ox as f32, job.oy as f32);
707            for unit in units(&words) {
708                if !unit_in_region(job, &unit) {
709                    continue;
710                }
711                let (l, t, r, b, text, conf) = unit;
712                let text = text.trim().to_string();
713                if text.is_empty() {
714                    continue;
715                }
716                cells.push((
717                    TextCell {
718                        text,
719                        l: (ox + l) / scale,
720                        t: (oy + t) / scale,
721                        r: (ox + r) / scale,
722                        b: (oy + b) / scale,
723                    },
724                    conf,
725                ));
726            }
727        }
728        cells
729    }
730
731    /// The crops of the regions `keep` selects.
732    fn crops(img: &RgbImage, regions: &[Region], scale: f32, keep: fn(&str) -> bool) -> Vec<Crop> {
733        regions
734            .iter()
735            .filter(|r| keep(r.label))
736            .filter_map(|r| crop_png(img, r, scale))
737            .collect()
738    }
739
740    /// OCR a page's text regions into line cells (page points) with their
741    /// confidence — the counterpart of [`crate::ocr::OcrModel::ocr_page`].
742    /// `scale` is image px per page point.
743    pub fn ocr_page(
744        &mut self,
745        img: &RgbImage,
746        regions: &[Region],
747        scale: f32,
748    ) -> Result<Vec<(TextCell, f32)>, String> {
749        let jobs = crate::timing::timed("ocr.prep", || {
750            Self::crops(img, regions, scale, is_text_label)
751        });
752        Ok(crate::timing::timed("ocr.rec", || {
753            self.recognize_all(&jobs, scale, group_lines)
754        }))
755    }
756
757    /// Word cells inside the page's table regions, for the cell matcher — the
758    /// counterpart of [`crate::ocr::OcrModel::ocr_table_words`].
759    pub fn ocr_table_words(
760        &mut self,
761        img: &RgbImage,
762        regions: &[Region],
763        scale: f32,
764    ) -> Result<Vec<(TextCell, f32)>, String> {
765        let jobs = Self::crops(img, regions, scale, crate::assemble::is_table_like);
766        Ok(self.recognize_all(&jobs, scale, word_units))
767    }
768
769    /// The clockwise angle the page content is rotated by in `img`
770    /// (`0`/`90`/`180`/`270`, the [`crate::orient::detect`] convention), from
771    /// Tesseract's orientation-and-script detection (`--psm 0 -l osd`).
772    /// `None` when the `osd` traineddata is not installed or OSD cannot make
773    /// a call (too little text) — the page then stays as rendered.
774    pub fn detect_orientation(&self, img: &RgbImage, scale: f32) -> Option<u16> {
775        if !self.has_osd {
776            debug_log!("docling-pdf: tesseract: no `osd` traineddata; orientation not probed");
777            return None;
778        }
779        let png = encode_png(img)?;
780        let mut cmd = self.command();
781        cmd.args(["--psm", "0", "-l", "osd", "--dpi"])
782            .arg(((scale * 72.0).round().max(1.0) as u32).to_string())
783            .args(["stdin", "stdout"]);
784        match self.run(cmd, &png) {
785            Ok(out) => parse_osd(&out),
786            Err(e) => {
787                debug_log!("docling-pdf: tesseract OSD failed ({e}); assuming upright");
788                None
789            }
790        }
791    }
792}
793
794/// The `Orientation in degrees:` line of an OSD report — Tesseract's clockwise
795/// angle of the text, the same convention as the probe's `deg` (its `Rotate:`
796/// line is the complementary correction).
797fn parse_osd(out: &str) -> Option<u16> {
798    let deg = out
799        .lines()
800        .find_map(|l| l.trim().strip_prefix("Orientation in degrees:"))?
801        .trim()
802        .parse::<u16>()
803        .ok()?;
804    matches!(deg, 0 | 90 | 180 | 270).then_some(deg)
805}
806
807#[cfg(test)]
808mod tests {
809    use super::*;
810
811    /// `ocr_lang` under the Tesseract engine: stems verbatim, BCP-47 tags and
812    /// the PP-OCR codes mapped, lists joined in order, junk rejected.
813    #[test]
814    fn lang_arg_maps_tags_and_keeps_stems() {
815        for (raw, want) in [
816            ("eng", "eng"),
817            ("en", "eng"),
818            ("EN-us", "eng"),
819            ("iso:en", "eng"),
820            ("english", "eng"),
821            ("ch", "chi_sim"),
822            ("ch_tra", "chi_tra"),
823            ("chinese_cht", "chi_tra"),
824            ("zh", "chi_sim"),
825            ("zh-Hant", "chi_tra"),
826            ("zh-TW", "chi_tra"),
827            ("iso:zh-CN", "chi_sim"),
828            ("de", "deu"),
829            ("de-Latf", "deu_latf"),
830            ("sr-Latn", "srp_latn"),
831            ("sr", "srp"),
832            ("nb", "nor"),
833            ("deu+fra", "deu+fra"),
834            ("iso:de + fr", "deu+fra"),
835            ("eng+eng", "eng"),
836            ("script/Cyrillic", "script/Cyrillic"),
837            ("chi_sim", "chi_sim"),
838            ("srp_latn", "srp_latn"),
839            ("custom_model", "custom_model"),
840        ] {
841            assert_eq!(lang_arg(raw).as_deref(), Ok(want), "{raw:?}");
842        }
843        for raw in ["", "xx", "en+", "de;u", "a b", "iso:", "deu fra"] {
844            assert!(lang_arg(raw).is_err(), "{raw:?} should be rejected");
845        }
846    }
847
848    /// The crop pad follows the glyph size: the 6 px floor for a 10 pt line
849    /// at the PDF path's 2.0 px/pt render, 14 px for a 48 px line of a
850    /// 300 dpi image at scale 1.0 — where 6 px clipped the tilde and the
851    /// cedilla and 12 px still grazed the cedilla — and capped for a tall
852    /// block.
853    #[test]
854    fn crop_pad_grows_with_the_region_and_is_capped() {
855        assert_eq!(crop_pad(20.0), CROP_PAD);
856        assert_eq!(crop_pad(48.0), 14);
857        assert_eq!(crop_pad(1000.0), CROP_PAD_MAX);
858        assert_eq!(crop_pad(f32::NAN), CROP_PAD);
859        let region = Region {
860            label: "text",
861            score: 0.9,
862            l: 100.0,
863            t: 100.0,
864            r: 300.0,
865            b: 148.0,
866        };
867        let img = RgbImage::from_pixel(400, 400, image::Rgb([255, 255, 255]));
868        let crop = crop_png(&img, &region, 1.0).expect("crop");
869        assert_eq!((crop.ox, crop.oy), (86, 86), "14 px pad on a 48 px line");
870        assert_eq!(crop.bounds, (100.0, 100.0, 300.0, 148.0));
871        let crop = crop_png(&img, &region, 0.4).expect("crop");
872        assert_eq!((crop.ox, crop.oy), (34, 34), "6 px floor on a 19 px line");
873    }
874
875    /// A neighbouring line the grown pad admits into the crop is not this
876    /// region's cell; a line whose layout box sits a few pixels off still is.
877    #[test]
878    fn units_outside_the_region_are_dropped() {
879        let crop = Crop {
880            ox: 86,
881            oy: 86,
882            png: Vec::new(),
883            bounds: (100.0, 100.0, 300.0, 148.0),
884        };
885        // Crop-relative: the region's own line spans y 14..62 inside the crop.
886        let own: Unit = (14.0, 14.0, 214.0, 62.0, "Ação".into(), 0.9);
887        // Bleeding 4 px past the box top (layout jitter) — still inside.
888        let jittered: Unit = (14.0, 6.0, 214.0, 58.0, "Ação".into(), 0.9);
889        // The next line down, starting where the region ends: centre outside.
890        let neighbour: Unit = (14.0, 62.0, 214.0, 110.0, "próxima".into(), 0.9);
891        assert!(unit_in_region(&crop, &own));
892        assert!(unit_in_region(&crop, &jittered));
893        assert!(!unit_in_region(&crop, &neighbour));
894    }
895
896    /// The `tsv` shape Tesseract 5 prints: only level-5 rows are words; a
897    /// block's two lines become two cells, words joined by single spaces,
898    /// boxes unioned, confidence averaged onto 0–1; table mode keeps words.
899    #[test]
900    fn tsv_words_group_into_lines() {
901        let tsv = "level\tpage_num\tblock_num\tpar_num\tline_num\tword_num\tleft\ttop\twidth\theight\tconf\ttext\n\
902                   1\t1\t0\t0\t0\t0\t0\t0\t600\t160\t-1\t\n\
903                   4\t1\t1\t1\t1\t0\t23\t28\t242\t25\t-1\t\n\
904                   5\t1\t1\t1\t1\t1\t23\t28\t64\t20\t93.2\tHello\n\
905                   5\t1\t1\t1\t1\t2\t94\t28\t93\t25\t92.4\tdocling\n\
906                   5\t1\t1\t1\t1\t3\t195\t28\t70\t20\t96.0\tworld\n\
907                   5\t1\t2\t1\t1\t1\t21\t98\t94\t20\t96.8\tSecond\n\
908                   5\t1\t2\t1\t1\t2\t125\t98\t44\t20\t50\t \n\
909                   5\t1\t2\t1\t1\t3\t179\t99\t44\t19\t96.8\t123\n";
910        let words = parse_tsv(tsv);
911        assert_eq!(words.len(), 5);
912        let lines = group_lines(&words);
913        assert_eq!(lines.len(), 2);
914        let (l, t, r, b, text, conf) = &lines[0];
915        assert_eq!(text, "Hello docling world");
916        assert_eq!((*l, *t, *r, *b), (23.0, 28.0, 265.0, 53.0));
917        assert!((conf - 0.9387).abs() < 1e-3, "{conf}");
918        assert_eq!(lines[1].4, "Second 123");
919        assert_eq!(word_units(&words).len(), 5);
920        assert!(parse_tsv("").is_empty());
921        assert!(parse_tsv("garbage\n5\t1\n").is_empty());
922    }
923
924    #[test]
925    fn osd_orientation_parses() {
926        let out = "Page number: 0\nOrientation in degrees: 270\nRotate: 90\n\
927                   Orientation confidence: 6.47\nScript: Latin\nScript confidence: 4.05\n";
928        assert_eq!(parse_osd(out), Some(270));
929        assert_eq!(parse_osd("Orientation in degrees: 0\n"), Some(0));
930        assert_eq!(parse_osd("Orientation in degrees: 45\n"), None);
931        assert_eq!(parse_osd("Too few characters. Skipping this page\n"), None);
932    }
933
934    #[test]
935    fn psm_is_range_checked() {
936        assert_eq!(parse_psm("6"), Some(6));
937        assert_eq!(parse_psm(" 13 "), Some(13));
938        assert_eq!(parse_psm("0"), Some(0));
939        for bad in ["14", "-1", "300", "six", ""] {
940            assert_eq!(parse_psm(bad), None, "{bad:?}");
941        }
942    }
943}