Skip to main content

docling_pdf/
tesseract.rs

1//! Tesseract as an alternative OCR engine (#460) — the system `tesseract`
2//! binary driven like docling's `TesseractOcrCliModel`: a subprocess per
3//! crop, `tsv` output on stdout, no bindings and no build-time dependency.
4//! Selected with `DOCLING_RS_OCR_ENGINE=tesseract` / `--ocr-engine tesseract`;
5//! PP-OCRv3 stays the default and the conformance engine.
6//!
7//! It plugs into the pipeline where the PP-OCR recognizer does — the same
8//! layout-region crops, the same [`TextCell`] output in page points — so
9//! everything downstream (orphan recovery, TableFormer word matching, the page
10//! `ocr_score`) is engine-agnostic. Differences that follow from the engine:
11//!
12//! - Tesseract segments its own lines and words inside a region crop, so the
13//!   projection-profile line splitter (`ocr_prep`) is not used; a region's
14//!   cells are its `tsv` lines (words joined by single spaces), a table's
15//!   cells its words — the granularity each consumer expects.
16//! - Orientation (#225) comes from Tesseract's own OSD (`--psm 0 -l osd`,
17//!   needs the `osd` traineddata) instead of the recognize-four-ways probe.
18//! - Languages are tessdata stems (`eng`, `deu+fra`, `script/Cyrillic`), not
19//!   the en/ch model switch — see [`lang_arg`] for what `ocr_lang` accepts
20//!   under this engine.
21//!
22//! Every crop runs as its own `tesseract` process, dealt across the worker's
23//! OCR lanes (`intra` threads, `DOCLING_RS_OCR_SESSIONS` overrides); the
24//! child is pinned to one OpenMP thread so the lanes, not libgomp, own the
25//! parallelism. Environment: `DOCLING_TESSERACT` (the binary; default
26//! `tesseract` on `PATH`), `DOCLING_RS_TESSERACT_PSM` (page segmentation
27//! mode 0–13, unset = Tesseract's default), `DOCLING_RS_TESSDATA_DIR`
28//! (`--tessdata-dir`; `TESSDATA_PREFIX` is honored by Tesseract itself).
29//! Runs are deterministic (Tesseract is; the results are placed by crop
30//! index), so pinned outputs stay stable.
31
32use std::io::Write;
33use std::process::{Command, Stdio};
34
35use image::{imageops, RgbImage};
36
37use crate::layout::Region;
38use crate::ocr_prep::is_text_label;
39use crate::pdfium_backend::TextCell;
40use docling_core::debug_log;
41
42/// How the `tesseract` binary is invoked.
43#[derive(Debug, Clone, PartialEq, Eq)]
44pub struct TesseractOptions {
45    /// Command or path of the executable (`DOCLING_TESSERACT`, default
46    /// `tesseract`).
47    pub cmd: String,
48    /// The `-l` argument — tessdata stems joined with `+`, already mapped by
49    /// [`lang_arg`]. `None` runs Tesseract's own default (`eng`).
50    pub lang: Option<String>,
51    /// Page segmentation mode (`--psm`, `DOCLING_RS_TESSERACT_PSM`, 0–13).
52    /// `None` keeps Tesseract's default (3, fully automatic) — the right
53    /// choice for layout-region crops; 6 (one uniform block) or 11 (sparse
54    /// text) help on odd inputs.
55    pub psm: Option<u8>,
56    /// `--tessdata-dir` (`DOCLING_RS_TESSDATA_DIR`); `None` leaves the lookup
57    /// to Tesseract (`TESSDATA_PREFIX`, then its build-time default).
58    pub tessdata_dir: Option<String>,
59}
60
61impl TesseractOptions {
62    /// The process-level options: `lang` from the caller (the mapped
63    /// `ocr_lang`), the rest from the environment. A `DOCLING_RS_TESSERACT_PSM`
64    /// outside 0–13 warns and is ignored.
65    pub fn from_env(lang: Option<String>) -> Self {
66        let psm = docling_core::env::nonempty("DOCLING_RS_TESSERACT_PSM").and_then(|raw| match raw
67            .trim()
68            .parse::<u8>()
69        {
70            Ok(n) if n <= 13 => Some(n),
71            _ => {
72                eprintln!(
73                    "docling-pdf: DOCLING_RS_TESSERACT_PSM={raw:?} is not a page \
74                         segmentation mode 0-13; using Tesseract's default"
75                );
76                None
77            }
78        });
79        Self {
80            cmd: docling_core::env::nonempty("DOCLING_TESSERACT")
81                .unwrap_or_else(|| "tesseract".to_string()),
82            lang,
83            psm,
84            tessdata_dir: docling_core::env::nonempty("DOCLING_RS_TESSDATA_DIR"),
85        }
86    }
87}
88
89/// Tesseract's `-l` argument for an `ocr_lang` value under this engine.
90///
91/// `+`-separated parts, each either a tessdata stem handed over verbatim —
92/// `eng`, `chi_sim`, `srp_latn`, `script/Cyrillic`, a traineddata file of your
93/// own (`[A-Za-z0-9_/][A-Za-z0-9_/-]*`, the identifier grammar docling's CLI
94/// model sanitizes with) — or a BCP-47 tag mapped onto the stem Tesseract
95/// names it by (its vocabulary is ISO 639-2/T: `de` → `deu`, `fr` → `fra`,
96/// `zh-Hans` → `chi_sim`, `zh-TW` → `chi_tra`, `sr-Latn` → `srp_latn`).
97/// A part is a tag when its primary subtag is two letters (docling's `iso:`
98/// prefix is accepted and optional, as it is for the PP-OCR engine, #388);
99/// region subtags are dropped, script subtags pick the variant. The PP-OCR
100/// codes map too (`en` → `eng`, `ch` → `chi_sim`, docling's legacy
101/// `english`/`chinese` and EasyOCR's `ch_sim`/`ch_tra`), so switching the
102/// engine never invalidates an `ocr_lang` that worked. Multiple languages
103/// are Tesseract's preference order. Unknown two-letter tags and malformed
104/// stems are errors (the message says what to write instead); whether a stem
105/// is *installed* is checked when the engine loads.
106pub fn lang_arg(raw: &str) -> Result<String, String> {
107    let mut stems = Vec::new();
108    for part in raw.split('+') {
109        let part = part.trim();
110        let part = part
111            .strip_prefix("iso:")
112            .or_else(|| part.strip_prefix("ISO:"))
113            .map_or(part, |tag| tag.trim());
114        if part.is_empty() {
115            return Err(format!("ocr_lang {raw:?} has an empty language entry"));
116        }
117        let mut subtags = part.split(['-', '_']);
118        let primary = subtags.next().unwrap_or_default();
119        let stem = if primary.len() == 2 && primary.bytes().all(|b| b.is_ascii_alphabetic()) {
120            let rest: Vec<String> = subtags.map(str::to_ascii_lowercase).collect();
121            bcp47_stem(&primary.to_ascii_lowercase(), &rest).ok_or_else(|| {
122                format!(
123                    "ocr_lang {part:?}: no Tesseract traineddata name is known for that \
124                     BCP-47 tag; give the tessdata stem directly (e.g. `deu`, `chi_sim`, \
125                     `script/Latin` — see `tesseract --list-langs`)"
126                )
127            })?
128        } else {
129            match part.to_ascii_lowercase().as_str() {
130                "english" => "eng".to_string(),
131                "chinese" => "chi_sim".to_string(),
132                "chinese_cht" => "chi_tra".to_string(),
133                _ => {
134                    let mut chars = part.chars();
135                    let head_ok = chars
136                        .next()
137                        .is_some_and(|c| c.is_ascii_alphanumeric() || c == '_' || c == '/');
138                    let tail_ok =
139                        chars.all(|c| c.is_ascii_alphanumeric() || matches!(c, '_' | '/' | '-'));
140                    if !(head_ok && tail_ok) {
141                        return Err(format!(
142                            "ocr_lang {part:?} is not a Tesseract language identifier \
143                             (letters, digits, `_`, `/`, `-`; e.g. `eng`, `deu+fra`, \
144                             `script/Cyrillic`)"
145                        ));
146                    }
147                    part.to_string()
148                }
149            }
150        };
151        if !stems.contains(&stem) {
152            stems.push(stem);
153        }
154    }
155    Ok(stems.join("+"))
156}
157
158/// The tessdata stem for a two-letter primary subtag plus its remaining
159/// subtags (lower-cased). Tesseract's names are ISO 639-2/T with a handful of
160/// deviations (docling's `_TESSERACT_CANONICAL_TO_CODE_DEVIATIONS`); the
161/// table covers the languages with a two-letter code among the traineddata
162/// files Tesseract ships.
163fn bcp47_stem(primary: &str, subtags: &[String]) -> Option<String> {
164    let has = |s: &str| subtags.iter().any(|t| t == s);
165    let stem = match primary {
166        // Script-dependent stems.
167        "zh" | "ch" => {
168            if has("hant") || has("tw") || has("hk") || has("mo") || has("tra") {
169                "chi_tra"
170            } else {
171                "chi_sim"
172            }
173        }
174        "sr" => {
175            if has("latn") {
176                "srp_latn"
177            } else {
178                "srp"
179            }
180        }
181        "az" => {
182            if has("cyrl") {
183                "aze_cyrl"
184            } else {
185                "aze"
186            }
187        }
188        "uz" => {
189            if has("cyrl") {
190                "uzb_cyrl"
191            } else {
192                "uzb"
193            }
194        }
195        "de" => {
196            if has("latf") {
197                "deu_latf"
198            } else {
199                "deu"
200            }
201        }
202        "ku" => "kmr",
203        "no" | "nb" | "nn" => "nor",
204        "af" => "afr",
205        "am" => "amh",
206        "ar" => "ara",
207        "as" => "asm",
208        "be" => "bel",
209        "bn" => "ben",
210        "bo" => "bod",
211        "bs" => "bos",
212        "br" => "bre",
213        "bg" => "bul",
214        "ca" => "cat",
215        "cs" => "ces",
216        "co" => "cos",
217        "cy" => "cym",
218        "da" => "dan",
219        "dv" => "div",
220        "dz" => "dzo",
221        "el" => "ell",
222        "en" => "eng",
223        "eo" => "epo",
224        "et" => "est",
225        "eu" => "eus",
226        "fo" => "fao",
227        "fa" => "fas",
228        "fi" => "fin",
229        "fr" => "fra",
230        "fy" => "fry",
231        "gd" => "gla",
232        "ga" => "gle",
233        "gl" => "glg",
234        "gu" => "guj",
235        "ht" => "hat",
236        "he" => "heb",
237        "hi" => "hin",
238        "hr" => "hrv",
239        "hu" => "hun",
240        "hy" => "hye",
241        "iu" => "iku",
242        "id" => "ind",
243        "is" => "isl",
244        "it" => "ita",
245        "jv" => "jav",
246        "ja" => "jpn",
247        "kn" => "kan",
248        "ka" => "kat",
249        "kk" => "kaz",
250        "km" => "khm",
251        "ky" => "kir",
252        "ko" => "kor",
253        "lo" => "lao",
254        "la" => "lat",
255        "lv" => "lav",
256        "lt" => "lit",
257        "lb" => "ltz",
258        "ml" => "mal",
259        "mr" => "mar",
260        "mk" => "mkd",
261        "mt" => "mlt",
262        "mn" => "mon",
263        "mi" => "mri",
264        "ms" => "msa",
265        "my" => "mya",
266        "ne" => "nep",
267        "nl" => "nld",
268        "oc" => "oci",
269        "or" => "ori",
270        "pa" => "pan",
271        "pl" => "pol",
272        "pt" => "por",
273        "ps" => "pus",
274        "qu" => "que",
275        "ro" => "ron",
276        "ru" => "rus",
277        "sa" => "san",
278        "si" => "sin",
279        "sk" => "slk",
280        "sl" => "slv",
281        "sd" => "snd",
282        "es" => "spa",
283        "sq" => "sqi",
284        "su" => "sun",
285        "sw" => "swa",
286        "sv" => "swe",
287        "ta" => "tam",
288        "tt" => "tat",
289        "te" => "tel",
290        "tg" => "tgk",
291        "th" => "tha",
292        "ti" => "tir",
293        "to" => "ton",
294        "tr" => "tur",
295        "ug" => "uig",
296        "uk" => "ukr",
297        "ur" => "urd",
298        "vi" => "vie",
299        "yi" => "yid",
300        "yo" => "yor",
301        _ => return None,
302    };
303    Some(stem.to_string())
304}
305
306/// One `tsv` word row (level 5): its position in Tesseract's block /
307/// paragraph / line hierarchy, its box in crop pixels and its confidence
308/// (0–100).
309#[derive(Debug, Clone, PartialEq)]
310struct Word {
311    block: u32,
312    par: u32,
313    line: u32,
314    l: f32,
315    t: f32,
316    r: f32,
317    b: f32,
318    conf: f32,
319    text: String,
320}
321
322/// Parse Tesseract's `tsv` output: the header names the columns (`level
323/// page_num block_num par_num line_num word_num left top width height conf
324/// text`); only level-5 rows with non-blank text are words — the block /
325/// paragraph / line rows carry no text and `conf -1`.
326fn parse_tsv(tsv: &str) -> Vec<Word> {
327    let mut lines = tsv.lines();
328    let Some(header) = lines.next() else {
329        return Vec::new();
330    };
331    let cols: Vec<&str> = header.split('\t').collect();
332    let col = |name: &str| cols.iter().position(|c| *c == name);
333    let (
334        Some(level),
335        Some(block),
336        Some(par),
337        Some(line),
338        Some(left),
339        Some(top),
340        Some(w),
341        Some(h),
342        Some(conf),
343        Some(text),
344    ) = (
345        col("level"),
346        col("block_num"),
347        col("par_num"),
348        col("line_num"),
349        col("left"),
350        col("top"),
351        col("width"),
352        col("height"),
353        col("conf"),
354        col("text"),
355    )
356    else {
357        return Vec::new();
358    };
359    let mut words = Vec::new();
360    for row in lines {
361        let f: Vec<&str> = row.split('\t').collect();
362        if f.len() <= text || f[level] != "5" {
363            continue;
364        }
365        let text = f[text].trim();
366        if text.is_empty() {
367            continue;
368        }
369        let num = |i: usize| f[i].trim().parse::<f32>().unwrap_or(0.0);
370        let idx = |i: usize| f[i].trim().parse::<u32>().unwrap_or(0);
371        let (l, t) = (num(left), num(top));
372        words.push(Word {
373            block: idx(block),
374            par: idx(par),
375            line: idx(line),
376            l,
377            t,
378            r: l + num(w),
379            b: t + num(h),
380            conf: num(conf).clamp(0.0, 100.0),
381            text: text.to_string(),
382        });
383    }
384    words
385}
386
387/// A recognized text unit in crop pixels: `(l, t, r, b, text, confidence)`,
388/// confidence on the 0–1 scale the page `ocr_score` expects.
389type Unit = (f32, f32, f32, f32, String, f32);
390
391/// Words → lines: consecutive words with the same block / paragraph / line
392/// index join with single spaces (the order `tsv` lists them in is reading
393/// order within the line), the box is their union and the confidence the
394/// mean of theirs.
395fn group_lines(words: &[Word]) -> Vec<Unit> {
396    let mut out: Vec<Unit> = Vec::new();
397    let mut key: Option<(u32, u32, u32)> = None;
398    let mut n = 0.0f32;
399    for w in words {
400        let k = (w.block, w.par, w.line);
401        if key == Some(k) {
402            let last = out.last_mut().expect("a line is open");
403            last.0 = last.0.min(w.l);
404            last.1 = last.1.min(w.t);
405            last.2 = last.2.max(w.r);
406            last.3 = last.3.max(w.b);
407            last.4.push(' ');
408            last.4.push_str(&w.text);
409            last.5 = (last.5 * n + w.conf / 100.0) / (n + 1.0);
410            n += 1.0;
411        } else {
412            key = Some(k);
413            n = 1.0;
414            out.push((w.l, w.t, w.r, w.b, w.text.clone(), w.conf / 100.0));
415        }
416    }
417    out
418}
419
420/// Each word on its own, for the table cell matcher.
421fn word_units(words: &[Word]) -> Vec<Unit> {
422    words
423        .iter()
424        .map(|w| (w.l, w.t, w.r, w.b, w.text.clone(), w.conf / 100.0))
425        .collect()
426}
427
428/// Minimum pixels of white paper added around a region crop: Tesseract's
429/// page segmentation wants text clear of the image border, and a layout box
430/// often sits on the ink.
431const CROP_PAD: u32 = 6;
432/// The pad also grows with the region ([`crop_pad`]): a layout box hugs the
433/// x-height/baseline band, and a diacritic — tilde, acute, cedilla — sits
434/// outside it by a fraction of the glyph size. At the PDF path's 2.0 px/pt
435/// render 6 px is a quarter of a 12 pt line and clears them; a 300 dpi PNG
436/// converts at scale 1.0 with 48 px glyphs, where the same 6 px clipped the
437/// tilde and the cedilla and Tesseract read "Acao" for "Ação" (found with
438/// #471's synthetic Portuguese scan). A quarter of the line (12 px) still
439/// landed on the cedilla's edge — its ink runs 12 px under that layout box —
440/// so 0.3 (14 px); the 12 pt PDF case goes from 6 to 7 px of white paper.
441const CROP_PAD_RATIO: f32 = 0.3;
442/// Cap on the grown pad, so a tall multi-line block does not drag whole
443/// neighbouring lines into its crop — and whatever partial neighbour the pad
444/// still admits is dropped by [`unit_in_region`], never a cell.
445const CROP_PAD_MAX: u32 = 48;
446
447/// The pad in image pixels for a region `h_px` tall: `CROP_PAD_RATIO` of the
448/// height, no less than [`CROP_PAD`], no more than [`CROP_PAD_MAX`].
449fn crop_pad(h_px: f32) -> u32 {
450    // `max` / `min` (not `clamp`) so a NaN height degrades to the minimum.
451    (h_px * CROP_PAD_RATIO)
452        .round()
453        .max(CROP_PAD as f32)
454        .min(CROP_PAD_MAX as f32) as u32
455}
456
457/// One region's crop for Tesseract.
458struct Crop {
459    /// The crop's origin in image pixels (its top-left corner on the page).
460    ox: u32,
461    oy: u32,
462    png: Vec<u8>,
463    /// The unpadded region box in image pixels, `(l, t, r, b)`: only units
464    /// centred inside it (plus [`CROP_PAD`] of slack for layout-box jitter)
465    /// count as this region's text — the pad exists to keep glyph extremities
466    /// whole, not to read the neighbours it lets in.
467    bounds: (f32, f32, f32, f32),
468}
469
470/// A region's padded crop as PNG bytes.
471fn crop_png(img: &RgbImage, region: &Region, scale: f32) -> Option<Crop> {
472    let (iw, ih) = img.dimensions();
473    let bounds = (
474        region.l * scale,
475        region.t * scale,
476        region.r * scale,
477        region.b * scale,
478    );
479    let pad = crop_pad(bounds.3 - bounds.1);
480    let l = (bounds.0.max(0.0) as u32).saturating_sub(pad);
481    let t = (bounds.1.max(0.0) as u32).saturating_sub(pad);
482    let r = ((bounds.2.max(0.0) as u32).saturating_add(pad)).min(iw);
483    let b = ((bounds.3.max(0.0) as u32).saturating_add(pad)).min(ih);
484    if r <= l || b <= t {
485        return None;
486    }
487    let crop = imageops::crop_imm(img, l, t, r - l, b - t).to_image();
488    Some(Crop {
489        ox: l,
490        oy: t,
491        png: encode_png(&crop)?,
492        bounds,
493    })
494}
495
496/// Whether a recognized unit (crop-relative px box) belongs to the crop's
497/// region: its centre lies inside the region box, with [`CROP_PAD`] of slack
498/// on every side. A neighbouring line the pad pulled into the crop has its
499/// centre outside and is discarded; a line the layout box merely sits a few
500/// pixels off from stays.
501fn unit_in_region(crop: &Crop, unit: &Unit) -> bool {
502    let (l, t, r, b, _, _) = unit;
503    let cx = crop.ox as f32 + (l + r) / 2.0;
504    let cy = crop.oy as f32 + (t + b) / 2.0;
505    let slack = CROP_PAD as f32;
506    let (bl, bt, br, bb) = crop.bounds;
507    cx >= bl - slack && cx <= br + slack && cy >= bt - slack && cy <= bb + slack
508}
509
510fn encode_png(img: &RgbImage) -> Option<Vec<u8>> {
511    let mut buf = std::io::Cursor::new(Vec::new());
512    img.write_to(&mut buf, image::ImageFormat::Png).ok()?;
513    Some(buf.into_inner())
514}
515
516/// The loaded engine: the validated options plus what `--list-langs` said.
517pub struct TesseractOcr {
518    opts: TesseractOptions,
519    /// Crops recognized concurrently (one child process each).
520    lanes: usize,
521    /// Whether the `osd` traineddata is installed — orientation detection
522    /// needs it.
523    has_osd: bool,
524}
525
526impl TesseractOcr {
527    /// Probe the binary (`--version`) and its languages (`--list-langs`),
528    /// checking every requested stem is installed. An error here is what the
529    /// pipeline shows as "OCR unavailable" (degrading like a missing model,
530    /// #244) — it names the binary, the missing stems and what is installed.
531    pub fn load(opts: TesseractOptions, lanes: usize) -> Result<Self, String> {
532        let lanes = docling_core::env::parse::<usize>("DOCLING_RS_OCR_SESSIONS")
533            .filter(|&n| n > 0)
534            .unwrap_or(lanes)
535            .clamp(1, 16);
536        let version = Command::new(&opts.cmd)
537            .arg("--version")
538            .stdin(Stdio::null())
539            .output()
540            .map_err(|e| {
541                format!(
542                    "tesseract binary {:?} not runnable ({e}); install tesseract-ocr or point \
543                     DOCLING_TESSERACT at it",
544                    opts.cmd
545                )
546            })?;
547        // Linux prints the version on stdout, older Windows builds on stderr.
548        let banner = [version.stdout, version.stderr]
549            .iter()
550            .map(|b| String::from_utf8_lossy(b).trim().to_string())
551            .find(|s| !s.is_empty())
552            .unwrap_or_default();
553        let banner = banner.lines().next().unwrap_or_default().to_string();
554        let mut list = Command::new(&opts.cmd);
555        list.arg("--list-langs").stdin(Stdio::null());
556        if let Some(dir) = &opts.tessdata_dir {
557            list.arg("--tessdata-dir").arg(dir);
558        }
559        let list = list
560            .output()
561            .map_err(|e| format!("tesseract --list-langs failed to run: {e}"))?;
562        // The header line ("List of available languages in ... (N):") is
563        // followed by one stem per line; Windows spells script packs with a
564        // backslash, `-l` wants the slash everywhere.
565        let installed: Vec<String> = String::from_utf8_lossy(&list.stdout)
566            .lines()
567            .skip(1)
568            .map(|l| l.trim().replace('\\', "/"))
569            .filter(|l| !l.is_empty())
570            .collect();
571        if installed.is_empty() {
572            return Err(format!(
573                "{banner}: no traineddata found (tesseract --list-langs printed nothing); \
574                 install a language pack (e.g. tesseract-ocr-eng) or set \
575                 DOCLING_RS_TESSDATA_DIR / TESSDATA_PREFIX"
576            ));
577        }
578        let wanted: Vec<&str> = match &opts.lang {
579            Some(l) => l.split('+').collect(),
580            None => vec!["eng"],
581        };
582        let missing: Vec<&str> = wanted
583            .iter()
584            .copied()
585            .filter(|w| !installed.iter().any(|i| i == w))
586            .collect();
587        if !missing.is_empty() {
588            return Err(format!(
589                "{banner}: no traineddata for {} (installed: {}); install the language \
590                 pack or pick an installed one with ocr_lang",
591                missing.join(", "),
592                installed.join(", ")
593            ));
594        }
595        let has_osd = installed.iter().any(|i| i == "osd");
596        debug_log!(
597            "docling-pdf: OCR engine {banner} (lang {}, psm {}, {} lane(s), osd {})",
598            opts.lang.as_deref().unwrap_or("eng"),
599            opts.psm.map_or("default".to_string(), |p| p.to_string()),
600            lanes,
601            if has_osd { "yes" } else { "no" }
602        );
603        Ok(Self {
604            opts,
605            lanes,
606            has_osd,
607        })
608    }
609
610    /// A child process with the common arguments: the binary, the language,
611    /// the data directory, and one OpenMP thread (the lanes own the
612    /// parallelism; libgomp's default of one thread per core per process
613    /// would oversubscribe every core `lanes` times — and Tesseract is not
614    /// faster multi-threaded on a small crop anyway).
615    fn command(&self) -> Command {
616        let mut cmd = Command::new(&self.opts.cmd);
617        cmd.env("OMP_THREAD_LIMIT", "1")
618            .stdin(Stdio::piped())
619            .stdout(Stdio::piped())
620            .stderr(Stdio::null());
621        if let Some(dir) = &self.opts.tessdata_dir {
622            cmd.arg("--tessdata-dir").arg(dir);
623        }
624        cmd
625    }
626
627    /// Feed `png` to a child on stdin and return its stdout. A child that
628    /// fails (a crop Tesseract rejects) is an `Err` the caller degrades
629    /// per crop.
630    fn run(&self, mut cmd: Command, png: &[u8]) -> Result<String, String> {
631        let mut child = cmd
632            .spawn()
633            .map_err(|e| format!("tesseract: spawn {:?}: {e}", self.opts.cmd))?;
634        // Write on a thread: a crop bigger than the pipe buffer would
635        // otherwise deadlock against a child that has started printing.
636        let mut stdin = child.stdin.take().expect("piped stdin");
637        let png = png.to_vec();
638        let writer = std::thread::spawn(move || stdin.write_all(&png));
639        let out = child
640            .wait_with_output()
641            .map_err(|e| format!("tesseract: wait: {e}"))?;
642        let _ = writer.join();
643        if !out.status.success() {
644            return Err(format!("tesseract exited with {}", out.status));
645        }
646        Ok(String::from_utf8_lossy(&out.stdout).into_owned())
647    }
648
649    /// Recognize one crop: `tesseract stdin stdout [-l L] [--psm N] --dpi D tsv`.
650    fn recognize(&self, png: &[u8], dpi: u32) -> Result<Vec<Word>, String> {
651        let mut cmd = self.command();
652        if let Some(lang) = &self.opts.lang {
653            cmd.arg("-l").arg(lang);
654        }
655        if let Some(psm) = self.opts.psm {
656            cmd.arg("--psm").arg(psm.to_string());
657        }
658        cmd.arg("--dpi").arg(dpi.to_string());
659        cmd.args(["stdin", "stdout", "tsv"]);
660        Ok(parse_tsv(&self.run(cmd, png)?))
661    }
662
663    /// Recognize every crop of `jobs` across the lanes, and return each
664    /// crop's units mapped to page points (`origin + px` / `scale`), in job
665    /// order; units centred outside the crop's region ([`unit_in_region`])
666    /// are dropped. A crop Tesseract fails on contributes nothing (logged
667    /// under `DOCLING_RS_DEBUG`), the rest of the page still reads out.
668    fn recognize_all(
669        &self,
670        jobs: &[Crop],
671        scale: f32,
672        units: fn(&[Word]) -> Vec<Unit>,
673    ) -> Vec<(TextCell, f32)> {
674        let dpi = (scale * 72.0).round().max(1.0) as u32;
675        let lanes = self.lanes.min(jobs.len()).max(1);
676        let next = std::sync::atomic::AtomicUsize::new(0);
677        let results: Vec<std::sync::Mutex<Option<Vec<Word>>>> =
678            jobs.iter().map(|_| std::sync::Mutex::new(None)).collect();
679        std::thread::scope(|s| {
680            for _ in 0..lanes {
681                s.spawn(|| loop {
682                    let i = next.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
683                    let Some(job) = jobs.get(i) else {
684                        break;
685                    };
686                    let words = match self.recognize(&job.png, dpi) {
687                        Ok(words) => words,
688                        Err(e) => {
689                            debug_log!("docling-pdf: tesseract: crop {i}: {e}; no text");
690                            Vec::new()
691                        }
692                    };
693                    *results[i].lock().expect("crop result slot") = Some(words);
694                });
695            }
696        });
697        let mut cells = Vec::new();
698        for (job, slot) in jobs.iter().zip(results) {
699            let words = slot
700                .into_inner()
701                .expect("crop result slot")
702                .unwrap_or_default();
703            let (ox, oy) = (job.ox as f32, job.oy as f32);
704            for unit in units(&words) {
705                if !unit_in_region(job, &unit) {
706                    continue;
707                }
708                let (l, t, r, b, text, conf) = unit;
709                let text = text.trim().to_string();
710                if text.is_empty() {
711                    continue;
712                }
713                cells.push((
714                    TextCell {
715                        text,
716                        l: (ox + l) / scale,
717                        t: (oy + t) / scale,
718                        r: (ox + r) / scale,
719                        b: (oy + b) / scale,
720                    },
721                    conf,
722                ));
723            }
724        }
725        cells
726    }
727
728    /// The crops of the regions `keep` selects.
729    fn crops(img: &RgbImage, regions: &[Region], scale: f32, keep: fn(&str) -> bool) -> Vec<Crop> {
730        regions
731            .iter()
732            .filter(|r| keep(r.label))
733            .filter_map(|r| crop_png(img, r, scale))
734            .collect()
735    }
736
737    /// OCR a page's text regions into line cells (page points) with their
738    /// confidence — the counterpart of [`crate::ocr::OcrModel::ocr_page`].
739    /// `scale` is image px per page point.
740    pub fn ocr_page(
741        &mut self,
742        img: &RgbImage,
743        regions: &[Region],
744        scale: f32,
745    ) -> Result<Vec<(TextCell, f32)>, String> {
746        let jobs = crate::timing::timed("ocr.prep", || {
747            Self::crops(img, regions, scale, is_text_label)
748        });
749        Ok(crate::timing::timed("ocr.rec", || {
750            self.recognize_all(&jobs, scale, group_lines)
751        }))
752    }
753
754    /// Word cells inside the page's table regions, for the cell matcher — the
755    /// counterpart of [`crate::ocr::OcrModel::ocr_table_words`].
756    pub fn ocr_table_words(
757        &mut self,
758        img: &RgbImage,
759        regions: &[Region],
760        scale: f32,
761    ) -> Result<Vec<(TextCell, f32)>, String> {
762        let jobs = Self::crops(img, regions, scale, crate::assemble::is_table_like);
763        Ok(self.recognize_all(&jobs, scale, word_units))
764    }
765
766    /// The clockwise angle the page content is rotated by in `img`
767    /// (`0`/`90`/`180`/`270`, the [`crate::orient::detect`] convention), from
768    /// Tesseract's orientation-and-script detection (`--psm 0 -l osd`).
769    /// `None` when the `osd` traineddata is not installed or OSD cannot make
770    /// a call (too little text) — the page then stays as rendered.
771    pub fn detect_orientation(&self, img: &RgbImage, scale: f32) -> Option<u16> {
772        if !self.has_osd {
773            debug_log!("docling-pdf: tesseract: no `osd` traineddata; orientation not probed");
774            return None;
775        }
776        let png = encode_png(img)?;
777        let mut cmd = self.command();
778        cmd.args(["--psm", "0", "-l", "osd", "--dpi"])
779            .arg(((scale * 72.0).round().max(1.0) as u32).to_string())
780            .args(["stdin", "stdout"]);
781        match self.run(cmd, &png) {
782            Ok(out) => parse_osd(&out),
783            Err(e) => {
784                debug_log!("docling-pdf: tesseract OSD failed ({e}); assuming upright");
785                None
786            }
787        }
788    }
789}
790
791/// The `Orientation in degrees:` line of an OSD report — Tesseract's clockwise
792/// angle of the text, the same convention as the probe's `deg` (its `Rotate:`
793/// line is the complementary correction).
794fn parse_osd(out: &str) -> Option<u16> {
795    let deg = out
796        .lines()
797        .find_map(|l| l.trim().strip_prefix("Orientation in degrees:"))?
798        .trim()
799        .parse::<u16>()
800        .ok()?;
801    matches!(deg, 0 | 90 | 180 | 270).then_some(deg)
802}
803
804#[cfg(test)]
805mod tests {
806    use super::*;
807
808    /// `ocr_lang` under the Tesseract engine: stems verbatim, BCP-47 tags and
809    /// the PP-OCR codes mapped, lists joined in order, junk rejected.
810    #[test]
811    fn lang_arg_maps_tags_and_keeps_stems() {
812        for (raw, want) in [
813            ("eng", "eng"),
814            ("en", "eng"),
815            ("EN-us", "eng"),
816            ("iso:en", "eng"),
817            ("english", "eng"),
818            ("ch", "chi_sim"),
819            ("ch_tra", "chi_tra"),
820            ("chinese_cht", "chi_tra"),
821            ("zh", "chi_sim"),
822            ("zh-Hant", "chi_tra"),
823            ("zh-TW", "chi_tra"),
824            ("iso:zh-CN", "chi_sim"),
825            ("de", "deu"),
826            ("de-Latf", "deu_latf"),
827            ("sr-Latn", "srp_latn"),
828            ("sr", "srp"),
829            ("nb", "nor"),
830            ("deu+fra", "deu+fra"),
831            ("iso:de + fr", "deu+fra"),
832            ("eng+eng", "eng"),
833            ("script/Cyrillic", "script/Cyrillic"),
834            ("chi_sim", "chi_sim"),
835            ("srp_latn", "srp_latn"),
836            ("custom_model", "custom_model"),
837        ] {
838            assert_eq!(lang_arg(raw).as_deref(), Ok(want), "{raw:?}");
839        }
840        for raw in ["", "xx", "en+", "de;u", "a b", "iso:", "deu fra"] {
841            assert!(lang_arg(raw).is_err(), "{raw:?} should be rejected");
842        }
843    }
844
845    /// The crop pad follows the glyph size: the 6 px floor for a 10 pt line
846    /// at the PDF path's 2.0 px/pt render, 14 px for a 48 px line of a
847    /// 300 dpi image at scale 1.0 — where 6 px clipped the tilde and the
848    /// cedilla and 12 px still grazed the cedilla — and capped for a tall
849    /// block.
850    #[test]
851    fn crop_pad_grows_with_the_region_and_is_capped() {
852        assert_eq!(crop_pad(20.0), CROP_PAD);
853        assert_eq!(crop_pad(48.0), 14);
854        assert_eq!(crop_pad(1000.0), CROP_PAD_MAX);
855        assert_eq!(crop_pad(f32::NAN), CROP_PAD);
856        let region = Region {
857            label: "text",
858            score: 0.9,
859            l: 100.0,
860            t: 100.0,
861            r: 300.0,
862            b: 148.0,
863        };
864        let img = RgbImage::from_pixel(400, 400, image::Rgb([255, 255, 255]));
865        let crop = crop_png(&img, &region, 1.0).expect("crop");
866        assert_eq!((crop.ox, crop.oy), (86, 86), "14 px pad on a 48 px line");
867        assert_eq!(crop.bounds, (100.0, 100.0, 300.0, 148.0));
868        let crop = crop_png(&img, &region, 0.4).expect("crop");
869        assert_eq!((crop.ox, crop.oy), (34, 34), "6 px floor on a 19 px line");
870    }
871
872    /// A neighbouring line the grown pad admits into the crop is not this
873    /// region's cell; a line whose layout box sits a few pixels off still is.
874    #[test]
875    fn units_outside_the_region_are_dropped() {
876        let crop = Crop {
877            ox: 86,
878            oy: 86,
879            png: Vec::new(),
880            bounds: (100.0, 100.0, 300.0, 148.0),
881        };
882        // Crop-relative: the region's own line spans y 14..62 inside the crop.
883        let own: Unit = (14.0, 14.0, 214.0, 62.0, "Ação".into(), 0.9);
884        // Bleeding 4 px past the box top (layout jitter) — still inside.
885        let jittered: Unit = (14.0, 6.0, 214.0, 58.0, "Ação".into(), 0.9);
886        // The next line down, starting where the region ends: centre outside.
887        let neighbour: Unit = (14.0, 62.0, 214.0, 110.0, "próxima".into(), 0.9);
888        assert!(unit_in_region(&crop, &own));
889        assert!(unit_in_region(&crop, &jittered));
890        assert!(!unit_in_region(&crop, &neighbour));
891    }
892
893    /// The `tsv` shape Tesseract 5 prints: only level-5 rows are words; a
894    /// block's two lines become two cells, words joined by single spaces,
895    /// boxes unioned, confidence averaged onto 0–1; table mode keeps words.
896    #[test]
897    fn tsv_words_group_into_lines() {
898        let tsv = "level\tpage_num\tblock_num\tpar_num\tline_num\tword_num\tleft\ttop\twidth\theight\tconf\ttext\n\
899                   1\t1\t0\t0\t0\t0\t0\t0\t600\t160\t-1\t\n\
900                   4\t1\t1\t1\t1\t0\t23\t28\t242\t25\t-1\t\n\
901                   5\t1\t1\t1\t1\t1\t23\t28\t64\t20\t93.2\tHello\n\
902                   5\t1\t1\t1\t1\t2\t94\t28\t93\t25\t92.4\tdocling\n\
903                   5\t1\t1\t1\t1\t3\t195\t28\t70\t20\t96.0\tworld\n\
904                   5\t1\t2\t1\t1\t1\t21\t98\t94\t20\t96.8\tSecond\n\
905                   5\t1\t2\t1\t1\t2\t125\t98\t44\t20\t50\t \n\
906                   5\t1\t2\t1\t1\t3\t179\t99\t44\t19\t96.8\t123\n";
907        let words = parse_tsv(tsv);
908        assert_eq!(words.len(), 5);
909        let lines = group_lines(&words);
910        assert_eq!(lines.len(), 2);
911        let (l, t, r, b, text, conf) = &lines[0];
912        assert_eq!(text, "Hello docling world");
913        assert_eq!((*l, *t, *r, *b), (23.0, 28.0, 265.0, 53.0));
914        assert!((conf - 0.9387).abs() < 1e-3, "{conf}");
915        assert_eq!(lines[1].4, "Second 123");
916        assert_eq!(word_units(&words).len(), 5);
917        assert!(parse_tsv("").is_empty());
918        assert!(parse_tsv("garbage\n5\t1\n").is_empty());
919    }
920
921    #[test]
922    fn osd_orientation_parses() {
923        let out = "Page number: 0\nOrientation in degrees: 270\nRotate: 90\n\
924                   Orientation confidence: 6.47\nScript: Latin\nScript confidence: 4.05\n";
925        assert_eq!(parse_osd(out), Some(270));
926        assert_eq!(parse_osd("Orientation in degrees: 0\n"), Some(0));
927        assert_eq!(parse_osd("Orientation in degrees: 45\n"), None);
928        assert_eq!(parse_osd("Too few characters. Skipping this page\n"), None);
929    }
930
931    #[test]
932    fn psm_env_is_range_checked() {
933        let opts = TesseractOptions::from_env(Some("eng".into()));
934        assert_eq!(opts.lang.as_deref(), Some("eng"));
935        assert!(opts.psm.is_none() || opts.psm.is_some_and(|p| p <= 13));
936    }
937}