Skip to main content

docling_pdf/
tesseract.rs

1//! Tesseract as an alternative OCR engine (#460) — the system `tesseract`
2//! binary driven like docling's `TesseractOcrCliModel`: a subprocess per
3//! crop, `tsv` output on stdout, no bindings and no build-time dependency.
4//! Selected with `DOCLING_RS_OCR_ENGINE=tesseract` / `--ocr-engine tesseract`;
5//! PP-OCRv3 stays the default and the conformance engine.
6//!
7//! It plugs into the pipeline where the PP-OCR recognizer does — the same
8//! layout-region crops, the same [`TextCell`] output in page points — so
9//! everything downstream (orphan recovery, TableFormer word matching, the page
10//! `ocr_score`) is engine-agnostic. Differences that follow from the engine:
11//!
12//! - Tesseract segments its own lines and words inside a region crop, so the
13//!   projection-profile line splitter (`ocr_prep`) is not used; a region's
14//!   cells are its `tsv` lines (words joined by single spaces), a table's
15//!   cells its words — the granularity each consumer expects.
16//! - Orientation (#225) comes from Tesseract's own OSD (`--psm 0 -l osd`,
17//!   needs the `osd` traineddata) instead of the recognize-four-ways probe.
18//! - Languages are tessdata stems (`eng`, `deu+fra`, `script/Cyrillic`), not
19//!   the en/ch model switch — see [`lang_arg`] for what `ocr_lang` accepts
20//!   under this engine.
21//!
22//! Every crop runs as its own `tesseract` process, dealt across the worker's
23//! OCR lanes (`intra` threads, `DOCLING_RS_OCR_SESSIONS` overrides); the
24//! child is pinned to one OpenMP thread so the lanes, not libgomp, own the
25//! parallelism. Environment: `DOCLING_TESSERACT` (the binary; default
26//! `tesseract` on `PATH`), `DOCLING_RS_TESSERACT_PSM` (page segmentation
27//! mode 0–13, unset = Tesseract's default), `DOCLING_RS_TESSDATA_DIR`
28//! (`--tessdata-dir`; `TESSDATA_PREFIX` is honored by Tesseract itself).
29//! Runs are deterministic (Tesseract is; the results are placed by crop
30//! index), so pinned outputs stay stable.
31
32use std::io::Write;
33use std::process::{Command, Stdio};
34
35use image::{imageops, RgbImage};
36
37use crate::layout::Region;
38use crate::ocr_prep::is_text_label;
39use crate::pdfium_backend::TextCell;
40use docling_core::debug_log;
41
42/// How the `tesseract` binary is invoked.
43#[derive(Debug, Clone, PartialEq, Eq)]
44pub struct TesseractOptions {
45    /// Command or path of the executable (`DOCLING_TESSERACT`, default
46    /// `tesseract`).
47    pub cmd: String,
48    /// The `-l` argument — tessdata stems joined with `+`, already mapped by
49    /// [`lang_arg`]. `None` runs Tesseract's own default (`eng`).
50    pub lang: Option<String>,
51    /// Page segmentation mode (`--psm`, `DOCLING_RS_TESSERACT_PSM`, 0–13).
52    /// `None` keeps Tesseract's default (3, fully automatic) — the right
53    /// choice for layout-region crops; 6 (one uniform block) or 11 (sparse
54    /// text) help on odd inputs.
55    pub psm: Option<u8>,
56    /// `--tessdata-dir` (`DOCLING_RS_TESSDATA_DIR`); `None` leaves the lookup
57    /// to Tesseract (`TESSDATA_PREFIX`, then its build-time default).
58    pub tessdata_dir: Option<String>,
59}
60
61impl TesseractOptions {
62    /// The process-level options: `lang` from the caller (the mapped
63    /// `ocr_lang`), the rest from the environment. A `DOCLING_RS_TESSERACT_PSM`
64    /// outside 0–13 warns and is ignored.
65    pub fn from_env(lang: Option<String>) -> Self {
66        let psm = docling_core::env::nonempty("DOCLING_RS_TESSERACT_PSM").and_then(|raw| match raw
67            .trim()
68            .parse::<u8>()
69        {
70            Ok(n) if n <= 13 => Some(n),
71            _ => {
72                eprintln!(
73                    "docling-pdf: DOCLING_RS_TESSERACT_PSM={raw:?} is not a page \
74                         segmentation mode 0-13; using Tesseract's default"
75                );
76                None
77            }
78        });
79        Self {
80            cmd: docling_core::env::nonempty("DOCLING_TESSERACT")
81                .unwrap_or_else(|| "tesseract".to_string()),
82            lang,
83            psm,
84            tessdata_dir: docling_core::env::nonempty("DOCLING_RS_TESSDATA_DIR"),
85        }
86    }
87}
88
89/// Tesseract's `-l` argument for an `ocr_lang` value under this engine.
90///
91/// `+`-separated parts, each either a tessdata stem handed over verbatim —
92/// `eng`, `chi_sim`, `srp_latn`, `script/Cyrillic`, a traineddata file of your
93/// own (`[A-Za-z0-9_/][A-Za-z0-9_/-]*`, the identifier grammar docling's CLI
94/// model sanitizes with) — or a BCP-47 tag mapped onto the stem Tesseract
95/// names it by (its vocabulary is ISO 639-2/T: `de` → `deu`, `fr` → `fra`,
96/// `zh-Hans` → `chi_sim`, `zh-TW` → `chi_tra`, `sr-Latn` → `srp_latn`).
97/// A part is a tag when its primary subtag is two letters (docling's `iso:`
98/// prefix is accepted and optional, as it is for the PP-OCR engine, #388);
99/// region subtags are dropped, script subtags pick the variant. The PP-OCR
100/// codes map too (`en` → `eng`, `ch` → `chi_sim`, docling's legacy
101/// `english`/`chinese` and EasyOCR's `ch_sim`/`ch_tra`), so switching the
102/// engine never invalidates an `ocr_lang` that worked. Multiple languages
103/// are Tesseract's preference order. Unknown two-letter tags and malformed
104/// stems are errors (the message says what to write instead); whether a stem
105/// is *installed* is checked when the engine loads.
106pub fn lang_arg(raw: &str) -> Result<String, String> {
107    let mut stems = Vec::new();
108    for part in raw.split('+') {
109        let part = part.trim();
110        let part = part
111            .strip_prefix("iso:")
112            .or_else(|| part.strip_prefix("ISO:"))
113            .map_or(part, |tag| tag.trim());
114        if part.is_empty() {
115            return Err(format!("ocr_lang {raw:?} has an empty language entry"));
116        }
117        let mut subtags = part.split(['-', '_']);
118        let primary = subtags.next().unwrap_or_default();
119        let stem = if primary.len() == 2 && primary.bytes().all(|b| b.is_ascii_alphabetic()) {
120            let rest: Vec<String> = subtags.map(str::to_ascii_lowercase).collect();
121            bcp47_stem(&primary.to_ascii_lowercase(), &rest).ok_or_else(|| {
122                format!(
123                    "ocr_lang {part:?}: no Tesseract traineddata name is known for that \
124                     BCP-47 tag; give the tessdata stem directly (e.g. `deu`, `chi_sim`, \
125                     `script/Latin` — see `tesseract --list-langs`)"
126                )
127            })?
128        } else {
129            match part.to_ascii_lowercase().as_str() {
130                "english" => "eng".to_string(),
131                "chinese" => "chi_sim".to_string(),
132                "chinese_cht" => "chi_tra".to_string(),
133                _ => {
134                    let mut chars = part.chars();
135                    let head_ok = chars
136                        .next()
137                        .is_some_and(|c| c.is_ascii_alphanumeric() || c == '_' || c == '/');
138                    let tail_ok =
139                        chars.all(|c| c.is_ascii_alphanumeric() || matches!(c, '_' | '/' | '-'));
140                    if !(head_ok && tail_ok) {
141                        return Err(format!(
142                            "ocr_lang {part:?} is not a Tesseract language identifier \
143                             (letters, digits, `_`, `/`, `-`; e.g. `eng`, `deu+fra`, \
144                             `script/Cyrillic`)"
145                        ));
146                    }
147                    part.to_string()
148                }
149            }
150        };
151        if !stems.contains(&stem) {
152            stems.push(stem);
153        }
154    }
155    Ok(stems.join("+"))
156}
157
158/// The tessdata stem for a two-letter primary subtag plus its remaining
159/// subtags (lower-cased). Tesseract's names are ISO 639-2/T with a handful of
160/// deviations (docling's `_TESSERACT_CANONICAL_TO_CODE_DEVIATIONS`); the
161/// table covers the languages with a two-letter code among the traineddata
162/// files Tesseract ships.
163fn bcp47_stem(primary: &str, subtags: &[String]) -> Option<String> {
164    let has = |s: &str| subtags.iter().any(|t| t == s);
165    let stem = match primary {
166        // Script-dependent stems.
167        "zh" | "ch" => {
168            if has("hant") || has("tw") || has("hk") || has("mo") || has("tra") {
169                "chi_tra"
170            } else {
171                "chi_sim"
172            }
173        }
174        "sr" => {
175            if has("latn") {
176                "srp_latn"
177            } else {
178                "srp"
179            }
180        }
181        "az" => {
182            if has("cyrl") {
183                "aze_cyrl"
184            } else {
185                "aze"
186            }
187        }
188        "uz" => {
189            if has("cyrl") {
190                "uzb_cyrl"
191            } else {
192                "uzb"
193            }
194        }
195        "de" => {
196            if has("latf") {
197                "deu_latf"
198            } else {
199                "deu"
200            }
201        }
202        "ku" => "kmr",
203        "no" | "nb" | "nn" => "nor",
204        "af" => "afr",
205        "am" => "amh",
206        "ar" => "ara",
207        "as" => "asm",
208        "be" => "bel",
209        "bn" => "ben",
210        "bo" => "bod",
211        "bs" => "bos",
212        "br" => "bre",
213        "bg" => "bul",
214        "ca" => "cat",
215        "cs" => "ces",
216        "co" => "cos",
217        "cy" => "cym",
218        "da" => "dan",
219        "dv" => "div",
220        "dz" => "dzo",
221        "el" => "ell",
222        "en" => "eng",
223        "eo" => "epo",
224        "et" => "est",
225        "eu" => "eus",
226        "fo" => "fao",
227        "fa" => "fas",
228        "fi" => "fin",
229        "fr" => "fra",
230        "fy" => "fry",
231        "gd" => "gla",
232        "ga" => "gle",
233        "gl" => "glg",
234        "gu" => "guj",
235        "ht" => "hat",
236        "he" => "heb",
237        "hi" => "hin",
238        "hr" => "hrv",
239        "hu" => "hun",
240        "hy" => "hye",
241        "iu" => "iku",
242        "id" => "ind",
243        "is" => "isl",
244        "it" => "ita",
245        "jv" => "jav",
246        "ja" => "jpn",
247        "kn" => "kan",
248        "ka" => "kat",
249        "kk" => "kaz",
250        "km" => "khm",
251        "ky" => "kir",
252        "ko" => "kor",
253        "lo" => "lao",
254        "la" => "lat",
255        "lv" => "lav",
256        "lt" => "lit",
257        "lb" => "ltz",
258        "ml" => "mal",
259        "mr" => "mar",
260        "mk" => "mkd",
261        "mt" => "mlt",
262        "mn" => "mon",
263        "mi" => "mri",
264        "ms" => "msa",
265        "my" => "mya",
266        "ne" => "nep",
267        "nl" => "nld",
268        "oc" => "oci",
269        "or" => "ori",
270        "pa" => "pan",
271        "pl" => "pol",
272        "pt" => "por",
273        "ps" => "pus",
274        "qu" => "que",
275        "ro" => "ron",
276        "ru" => "rus",
277        "sa" => "san",
278        "si" => "sin",
279        "sk" => "slk",
280        "sl" => "slv",
281        "sd" => "snd",
282        "es" => "spa",
283        "sq" => "sqi",
284        "su" => "sun",
285        "sw" => "swa",
286        "sv" => "swe",
287        "ta" => "tam",
288        "tt" => "tat",
289        "te" => "tel",
290        "tg" => "tgk",
291        "th" => "tha",
292        "ti" => "tir",
293        "to" => "ton",
294        "tr" => "tur",
295        "ug" => "uig",
296        "uk" => "ukr",
297        "ur" => "urd",
298        "vi" => "vie",
299        "yi" => "yid",
300        "yo" => "yor",
301        _ => return None,
302    };
303    Some(stem.to_string())
304}
305
306/// One `tsv` word row (level 5): its position in Tesseract's block /
307/// paragraph / line hierarchy, its box in crop pixels and its confidence
308/// (0–100).
309#[derive(Debug, Clone, PartialEq)]
310struct Word {
311    block: u32,
312    par: u32,
313    line: u32,
314    l: f32,
315    t: f32,
316    r: f32,
317    b: f32,
318    conf: f32,
319    text: String,
320}
321
322/// Parse Tesseract's `tsv` output: the header names the columns (`level
323/// page_num block_num par_num line_num word_num left top width height conf
324/// text`); only level-5 rows with non-blank text are words — the block /
325/// paragraph / line rows carry no text and `conf -1`.
326fn parse_tsv(tsv: &str) -> Vec<Word> {
327    let mut lines = tsv.lines();
328    let Some(header) = lines.next() else {
329        return Vec::new();
330    };
331    let cols: Vec<&str> = header.split('\t').collect();
332    let col = |name: &str| cols.iter().position(|c| *c == name);
333    let (
334        Some(level),
335        Some(block),
336        Some(par),
337        Some(line),
338        Some(left),
339        Some(top),
340        Some(w),
341        Some(h),
342        Some(conf),
343        Some(text),
344    ) = (
345        col("level"),
346        col("block_num"),
347        col("par_num"),
348        col("line_num"),
349        col("left"),
350        col("top"),
351        col("width"),
352        col("height"),
353        col("conf"),
354        col("text"),
355    )
356    else {
357        return Vec::new();
358    };
359    let mut words = Vec::new();
360    for row in lines {
361        let f: Vec<&str> = row.split('\t').collect();
362        if f.len() <= text || f[level] != "5" {
363            continue;
364        }
365        let text = f[text].trim();
366        if text.is_empty() {
367            continue;
368        }
369        let num = |i: usize| f[i].trim().parse::<f32>().unwrap_or(0.0);
370        let idx = |i: usize| f[i].trim().parse::<u32>().unwrap_or(0);
371        let (l, t) = (num(left), num(top));
372        words.push(Word {
373            block: idx(block),
374            par: idx(par),
375            line: idx(line),
376            l,
377            t,
378            r: l + num(w),
379            b: t + num(h),
380            conf: num(conf).clamp(0.0, 100.0),
381            text: text.to_string(),
382        });
383    }
384    words
385}
386
387/// A recognized text unit in crop pixels: `(l, t, r, b, text, confidence)`,
388/// confidence on the 0–1 scale the page `ocr_score` expects.
389type Unit = (f32, f32, f32, f32, String, f32);
390
391/// Words → lines: consecutive words with the same block / paragraph / line
392/// index join with single spaces (the order `tsv` lists them in is reading
393/// order within the line), the box is their union and the confidence the
394/// mean of theirs.
395fn group_lines(words: &[Word]) -> Vec<Unit> {
396    let mut out: Vec<Unit> = Vec::new();
397    let mut key: Option<(u32, u32, u32)> = None;
398    let mut n = 0.0f32;
399    for w in words {
400        let k = (w.block, w.par, w.line);
401        if key == Some(k) {
402            let last = out.last_mut().expect("a line is open");
403            last.0 = last.0.min(w.l);
404            last.1 = last.1.min(w.t);
405            last.2 = last.2.max(w.r);
406            last.3 = last.3.max(w.b);
407            last.4.push(' ');
408            last.4.push_str(&w.text);
409            last.5 = (last.5 * n + w.conf / 100.0) / (n + 1.0);
410            n += 1.0;
411        } else {
412            key = Some(k);
413            n = 1.0;
414            out.push((w.l, w.t, w.r, w.b, w.text.clone(), w.conf / 100.0));
415        }
416    }
417    out
418}
419
420/// Each word on its own, for the table cell matcher.
421fn word_units(words: &[Word]) -> Vec<Unit> {
422    words
423        .iter()
424        .map(|w| (w.l, w.t, w.r, w.b, w.text.clone(), w.conf / 100.0))
425        .collect()
426}
427
428/// Pixels of white paper added around a region crop: Tesseract's page
429/// segmentation wants text clear of the image border, and a layout box
430/// often sits on the ink.
431const CROP_PAD: u32 = 6;
432
433/// A region's crop as PNG bytes, with the crop's origin in image pixels.
434fn crop_png(img: &RgbImage, region: &Region, scale: f32) -> Option<(u32, u32, Vec<u8>)> {
435    let (iw, ih) = img.dimensions();
436    let l = ((region.l * scale).max(0.0) as u32).saturating_sub(CROP_PAD);
437    let t = ((region.t * scale).max(0.0) as u32).saturating_sub(CROP_PAD);
438    let r = (((region.r * scale).max(0.0) as u32).saturating_add(CROP_PAD)).min(iw);
439    let b = (((region.b * scale).max(0.0) as u32).saturating_add(CROP_PAD)).min(ih);
440    if r <= l || b <= t {
441        return None;
442    }
443    let crop = imageops::crop_imm(img, l, t, r - l, b - t).to_image();
444    Some((l, t, encode_png(&crop)?))
445}
446
447fn encode_png(img: &RgbImage) -> Option<Vec<u8>> {
448    let mut buf = std::io::Cursor::new(Vec::new());
449    img.write_to(&mut buf, image::ImageFormat::Png).ok()?;
450    Some(buf.into_inner())
451}
452
453/// The loaded engine: the validated options plus what `--list-langs` said.
454pub struct TesseractOcr {
455    opts: TesseractOptions,
456    /// Crops recognized concurrently (one child process each).
457    lanes: usize,
458    /// Whether the `osd` traineddata is installed — orientation detection
459    /// needs it.
460    has_osd: bool,
461}
462
463impl TesseractOcr {
464    /// Probe the binary (`--version`) and its languages (`--list-langs`),
465    /// checking every requested stem is installed. An error here is what the
466    /// pipeline shows as "OCR unavailable" (degrading like a missing model,
467    /// #244) — it names the binary, the missing stems and what is installed.
468    pub fn load(opts: TesseractOptions, lanes: usize) -> Result<Self, String> {
469        let lanes = docling_core::env::parse::<usize>("DOCLING_RS_OCR_SESSIONS")
470            .filter(|&n| n > 0)
471            .unwrap_or(lanes)
472            .clamp(1, 16);
473        let version = Command::new(&opts.cmd)
474            .arg("--version")
475            .stdin(Stdio::null())
476            .output()
477            .map_err(|e| {
478                format!(
479                    "tesseract binary {:?} not runnable ({e}); install tesseract-ocr or point \
480                     DOCLING_TESSERACT at it",
481                    opts.cmd
482                )
483            })?;
484        // Linux prints the version on stdout, older Windows builds on stderr.
485        let banner = [version.stdout, version.stderr]
486            .iter()
487            .map(|b| String::from_utf8_lossy(b).trim().to_string())
488            .find(|s| !s.is_empty())
489            .unwrap_or_default();
490        let banner = banner.lines().next().unwrap_or_default().to_string();
491        let mut list = Command::new(&opts.cmd);
492        list.arg("--list-langs").stdin(Stdio::null());
493        if let Some(dir) = &opts.tessdata_dir {
494            list.arg("--tessdata-dir").arg(dir);
495        }
496        let list = list
497            .output()
498            .map_err(|e| format!("tesseract --list-langs failed to run: {e}"))?;
499        // The header line ("List of available languages in ... (N):") is
500        // followed by one stem per line; Windows spells script packs with a
501        // backslash, `-l` wants the slash everywhere.
502        let installed: Vec<String> = String::from_utf8_lossy(&list.stdout)
503            .lines()
504            .skip(1)
505            .map(|l| l.trim().replace('\\', "/"))
506            .filter(|l| !l.is_empty())
507            .collect();
508        if installed.is_empty() {
509            return Err(format!(
510                "{banner}: no traineddata found (tesseract --list-langs printed nothing); \
511                 install a language pack (e.g. tesseract-ocr-eng) or set \
512                 DOCLING_RS_TESSDATA_DIR / TESSDATA_PREFIX"
513            ));
514        }
515        let wanted: Vec<&str> = match &opts.lang {
516            Some(l) => l.split('+').collect(),
517            None => vec!["eng"],
518        };
519        let missing: Vec<&str> = wanted
520            .iter()
521            .copied()
522            .filter(|w| !installed.iter().any(|i| i == w))
523            .collect();
524        if !missing.is_empty() {
525            return Err(format!(
526                "{banner}: no traineddata for {} (installed: {}); install the language \
527                 pack or pick an installed one with ocr_lang",
528                missing.join(", "),
529                installed.join(", ")
530            ));
531        }
532        let has_osd = installed.iter().any(|i| i == "osd");
533        debug_log!(
534            "docling-pdf: OCR engine {banner} (lang {}, psm {}, {} lane(s), osd {})",
535            opts.lang.as_deref().unwrap_or("eng"),
536            opts.psm.map_or("default".to_string(), |p| p.to_string()),
537            lanes,
538            if has_osd { "yes" } else { "no" }
539        );
540        Ok(Self {
541            opts,
542            lanes,
543            has_osd,
544        })
545    }
546
547    /// A child process with the common arguments: the binary, the language,
548    /// the data directory, and one OpenMP thread (the lanes own the
549    /// parallelism; libgomp's default of one thread per core per process
550    /// would oversubscribe every core `lanes` times — and Tesseract is not
551    /// faster multi-threaded on a small crop anyway).
552    fn command(&self) -> Command {
553        let mut cmd = Command::new(&self.opts.cmd);
554        cmd.env("OMP_THREAD_LIMIT", "1")
555            .stdin(Stdio::piped())
556            .stdout(Stdio::piped())
557            .stderr(Stdio::null());
558        if let Some(dir) = &self.opts.tessdata_dir {
559            cmd.arg("--tessdata-dir").arg(dir);
560        }
561        cmd
562    }
563
564    /// Feed `png` to a child on stdin and return its stdout. A child that
565    /// fails (a crop Tesseract rejects) is an `Err` the caller degrades
566    /// per crop.
567    fn run(&self, mut cmd: Command, png: &[u8]) -> Result<String, String> {
568        let mut child = cmd
569            .spawn()
570            .map_err(|e| format!("tesseract: spawn {:?}: {e}", self.opts.cmd))?;
571        // Write on a thread: a crop bigger than the pipe buffer would
572        // otherwise deadlock against a child that has started printing.
573        let mut stdin = child.stdin.take().expect("piped stdin");
574        let png = png.to_vec();
575        let writer = std::thread::spawn(move || stdin.write_all(&png));
576        let out = child
577            .wait_with_output()
578            .map_err(|e| format!("tesseract: wait: {e}"))?;
579        let _ = writer.join();
580        if !out.status.success() {
581            return Err(format!("tesseract exited with {}", out.status));
582        }
583        Ok(String::from_utf8_lossy(&out.stdout).into_owned())
584    }
585
586    /// Recognize one crop: `tesseract stdin stdout [-l L] [--psm N] --dpi D tsv`.
587    fn recognize(&self, png: &[u8], dpi: u32) -> Result<Vec<Word>, String> {
588        let mut cmd = self.command();
589        if let Some(lang) = &self.opts.lang {
590            cmd.arg("-l").arg(lang);
591        }
592        if let Some(psm) = self.opts.psm {
593            cmd.arg("--psm").arg(psm.to_string());
594        }
595        cmd.arg("--dpi").arg(dpi.to_string());
596        cmd.args(["stdin", "stdout", "tsv"]);
597        Ok(parse_tsv(&self.run(cmd, png)?))
598    }
599
600    /// Recognize every crop of `jobs` — `(origin l, origin t, png)` — across
601    /// the lanes, and return each crop's units mapped to page points
602    /// (`origin + px` / `scale`), in job order. A crop Tesseract fails on
603    /// contributes nothing (logged under `DOCLING_RS_DEBUG`), the rest of the
604    /// page still reads out.
605    fn recognize_all(
606        &self,
607        jobs: &[(u32, u32, Vec<u8>)],
608        scale: f32,
609        units: fn(&[Word]) -> Vec<Unit>,
610    ) -> Vec<(TextCell, f32)> {
611        let dpi = (scale * 72.0).round().max(1.0) as u32;
612        let lanes = self.lanes.min(jobs.len()).max(1);
613        let next = std::sync::atomic::AtomicUsize::new(0);
614        let results: Vec<std::sync::Mutex<Option<Vec<Word>>>> =
615            jobs.iter().map(|_| std::sync::Mutex::new(None)).collect();
616        std::thread::scope(|s| {
617            for _ in 0..lanes {
618                s.spawn(|| loop {
619                    let i = next.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
620                    let Some((_, _, png)) = jobs.get(i) else {
621                        break;
622                    };
623                    let words = match self.recognize(png, dpi) {
624                        Ok(words) => words,
625                        Err(e) => {
626                            debug_log!("docling-pdf: tesseract: crop {i}: {e}; no text");
627                            Vec::new()
628                        }
629                    };
630                    *results[i].lock().expect("crop result slot") = Some(words);
631                });
632            }
633        });
634        let mut cells = Vec::new();
635        for ((ox, oy, _), slot) in jobs.iter().zip(results) {
636            let words = slot
637                .into_inner()
638                .expect("crop result slot")
639                .unwrap_or_default();
640            for (l, t, r, b, text, conf) in units(&words) {
641                let text = text.trim().to_string();
642                if text.is_empty() {
643                    continue;
644                }
645                cells.push((
646                    TextCell {
647                        text,
648                        l: (*ox as f32 + l) / scale,
649                        t: (*oy as f32 + t) / scale,
650                        r: (*ox as f32 + r) / scale,
651                        b: (*oy as f32 + b) / scale,
652                    },
653                    conf,
654                ));
655            }
656        }
657        cells
658    }
659
660    /// The crops of the regions `keep` selects.
661    fn crops(
662        img: &RgbImage,
663        regions: &[Region],
664        scale: f32,
665        keep: fn(&str) -> bool,
666    ) -> Vec<(u32, u32, Vec<u8>)> {
667        regions
668            .iter()
669            .filter(|r| keep(r.label))
670            .filter_map(|r| crop_png(img, r, scale))
671            .collect()
672    }
673
674    /// OCR a page's text regions into line cells (page points) with their
675    /// confidence — the counterpart of [`crate::ocr::OcrModel::ocr_page`].
676    /// `scale` is image px per page point.
677    pub fn ocr_page(
678        &mut self,
679        img: &RgbImage,
680        regions: &[Region],
681        scale: f32,
682    ) -> Result<Vec<(TextCell, f32)>, String> {
683        let jobs = crate::timing::timed("ocr.prep", || {
684            Self::crops(img, regions, scale, is_text_label)
685        });
686        Ok(crate::timing::timed("ocr.rec", || {
687            self.recognize_all(&jobs, scale, group_lines)
688        }))
689    }
690
691    /// Word cells inside the page's table regions, for the cell matcher — the
692    /// counterpart of [`crate::ocr::OcrModel::ocr_table_words`].
693    pub fn ocr_table_words(
694        &mut self,
695        img: &RgbImage,
696        regions: &[Region],
697        scale: f32,
698    ) -> Result<Vec<(TextCell, f32)>, String> {
699        let jobs = Self::crops(img, regions, scale, crate::assemble::is_table_like);
700        Ok(self.recognize_all(&jobs, scale, word_units))
701    }
702
703    /// The clockwise angle the page content is rotated by in `img`
704    /// (`0`/`90`/`180`/`270`, the [`crate::orient::detect`] convention), from
705    /// Tesseract's orientation-and-script detection (`--psm 0 -l osd`).
706    /// `None` when the `osd` traineddata is not installed or OSD cannot make
707    /// a call (too little text) — the page then stays as rendered.
708    pub fn detect_orientation(&self, img: &RgbImage, scale: f32) -> Option<u16> {
709        if !self.has_osd {
710            debug_log!("docling-pdf: tesseract: no `osd` traineddata; orientation not probed");
711            return None;
712        }
713        let png = encode_png(img)?;
714        let mut cmd = self.command();
715        cmd.args(["--psm", "0", "-l", "osd", "--dpi"])
716            .arg(((scale * 72.0).round().max(1.0) as u32).to_string())
717            .args(["stdin", "stdout"]);
718        match self.run(cmd, &png) {
719            Ok(out) => parse_osd(&out),
720            Err(e) => {
721                debug_log!("docling-pdf: tesseract OSD failed ({e}); assuming upright");
722                None
723            }
724        }
725    }
726}
727
728/// The `Orientation in degrees:` line of an OSD report — Tesseract's clockwise
729/// angle of the text, the same convention as the probe's `deg` (its `Rotate:`
730/// line is the complementary correction).
731fn parse_osd(out: &str) -> Option<u16> {
732    let deg = out
733        .lines()
734        .find_map(|l| l.trim().strip_prefix("Orientation in degrees:"))?
735        .trim()
736        .parse::<u16>()
737        .ok()?;
738    matches!(deg, 0 | 90 | 180 | 270).then_some(deg)
739}
740
741#[cfg(test)]
742mod tests {
743    use super::*;
744
745    /// `ocr_lang` under the Tesseract engine: stems verbatim, BCP-47 tags and
746    /// the PP-OCR codes mapped, lists joined in order, junk rejected.
747    #[test]
748    fn lang_arg_maps_tags_and_keeps_stems() {
749        for (raw, want) in [
750            ("eng", "eng"),
751            ("en", "eng"),
752            ("EN-us", "eng"),
753            ("iso:en", "eng"),
754            ("english", "eng"),
755            ("ch", "chi_sim"),
756            ("ch_tra", "chi_tra"),
757            ("chinese_cht", "chi_tra"),
758            ("zh", "chi_sim"),
759            ("zh-Hant", "chi_tra"),
760            ("zh-TW", "chi_tra"),
761            ("iso:zh-CN", "chi_sim"),
762            ("de", "deu"),
763            ("de-Latf", "deu_latf"),
764            ("sr-Latn", "srp_latn"),
765            ("sr", "srp"),
766            ("nb", "nor"),
767            ("deu+fra", "deu+fra"),
768            ("iso:de + fr", "deu+fra"),
769            ("eng+eng", "eng"),
770            ("script/Cyrillic", "script/Cyrillic"),
771            ("chi_sim", "chi_sim"),
772            ("srp_latn", "srp_latn"),
773            ("custom_model", "custom_model"),
774        ] {
775            assert_eq!(lang_arg(raw).as_deref(), Ok(want), "{raw:?}");
776        }
777        for raw in ["", "xx", "en+", "de;u", "a b", "iso:", "deu fra"] {
778            assert!(lang_arg(raw).is_err(), "{raw:?} should be rejected");
779        }
780    }
781
782    /// The `tsv` shape Tesseract 5 prints: only level-5 rows are words; a
783    /// block's two lines become two cells, words joined by single spaces,
784    /// boxes unioned, confidence averaged onto 0–1; table mode keeps words.
785    #[test]
786    fn tsv_words_group_into_lines() {
787        let tsv = "level\tpage_num\tblock_num\tpar_num\tline_num\tword_num\tleft\ttop\twidth\theight\tconf\ttext\n\
788                   1\t1\t0\t0\t0\t0\t0\t0\t600\t160\t-1\t\n\
789                   4\t1\t1\t1\t1\t0\t23\t28\t242\t25\t-1\t\n\
790                   5\t1\t1\t1\t1\t1\t23\t28\t64\t20\t93.2\tHello\n\
791                   5\t1\t1\t1\t1\t2\t94\t28\t93\t25\t92.4\tdocling\n\
792                   5\t1\t1\t1\t1\t3\t195\t28\t70\t20\t96.0\tworld\n\
793                   5\t1\t2\t1\t1\t1\t21\t98\t94\t20\t96.8\tSecond\n\
794                   5\t1\t2\t1\t1\t2\t125\t98\t44\t20\t50\t \n\
795                   5\t1\t2\t1\t1\t3\t179\t99\t44\t19\t96.8\t123\n";
796        let words = parse_tsv(tsv);
797        assert_eq!(words.len(), 5);
798        let lines = group_lines(&words);
799        assert_eq!(lines.len(), 2);
800        let (l, t, r, b, text, conf) = &lines[0];
801        assert_eq!(text, "Hello docling world");
802        assert_eq!((*l, *t, *r, *b), (23.0, 28.0, 265.0, 53.0));
803        assert!((conf - 0.9387).abs() < 1e-3, "{conf}");
804        assert_eq!(lines[1].4, "Second 123");
805        assert_eq!(word_units(&words).len(), 5);
806        assert!(parse_tsv("").is_empty());
807        assert!(parse_tsv("garbage\n5\t1\n").is_empty());
808    }
809
810    #[test]
811    fn osd_orientation_parses() {
812        let out = "Page number: 0\nOrientation in degrees: 270\nRotate: 90\n\
813                   Orientation confidence: 6.47\nScript: Latin\nScript confidence: 4.05\n";
814        assert_eq!(parse_osd(out), Some(270));
815        assert_eq!(parse_osd("Orientation in degrees: 0\n"), Some(0));
816        assert_eq!(parse_osd("Orientation in degrees: 45\n"), None);
817        assert_eq!(parse_osd("Too few characters. Skipping this page\n"), None);
818    }
819
820    #[test]
821    fn psm_env_is_range_checked() {
822        let opts = TesseractOptions::from_env(Some("eng".into()));
823        assert_eq!(opts.lang.as_deref(), Some("eng"));
824        assert!(opts.psm.is_none() || opts.psm.is_some_and(|p| p <= 13));
825    }
826}