Skip to main content

docling_pdf/
picture.rs

1//! OCR for a standalone picture (#645): the text the recognizer reads off an
2//! image embedded in a non-PDF document — a DOCX/PPTX screenshot, an HTML
3//! figure, a sampled video frame — so the picture-OCR enrichment in
4//! `docling` can attach it to the picture as docling's description
5//! annotation. The models are the ML pipeline's own (the recognizer pair,
6//! the text detector, Tesseract under `ocr_engine`), loaded lazily on the
7//! first picture exactly as they are for a scanned page, so a converter that
8//! OCRs pictures and scans alike holds one set of sessions.
9//!
10//! The picture is treated as the image input `Pipeline::convert_image` sees:
11//! its own scale-1.0 page whose one text region is the whole image, read at
12//! docling's effective OCR resolution for images (3 px/pt shrunk to
13//! RapidOCR's 2000 px longer side, #570) — no layout pass, since the caller
14//! already knows the whole image is the picture. The detector's boxes are
15//! the line crops when `ocr_det.onnx` is installed (#570); the ink-projection
16//! strips otherwise. Lines under `DOCLING_RS_OCR_TEXT_SCORE` are dropped by
17//! the recognizer, as on a page.
18
19use image::RgbImage;
20
21use crate::layout::Region;
22use crate::pdfium_backend::TextCell;
23use crate::{ocr, ocr_input, page_ocr_scale, EnrichSlot, PdfError, Pipeline};
24
25/// What the OCR read off one picture: the lines in reading order (top to
26/// bottom, left to right within a row, rows joined by `\n`) and the engine
27/// that read them — docling's `PictureDescriptionData.provenance`.
28#[derive(Debug, Clone, PartialEq, Eq)]
29pub struct PictureText {
30    pub text: String,
31    /// `ppocr` or `tesseract` — the [`ocr::OcrEngine`] that ran.
32    pub provenance: &'static str,
33}
34
35impl Pipeline {
36    /// Decode an embedded picture's bytes (PNG/JPEG/GIF/BMP/TIFF/WebP) under
37    /// the standalone-image pixel cap (`DOCLING_RS_MAX_IMAGE_PIXELS`, 30000
38    /// per side) — a crafted header never allocates a multi-GB bitmap.
39    pub fn decode_picture(&self, bytes: &[u8]) -> Result<RgbImage, PdfError> {
40        crate::decode_image_limited(bytes)
41    }
42
43    /// The DocumentFigureClassifier's predictions for a picture (descending
44    /// confidence), loading the model on first use. `None` when the model is
45    /// not installed (warned once by the loader) or inference failed — the
46    /// caller decides what an unclassifiable picture means (the picture-OCR
47    /// class filter lets it through).
48    pub fn classify_picture(&mut self, img: &RgbImage) -> Option<Vec<docling_core::PictureClass>> {
49        let mut guard = self.classifier.lock().unwrap_or_else(|p| p.into_inner());
50        if matches!(*guard, EnrichSlot::Unloaded) {
51            *guard = match crate::enrich::PictureClassifier::load_with(crate::intra_threads()) {
52                Some(m) => EnrichSlot::Ready(m),
53                None => EnrichSlot::Missing,
54            };
55        }
56        let EnrichSlot::Ready(model) = &mut *guard else {
57            return None;
58        };
59        match model.classify(img) {
60            Ok(classes) => Some(classes),
61            Err(e) => {
62                eprintln!("docling-pdf: picture classifier: {e}");
63                None
64            }
65        }
66    }
67
68    /// OCR a picture: `Ok(None)` when the engine read no text (or could not
69    /// load — the recognizer's one-time warning says so, and `skip_ocr`
70    /// short-circuits without one), `Err` only for a failed inference.
71    pub fn ocr_picture(&mut self, img: &RgbImage) -> Result<Option<PictureText>, PdfError> {
72        let provenance = match self.ocr_engine {
73            ocr::OcrEngine::PpOcr => "ppocr",
74            ocr::OcrEngine::Tesseract => "tesseract",
75        };
76        let (w, h) = (img.width() as f32, img.height() as f32);
77        if w < 1.0 || h < 1.0 {
78            return Ok(None);
79        }
80        // The image is its own page at 1 px per point; the OCR reads it at
81        // docling's image resolution unless `ocr_scale` says otherwise.
82        let ocr_scale = page_ocr_scale(self.ocr_scale, w, h, 1.0);
83        let mut cache = None;
84        let (view, scale) = ocr_input(&mut cache, img, 1.0, ocr_scale);
85        let region = Region {
86            label: "text",
87            score: 1.0,
88            l: 0.0,
89            t: 0.0,
90            r: w,
91            b: h,
92        };
93        let worker = self.primary()?;
94        let detected = match worker.det_model() {
95            Some(det) => Some(det.detect(view).map_err(PdfError::Ocr)?),
96            None => None,
97        };
98        let Some(model) = worker.ocr_model()? else {
99            return Ok(None);
100        };
101        let cells = model
102            .ocr_page_with(
103                view,
104                std::slice::from_ref(&region),
105                scale,
106                detected.as_deref(),
107            )
108            .map_err(PdfError::Ocr)?;
109        let text = lines_text(cells.into_iter().map(|(c, _)| c).collect());
110        Ok((!text.is_empty()).then_some(PictureText { text, provenance }))
111    }
112}
113
114/// Join OCR line cells into text in reading order: rows top to bottom, cells
115/// left to right within a row. A cell joins the row whose vertical extent
116/// contains its centre — two columns of a screenshot's dialog read across,
117/// the way a human scans it; a line that sits lower starts a new row. Empty
118/// cells are dropped.
119pub(crate) fn lines_text(mut cells: Vec<TextCell>) -> String {
120    cells.retain(|c| !c.text.trim().is_empty());
121    cells.sort_by(|a, b| a.t.total_cmp(&b.t).then(a.l.total_cmp(&b.l)));
122    let mut rows: Vec<(f32, f32, Vec<TextCell>)> = Vec::new();
123    for cell in cells {
124        let mid = (cell.t + cell.b) / 2.0;
125        match rows.last_mut() {
126            Some((t, b, row)) if mid >= *t && mid <= *b => {
127                *t = t.min(cell.t);
128                *b = b.max(cell.b);
129                row.push(cell);
130            }
131            _ => rows.push((cell.t, cell.b, vec![cell])),
132        }
133    }
134    rows.iter_mut()
135        .map(|(_, _, row)| {
136            row.sort_by(|a, b| a.l.total_cmp(&b.l));
137            row.iter()
138                .map(|c| c.text.trim())
139                .collect::<Vec<_>>()
140                .join(" ")
141        })
142        .collect::<Vec<_>>()
143        .join("\n")
144}
145
146#[cfg(test)]
147mod tests {
148    use super::*;
149
150    fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
151        TextCell {
152            text: text.into(),
153            l,
154            t,
155            r,
156            b,
157        }
158    }
159
160    /// Rows read across, then down; a cell whose centre sits below the
161    /// current row starts a new one, and empty cells vanish.
162    #[test]
163    fn lines_text_reads_rows_across_then_down() {
164        let cells = vec![
165            cell("right", 200.0, 10.0, 260.0, 30.0),
166            cell("below", 10.0, 40.0, 80.0, 60.0),
167            cell("left", 10.0, 12.0, 60.0, 32.0),
168            cell("  ", 10.0, 70.0, 20.0, 80.0),
169        ];
170        assert_eq!(lines_text(cells), "left right\nbelow");
171        assert_eq!(lines_text(Vec::new()), "");
172    }
173}