docling_pdf/picture.rs
1//! OCR for a standalone picture (#645): the text the recognizer reads off an
2//! image embedded in a non-PDF document — a DOCX/PPTX screenshot, an HTML
3//! figure, a sampled video frame — so the picture-OCR enrichment in
4//! `docling` can attach it to the picture as docling's description
5//! annotation. The models are the ML pipeline's own (the recognizer pair,
6//! the text detector, Tesseract under `ocr_engine`), loaded lazily on the
7//! first picture exactly as they are for a scanned page, so a converter that
8//! OCRs pictures and scans alike holds one set of sessions.
9//!
10//! The picture is treated as the image input `Pipeline::convert_image` sees:
11//! its own scale-1.0 page whose one text region is the whole image, read at
12//! docling's effective OCR resolution for images (3 px/pt shrunk to
13//! RapidOCR's 2000 px longer side, #570) — no layout pass, since the caller
14//! already knows the whole image is the picture. The detector's boxes are
15//! the line crops when `ocr_det.onnx` is installed (#570); the ink-projection
16//! strips otherwise. Lines under `DOCLING_RS_OCR_TEXT_SCORE` are dropped by
17//! the recognizer, as on a page.
18
19use image::RgbImage;
20
21use crate::layout::Region;
22use crate::pdfium_backend::TextCell;
23use crate::{ocr, ocr_input, page_ocr_scale, EnrichSlot, PdfError, Pipeline};
24
25/// What the OCR read off one picture: the lines in reading order (top to
26/// bottom, left to right within a row, rows joined by `\n`) and the engine
27/// that read them — docling's `PictureDescriptionData.provenance`.
28#[derive(Debug, Clone, PartialEq, Eq)]
29pub struct PictureText {
30 pub text: String,
31 /// `ppocr` or `tesseract` — the [`ocr::OcrEngine`] that ran.
32 pub provenance: &'static str,
33}
34
35impl Pipeline {
36 /// Decode an embedded picture's bytes (PNG/JPEG/GIF/BMP/TIFF/WebP) under
37 /// the standalone-image pixel cap (`DOCLING_RS_MAX_IMAGE_PIXELS`, 30000
38 /// per side) — a crafted header never allocates a multi-GB bitmap.
39 pub fn decode_picture(&self, bytes: &[u8]) -> Result<RgbImage, PdfError> {
40 crate::decode_image_limited(bytes)
41 }
42
43 /// The DocumentFigureClassifier's predictions for a picture (descending
44 /// confidence), loading the model on first use. `None` when the model is
45 /// not installed (warned once by the loader) or inference failed — the
46 /// caller decides what an unclassifiable picture means (the picture-OCR
47 /// class filter lets it through).
48 pub fn classify_picture(&mut self, img: &RgbImage) -> Option<Vec<docling_core::PictureClass>> {
49 let mut guard = self.classifier.lock().unwrap_or_else(|p| p.into_inner());
50 if matches!(*guard, EnrichSlot::Unloaded) {
51 *guard = match crate::enrich::PictureClassifier::load_with(crate::intra_threads()) {
52 Some(m) => EnrichSlot::Ready(m),
53 None => EnrichSlot::Missing,
54 };
55 }
56 let EnrichSlot::Ready(model) = &mut *guard else {
57 return None;
58 };
59 match model.classify(img) {
60 Ok(classes) => Some(classes),
61 Err(e) => {
62 eprintln!("docling-pdf: picture classifier: {e}");
63 None
64 }
65 }
66 }
67
68 /// OCR a picture: `Ok(None)` when the engine read no text (or could not
69 /// load — the recognizer's one-time warning says so, and `skip_ocr`
70 /// short-circuits without one), `Err` only for a failed inference.
71 pub fn ocr_picture(&mut self, img: &RgbImage) -> Result<Option<PictureText>, PdfError> {
72 let provenance = match self.ocr_engine {
73 ocr::OcrEngine::PpOcr => "ppocr",
74 ocr::OcrEngine::Tesseract => "tesseract",
75 };
76 let (w, h) = (img.width() as f32, img.height() as f32);
77 if w < 1.0 || h < 1.0 {
78 return Ok(None);
79 }
80 // The image is its own page at 1 px per point; the OCR reads it at
81 // docling's image resolution unless `ocr_scale` says otherwise.
82 let ocr_scale = page_ocr_scale(self.ocr_scale, w, h, 1.0);
83 let mut cache = None;
84 let (view, scale) = ocr_input(&mut cache, img, 1.0, ocr_scale);
85 let region = Region {
86 label: "text",
87 score: 1.0,
88 l: 0.0,
89 t: 0.0,
90 r: w,
91 b: h,
92 };
93 let worker = self.primary()?;
94 let detected = match worker.det_model() {
95 Some(det) => Some(det.detect(view).map_err(PdfError::Ocr)?),
96 None => None,
97 };
98 let Some(model) = worker.ocr_model()? else {
99 return Ok(None);
100 };
101 let cells = model
102 .ocr_page_with(
103 view,
104 std::slice::from_ref(®ion),
105 scale,
106 detected.as_deref(),
107 )
108 .map_err(PdfError::Ocr)?;
109 let text = lines_text(cells.into_iter().map(|(c, _)| c).collect());
110 Ok((!text.is_empty()).then_some(PictureText { text, provenance }))
111 }
112}
113
114/// Join OCR line cells into text in reading order: rows top to bottom, cells
115/// left to right within a row. A cell joins the row whose vertical extent
116/// contains its centre — two columns of a screenshot's dialog read across,
117/// the way a human scans it; a line that sits lower starts a new row. Empty
118/// cells are dropped.
119pub(crate) fn lines_text(mut cells: Vec<TextCell>) -> String {
120 cells.retain(|c| !c.text.trim().is_empty());
121 cells.sort_by(|a, b| a.t.total_cmp(&b.t).then(a.l.total_cmp(&b.l)));
122 let mut rows: Vec<(f32, f32, Vec<TextCell>)> = Vec::new();
123 for cell in cells {
124 let mid = (cell.t + cell.b) / 2.0;
125 match rows.last_mut() {
126 Some((t, b, row)) if mid >= *t && mid <= *b => {
127 *t = t.min(cell.t);
128 *b = b.max(cell.b);
129 row.push(cell);
130 }
131 _ => rows.push((cell.t, cell.b, vec![cell])),
132 }
133 }
134 rows.iter_mut()
135 .map(|(_, _, row)| {
136 row.sort_by(|a, b| a.l.total_cmp(&b.l));
137 row.iter()
138 .map(|c| c.text.trim())
139 .collect::<Vec<_>>()
140 .join(" ")
141 })
142 .collect::<Vec<_>>()
143 .join("\n")
144}
145
146#[cfg(test)]
147mod tests {
148 use super::*;
149
150 fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
151 TextCell {
152 text: text.into(),
153 l,
154 t,
155 r,
156 b,
157 }
158 }
159
160 /// Rows read across, then down; a cell whose centre sits below the
161 /// current row starts a new one, and empty cells vanish.
162 #[test]
163 fn lines_text_reads_rows_across_then_down() {
164 let cells = vec![
165 cell("right", 200.0, 10.0, 260.0, 30.0),
166 cell("below", 10.0, 40.0, 80.0, 60.0),
167 cell("left", 10.0, 12.0, 60.0, 32.0),
168 cell(" ", 10.0, 70.0, 20.0, 80.0),
169 ];
170 assert_eq!(lines_text(cells), "left right\nbelow");
171 assert_eq!(lines_text(Vec::new()), "");
172 }
173}