Skip to main content

docling_pdf/
pdfium_backend.rs

1//! pdfium-based text extraction and page rendering.
2//!
3//! Text is reconstructed the way docling's `docling-parse` does it, so the
4//! output spacing matches the groundtruth: the page's **character** stream is
5//! grouped into **words** (split at a horizontal gap wider than a fraction of
6//! the font height — font-relative, so letter-tracking in display titles does
7//! not split a word) and words into **lines** (by baseline). pdfium-render's
8//! safe API only exposes whole style runs / `GetBoundedText`, so the character
9//! loop is driven through the raw `PdfiumLibraryBindings` FFI on a second handle
10//! to the same bytes (no fork; stays publishable).
11
12#[cfg(feature = "ocr-prep")]
13use image::RgbImage;
14#[cfg(feature = "ml")]
15use pdfium_render::prelude::*;
16
17/// A run of text with its bounding box, in PDF points with a **top-left** origin
18/// (pdfium's native origin is bottom-left; we flip it to match docling's
19/// `BoundingBox(..., origin=TOPLEFT)`).
20#[derive(Debug, Clone)]
21pub struct TextCell {
22    pub text: String,
23    pub l: f32,
24    pub t: f32,
25    pub r: f32,
26    pub b: f32,
27}
28
29/// Pixels-per-point used to render page images. Layout is scale-invariant (it
30/// scales normalized boxes by the page point size), but OCR benefits from the
31/// extra resolution.
32pub const RENDER_SCALE: f32 = 2.0;
33
34/// One page's geometry, extracted text cells, and a rendered RGB image. The
35/// image is rendered at [`RENDER_SCALE`] pixels per PDF point; `image px =
36/// page point × scale`.
37#[derive(Clone)]
38pub struct PdfPage {
39    pub width: f32,
40    pub height: f32,
41    pub scale: f32,
42    pub cells: Vec<TextCell>,
43    /// Same text grouped for code regions: split only at pdfium space glyphs, so
44    /// monospace runs keep their source spacing instead of the prose heuristic's.
45    pub code_cells: Vec<TextCell>,
46    /// Per-word cells (one per word, not joined into lines) for TableFormer cell
47    /// matching.
48    pub word_cells: Vec<TextCell>,
49    /// The rendered page bitmap. Present whenever pixels are available at all
50    /// (`ocr-prep` ⊂ `ml`): the native pipeline renders it with pdfium, the
51    /// browser pipeline receives it from the host canvas. Picture regions are
52    /// cropped out of it.
53    #[cfg(feature = "ocr-prep")]
54    pub image: RgbImage,
55    /// The **scale-1.0** page image the layout model runs on (parity with
56    /// docling's `PyPdfiumDocumentBackend`: the layout stage calls
57    /// `page.get_image(scale=1.0)`, which that backend serves as pdfium at
58    /// 1.5×, PIL-BICUBIC down to point size — a *different* image from the 2×
59    /// OCR/crop bitmap above, and a different resampling regime than
60    /// stretching that bitmap; docling 2.123+'s default docling-parse backend
61    /// renders the same request with its own Blend2D/FreeType renderer, whose
62    /// glyph anti-aliasing differs — #478). `None` on paths without a pdfium
63    /// renderer (browser, METS/TIFF), which fall back to stretching
64    /// [`Self::image`].
65    #[cfg(feature = "ocr-prep")]
66    pub image_layout: Option<RgbImage>,
67    /// Hyperlink annotations on the page (rect in top-left page coords + target
68    /// URI), restricted to web/mail/tel schemes. Used only by strict Markdown.
69    pub links: Vec<LinkAnnot>,
70    /// The page's `/Rotate` value (0/90/180/270) when it was normalized away
71    /// before inference: a scanned page with `/Rotate` displays its raster
72    /// rotated, which turns OCR into garbage — so extraction un-rotates the
73    /// bitmaps (and swaps `width`/`height`) and records the display rotation
74    /// here. Assembly rotates the finished geometry *back* by this many
75    /// degrees clockwise, so emitted locations and the page size stay in
76    /// display space (matching docling and every PDF viewer). Always 0 for
77    /// text-layer pages (their cells live in display space already) and on
78    /// paths without a pdfium renderer.
79    pub rotation: u16,
80}
81
82impl PdfPage {
83    /// A page built from recognized cells alone — the browser pipeline's
84    /// shape (#157), where the bitmap lives on the JS side. Exists so callers
85    /// compile identically with and without the `ml` feature: under a
86    /// feature-unified workspace build the struct carries the `image` field,
87    /// which a plain literal in a non-`ml` consumer can't spell.
88    #[cfg(feature = "ocr-prep")]
89    pub fn from_cells(width: f32, height: f32, scale: f32, cells: Vec<TextCell>) -> Self {
90        Self {
91            width,
92            height,
93            scale,
94            cells,
95            code_cells: Vec::new(),
96            word_cells: Vec::new(),
97            #[cfg(feature = "ocr-prep")]
98            image: RgbImage::new(0, 0),
99            #[cfg(feature = "ocr-prep")]
100            image_layout: None,
101            links: Vec::new(),
102            rotation: 0,
103        }
104    }
105
106    /// Same as [`from_cells`](Self::from_cells) but carrying the rendered page
107    /// bitmap, so picture regions can be cropped out of it (#157: the browser
108    /// pipeline gets the same figure bytes the native one does).
109    #[cfg(feature = "ocr-prep")]
110    pub fn from_cells_with_image(
111        width: f32,
112        height: f32,
113        scale: f32,
114        cells: Vec<TextCell>,
115        image: RgbImage,
116    ) -> Self {
117        Self {
118            image,
119            ..Self::from_cells(width, height, scale, cells)
120        }
121    }
122
123    /// Un-rotate the page's bitmaps by `deg` (clockwise 90° steps) and record
124    /// the compensating display rotation, composing with any rotation already
125    /// recorded: the raster becomes upright for inference while assembly
126    /// still maps the finished geometry back into display space. Handles both
127    /// `/Rotate` normalization (extraction) and content-detected orientation
128    /// (#225) — the two compose additively (axis-aligned 90° rotations
129    /// commute through the dimension swaps). Link rectangles follow the
130    /// raster; `width`/`height` swap on odd quarter-turns.
131    #[cfg(feature = "ocr-prep")]
132    pub(crate) fn unrotate(&mut self, deg: u16) {
133        if deg == 0 {
134            return;
135        }
136        use image::imageops::{rotate180, rotate270, rotate90};
137        // Display = upright rotated `deg`° clockwise, so upright = display
138        // rotated the complementary amount clockwise.
139        let un = |img: &RgbImage| match deg {
140            90 => rotate270(img),
141            180 => rotate180(img),
142            _ => rotate90(img),
143        };
144        if self.image.width() > 1 {
145            self.image = un(&self.image);
146        }
147        self.image_layout = self.image_layout.as_ref().map(&un);
148        let (width, height) = (self.width, self.height);
149        // Link rects follow the raster from display into upright space (the
150        // inverse of the geometry rotation assembly applies at the end).
151        for l in &mut self.links {
152            let (nl, nt, nr, nb) = match deg {
153                90 => (l.t, width - l.r, l.b, width - l.l),
154                180 => (width - l.r, height - l.b, width - l.l, height - l.t),
155                _ => (height - l.b, l.l, height - l.t, l.r),
156            };
157            (l.l, l.t, l.r, l.b) = (nl, nt, nr, nb);
158        }
159        if deg != 180 {
160            (self.width, self.height) = (height, width);
161        }
162        self.rotation = (self.rotation + deg) % 360;
163    }
164}
165
166/// A PDF link annotation: its rectangle (top-left page coordinates, matching
167/// [`TextCell`]) and target URI.
168#[derive(Debug, Clone)]
169pub struct LinkAnnot {
170    pub l: f32,
171    pub t: f32,
172    pub r: f32,
173    pub b: f32,
174    pub uri: String,
175}
176
177#[cfg(feature = "ml")]
178/// A parsed PDF: per-page text cells and page images.
179pub struct PdfDocument {
180    pub pages: Vec<PdfPage>,
181}
182
183/// Whether to use the docling-parse line sanitizer ([`crate::dp_lines`]) for prose
184/// reconstruction — the default. Set `DOCLING_LEGACY_LINES` to fall back to the
185/// older gap-heuristic `lines_from_glyphs`.
186pub(crate) fn use_dp_lines() -> bool {
187    !docling_core::env::flag("DOCLING_LEGACY_LINES")
188}
189
190/// Whether to source **word** cells from the pure-Rust parser (roadmap item 6),
191/// the default. The parser's `word_cells` reproduce docling-parse's word grouping
192/// byte-for-byte — the per-word tokens TableFormer matches table-grid cells
193/// against — which moves table extraction closer to docling on the heavy
194/// multi-column fixtures. Set `DOCLING_PDFIUM_WORDS` to keep pdfium's word cells,
195/// or `DOCLING_PDFIUM_TEXT` to fall back to pdfium for all text.
196pub(crate) fn use_parser_words() -> bool {
197    !docling_core::env::flag("DOCLING_PDFIUM_WORDS")
198        && !docling_core::env::flag("DOCLING_PDFIUM_TEXT")
199}
200
201/// Whether to source **code** cells from the parser too (the default) — the last
202/// text layer to leave pdfium, fully retiring its text path. The parser's
203/// gap-based code grouping ([`code_cells_from_glyphs`]) reconstructs monospace
204/// spacing from positioning gaps (`function add(a, b) { … }`), so it no longer
205/// drops the inter-token spaces the old space-glyph-only grouping lost
206/// (`functionadd`). Reverts to pdfium with `DOCLING_PDFIUM_WORDS` (alongside word
207/// cells) or `DOCLING_PDFIUM_TEXT` (all text).
208pub(crate) fn use_parser_code() -> bool {
209    use_parser_words()
210}
211
212#[cfg(feature = "ml")]
213/// Try binding pdfium from a directory (or a literal library file path):
214/// `<dir>/<platform library name>` first, else `<dir>` itself as the file.
215fn try_bind_dir(path: &str) -> Option<Box<dyn pdfium_render::prelude::PdfiumLibraryBindings>> {
216    let name = Pdfium::pdfium_platform_library_name_at_path(path);
217    if let Ok(b) = Pdfium::bind_to_library(&name) {
218        return Some(b);
219    }
220    Pdfium::bind_to_library(path).ok()
221}
222
223#[cfg(feature = "ml")]
224/// Bind to the pdfium dynamic library. Honors `PDFIUM_DYNAMIC_LIB_PATH` (a
225/// directory or file) first; else falls back to `.pdfium/lib` relative to the
226/// current directory (the layout `scripts/install/download_dependencies.sh` and
227/// `scripts/install/pdf_setup.sh` both produce); else the system library.
228fn bind() -> Result<Pdfium, PdfiumError> {
229    if let Some(path) = docling_core::env::nonempty("PDFIUM_DYNAMIC_LIB_PATH") {
230        if let Some(b) = try_bind_dir(&path) {
231            return Ok(Pdfium::new(b));
232        }
233    }
234    // No env var (or it didn't resolve): fall back to `.pdfium/lib` relative to
235    // the current directory — mirroring `layout.rs`/`ocr.rs`'s `.models/…`
236    // defaults — the layout `scripts/install/download_dependencies.sh` (and
237    // `scripts/install/pdf_setup.sh`) produce, so a checkout with the dependencies
238    // downloaded next to it needs no env var at all.
239    if let Some(b) = try_bind_dir(&crate::resolve_asset(".pdfium/lib")) {
240        return Ok(Pdfium::new(b));
241    }
242    Pdfium::bind_to_system_library().map(Pdfium::new)
243}
244
245#[cfg(feature = "ml")]
246impl PdfDocument {
247    /// Parse a PDF from bytes, optionally decrypting with `password`.
248    ///
249    /// Note: this materialises **every** page's rendered bitmap in memory at
250    /// once. For large documents prefer [`for_each_page`], which streams.
251    pub fn open(bytes: &[u8], password: Option<&str>) -> Result<Self, PdfiumError> {
252        let pdfium = bind()?;
253        let ffi = FfiText::load(pdfium.bindings(), bytes, password);
254        let doc = pdfium.load_pdf_from_byte_slice(bytes, password)?;
255        let dparse = crate::dparse_render::Doc::open_if_enabled(bytes, password);
256        let mut rust = rust_parser_cells(bytes);
257        let mut pages = Vec::new();
258        for (i, page) in doc.pages().iter().enumerate() {
259            let rc = rust.as_mut().map(|p| p.cells_timed(i));
260            pages.push(extract_page(
261                &page,
262                &ffi,
263                i as i32,
264                rc,
265                true,
266                true,
267                dparse.as_ref(),
268            )?);
269        }
270        Ok(PdfDocument { pages })
271    }
272}
273
274#[cfg(feature = "ml")]
275/// Per-page prose line cells from the pure-Rust text parser. This is the
276/// **default** text layer (it matches docling-parse's char geometry and is a
277/// strict improvement on byte-conformance — e.g. it recovers the Arabic
278/// sentence-period attachment in `right_to_left_01`). Set `DOCLING_PDFIUM_TEXT`
279/// to fall back to pdfium's text layer. The parser returns an empty page when a
280/// PDF (or a page) has no parseable text layer; the caller keeps pdfium's cells
281/// in that case, so scanned/edge-case pages are unaffected.
282fn rust_parser_cells(bytes: &[u8]) -> Option<crate::textparse::PageTextParser> {
283    if docling_core::env::flag("DOCLING_PDFIUM_TEXT") {
284        return None;
285    }
286    // Only the document load happens here; pages are parsed as the walk
287    // reaches them (`cells_timed`), so nothing is decoded for pages outside
288    // a `--pages` window and the parse overlaps the workers' inference.
289    crate::timing::timed("textparse.open", || {
290        crate::textparse::PageTextParser::open(bytes)
291    })
292}
293
294impl crate::textparse::PageTextParser {
295    /// [`cells`](Self::cells) under the `textparse` timing stage (per page).
296    fn cells_timed(&mut self, index: usize) -> crate::textparse::PageParserCells {
297        crate::timing::timed("textparse", || self.cells(index))
298    }
299}
300
301#[cfg(feature = "ml")]
302/// Number of pages in a PDF, without rendering any of them — used to decide
303/// whether a document is worth spinning up the parallel worker pool.
304pub fn page_count(bytes: &[u8], password: Option<&str>) -> Result<usize, PdfiumError> {
305    crate::timing::timed("pdfium.page_count", || {
306        let pdfium = bind()?;
307        let doc = pdfium.load_pdf_from_byte_slice(bytes, password)?;
308        Ok(doc.pages().len() as usize)
309    })
310}
311
312#[cfg(feature = "ml")]
313/// Render + extract pages one at a time, handing each (owned) [`PdfPage`] to `f`.
314/// Only one page bitmap is resident at a time — a rendered page is ~5 MB, so a
315/// large PDF would otherwise hold gigabytes of bitmaps at once. `f` receives the
316/// zero-based page index and the total page count.
317///
318/// `render_image` controls whether the page bitmap is rasterized at all: layout,
319/// OCR, TableFormer, and picture cropping all need it, but a caller that skips
320/// every one of those (the `no_ocr` fast path) doesn't, and rasterizing +
321/// downsampling a page is by far the most expensive step per page — skipping it
322/// is most of `no_ocr`'s speedup. `PdfPage::image` is a 1×1 placeholder when
323/// `false`; do not read it.
324///
325/// `extract_text` decodes the page's text layer (parser or pdfium cells); pass
326/// `false` when full-page OCR is forced and the cells would be discarded
327/// unread (docling#4061).
328///
329/// `range` restricts the walk to a **0-based inclusive** page window (issue
330/// #80's `--pages`); out-of-window pages are skipped *before* text extraction
331/// and rasterization, so a 3-page window over a 500-page PDF costs three
332/// pages, not five hundred. `f` still receives the absolute page index, so
333/// downstream page numbering refers to the source document.
334///
335/// `E` is the caller's error type; pdfium errors convert into it via `From`.
336pub fn for_each_page<E, F>(
337    bytes: &[u8],
338    password: Option<&str>,
339    render_image: bool,
340    extract_text: bool,
341    range: Option<(usize, usize)>,
342    mut f: F,
343) -> Result<(), E>
344where
345    E: From<PdfiumError>,
346    F: FnMut(usize, usize, PdfPage) -> Result<(), E>,
347{
348    let pdfium = bind()?;
349    let (ffi, doc) = crate::timing::timed("pdfium.open", || {
350        let ffi = FfiText::load(pdfium.bindings(), bytes, password);
351        pdfium
352            .load_pdf_from_byte_slice(bytes, password)
353            .map(|doc| (ffi, doc))
354    })?;
355    // `extract_text = false` (full-page OCR forced, docling#4061 / 2.122):
356    // the text layer would be cleared unread, so neither the pure-Rust parser
357    // nor pdfium's text page is decoded at all — on vector-dense pages (CAD
358    // drawings as 100k+ path segments) that decode is most of the page cost.
359    let mut rust = if extract_text {
360        rust_parser_cells(bytes)
361    } else {
362        None
363    };
364    // The opt-in docling-parse renderer (#478): when active, the page images
365    // the models see come from it instead of pdfium.
366    let dparse = if render_image {
367        crate::dparse_render::Doc::open_if_enabled(bytes, password)
368    } else {
369        None
370    };
371    let pages = doc.pages();
372    let total = pages.len() as usize;
373    let (first, last) = range.unwrap_or((0, total.saturating_sub(1)));
374    // Index the window directly: iterating `pages.iter()` from page 0 and
375    // skipping to `first` loads (and closes) every page before the window —
376    // ~0.7 ms each, 1.3 s of pure overhead for a one-page window over the
377    // 1913-page .NET reference.
378    for i in first..=last {
379        if i >= total {
380            break;
381        }
382        let page = pages.get(i as pdfium_render::prelude::PdfPageIndex)?;
383        let rc = rust.as_mut().map(|p| p.cells_timed(i));
384        let extracted = extract_page(
385            &page,
386            &ffi,
387            i as i32,
388            rc,
389            render_image,
390            extract_text,
391            dparse.as_ref(),
392        )?;
393        f(i, total, extracted)?;
394    }
395    // Tearing down the parsed document (hundreds of thousands of lopdf
396    // objects on a long PDF — 250 ms for the 1913-page .NET reference) is
397    // nobody's business but the allocator's: hand it to a detached thread so
398    // the last page's output isn't held up by it. `Arc`, not `Rc`, in the
399    // caches is what makes the parser `Send`.
400    if let Some(parser) = rust {
401        std::thread::spawn(move || crate::timing::timed("textparse.close", || drop(parser)));
402    }
403    Ok(())
404}
405
406/// One rasterized page from [`render_pages`] (#243): the absolute 1-based page
407/// number in the source document, the pixel dimensions, and the PNG bytes.
408#[cfg(feature = "ml")]
409#[derive(Debug, Clone)]
410pub struct RenderedPage {
411    pub page_no: usize,
412    pub width: u32,
413    pub height: u32,
414    pub png: Vec<u8>,
415}
416
417/// Upper bound, in pixels, on a rendered page bitmap's side. A crafted PDF can
418/// declare an enormous `MediaBox` in a few hundred bytes; the page render then
419/// asks pdfium — and `into_rgb8` — to allocate `w * h * 4` bytes. At the
420/// pipeline's 3x supersample a 12000 pt box is 36000x36000 ~ 5 GB: pdfium
421/// returns an opaque internal error at the extreme, and just below it the
422/// `image` crate *panics* (a `TryReserveError`, not a recoverable error) when
423/// the allocation fails. Real pages, even large-format (A0 at 3x ~ 10110 px),
424/// stay well under this cap; it only rejects the implausible, turning an abort
425/// into a clean error. Mirrors `decode_image_limited`'s guard on the
426/// standalone-image path. `DOCLING_RS_MAX_RENDER_PIXELS` overrides it.
427#[cfg(feature = "ml")]
428fn max_render_side() -> u32 {
429    static M: std::sync::OnceLock<u32> = std::sync::OnceLock::new();
430    *M.get_or_init(|| docling_core::env::parse("DOCLING_RS_MAX_RENDER_PIXELS").unwrap_or(15_000))
431}
432
433/// Round float pixel dimensions to the `i32` pdfium wants, rejecting a page
434/// whose bitmap would exceed [`max_render_side`] on either side before either
435/// pdfium or `image` tries to allocate it.
436#[cfg(feature = "ml")]
437fn checked_render_dims(
438    w_px: f64,
439    h_px: f64,
440    page_no: usize,
441) -> Result<(i32, i32), crate::PdfError> {
442    let cap = max_render_side();
443    let w = w_px.round().max(1.0);
444    let h = h_px.round().max(1.0);
445    if w > f64::from(cap) || h > f64::from(cap) {
446        return Err(crate::PdfError::Pdfium(format!(
447            "page {page_no}: render size {w:.0}x{h:.0} px exceeds the {cap}px per-side cap \
448             (raise DOCLING_RS_MAX_RENDER_PIXELS); the page's declared size is implausibly large"
449        )));
450    }
451    Ok((w as i32, h as i32))
452}
453
454#[cfg(feature = "ml")]
455/// Rasterize a PDF's pages to PNG (#243) — the lean path behind serve's
456/// `to=images`: pdfium render only, no text extraction, no models, and only
457/// one page bitmap resident at a time (each is PNG-encoded and dropped before
458/// the next renders). `scale` is pixels per PDF point — 2.0 matches the
459/// pipeline's [`RENDER_SCALE`] (144 dpi). Unlike the pipeline's render there
460/// is no 1.5× supersample + downsample pass: that dance exists only because
461/// TableFormer is pixel-pinned to docling's bitmaps, and nothing downstream
462/// of this output is — a single render is nearly twice as fast.
463///
464/// `range` is a **1-based** inclusive page window (issue #80's `pages`
465/// semantics: the end clamps to the document, a start past the end errors).
466///
467/// pdfium is not thread-safe — callers must serialize this against any other
468/// pdfium use (docling-serve holds its pipeline mutex around this call for
469/// exactly that reason).
470pub fn render_pages(
471    bytes: &[u8],
472    password: Option<&str>,
473    range: Option<(usize, usize)>,
474    scale: f32,
475) -> Result<Vec<RenderedPage>, crate::PdfError> {
476    let pdfium = bind()?;
477    let doc = pdfium.load_pdf_from_byte_slice(bytes, password)?;
478    let pages = doc.pages();
479    let total = pages.len() as usize;
480    let (first, last) = match range {
481        None => (0, total.saturating_sub(1)),
482        Some((first, last)) => {
483            if first == 0 || last < first {
484                return Err(crate::PdfError::Pdfium(format!(
485                    "invalid page range {first}-{last} (pages are 1-based, first <= last)"
486                )));
487            }
488            if first > total {
489                return Err(crate::PdfError::Pdfium(format!(
490                    "page range {first}-{last} is outside the document ({total} page(s))"
491                )));
492            }
493            (first - 1, last.min(total) - 1)
494        }
495    };
496    let mut out = Vec::with_capacity(last.saturating_sub(first) + 1);
497    for i in first..=last {
498        if i >= total {
499            break;
500        }
501        let page = pages.get(i as pdfium_render::prelude::PdfPageIndex)?;
502        // pdfium applies /Rotate itself, so the bitmap is the page as a viewer
503        // shows it — no orientation handling needed (the pipeline's scanned-page
504        // un-rotation is an OCR-conformance concern, not a display one).
505        let (tw, th) = checked_render_dims(
506            f64::from(page.width().value * scale),
507            f64::from(page.height().value * scale),
508            i + 1,
509        )?;
510        let cfg = PdfRenderConfig::new()
511            .set_target_width(tw)
512            .set_target_height(th);
513        let bitmap = crate::timing::timed("pdfium.rasterize", || {
514            page.render_with_config(&cfg)
515                .map(|b| b.as_image().into_rgb8())
516        })?;
517        let mut png = Vec::new();
518        bitmap
519            .write_to(&mut std::io::Cursor::new(&mut png), image::ImageFormat::Png)
520            .map_err(|e| crate::PdfError::Pdfium(format!("PNG-encoding page {}: {e}", i + 1)))?;
521        out.push(RenderedPage {
522            page_no: i + 1,
523            width: bitmap.width(),
524            height: bitmap.height(),
525            png,
526        });
527    }
528    Ok(out)
529}
530
531#[cfg(feature = "ml")]
532fn extract_page(
533    page: &pdfium_render::prelude::PdfPage<'_>,
534    ffi: &FfiText<'_>,
535    index: i32,
536    rust_cells: Option<crate::textparse::PageParserCells>,
537    render_image: bool,
538    extract_text: bool,
539    dparse: Option<&crate::dparse_render::Doc>,
540) -> Result<PdfPage, PdfiumError> {
541    // pdfium reports the page size (and renders) in the *display* frame —
542    // `/Rotate` applied — while every text coordinate (its own text page, the
543    // pure-Rust parser's MediaBox-based glyphs, link annotation rects) lives
544    // in the unrotated frame (docling#4008, 2.121). Keep the unrotated box
545    // around for the y-flips and bring every rect into the display frame.
546    let width = page.width().value;
547    let height = page.height().value;
548    let rotation = match page.rotation() {
549        Ok(PdfPageRenderRotation::Degrees90) => 90u16,
550        Ok(PdfPageRenderRotation::Degrees180) => 180,
551        Ok(PdfPageRenderRotation::Degrees270) => 270,
552        _ => 0,
553    };
554    let (unrot_w, unrot_h) = if rotation == 90 || rotation == 270 {
555        (height, width)
556    } else {
557        (width, height)
558    };
559
560    // Default: use the pure-Rust text parser instead of pdfium's text layer
561    // (override with `DOCLING_PDFIUM_TEXT`). Prose line cells always come from the
562    // parser; word and code cells do too unless `DOCLING_PDFIUM_WORDS` keeps them
563    // on pdfium (the parser's word grouping reproduces docling-parse's, which
564    // TableFormer matches against — roadmap item 6). A page the parser couldn't
565    // read (no text layer) keeps pdfium's cells.
566    let rc = rust_cells.unwrap_or_default();
567    let need_pdfium_prose = extract_text && rc.prose.is_empty();
568    let need_pdfium_words = extract_text && (!use_parser_words() || rc.words.is_empty());
569    let need_pdfium_code = extract_text && (!use_parser_code() || rc.code.is_empty());
570
571    // The parser covers prose/words/code from one shared glyph pass, so on the
572    // common (parser-succeeded) page all three are already satisfied and this
573    // pdfium FFI call — otherwise fully discarded below — is skipped outright.
574    let (mut cells, mut code_cells, mut word_cells) =
575        if need_pdfium_prose || need_pdfium_words || need_pdfium_code {
576            let (mut cells, code_cells, word_cells) =
577                crate::timing::timed("ffi.page_cells", || ffi.page_cells(index, unrot_h));
578            if cells.is_empty() {
579                cells = segment_cells(&page.text()?, unrot_h);
580            }
581            (cells, code_cells, word_cells)
582        } else {
583            (Vec::new(), Vec::new(), Vec::new())
584        };
585    if !rc.prose.is_empty() {
586        cells = rc.prose;
587    }
588    if use_parser_words() && !rc.words.is_empty() {
589        word_cells = rc.words;
590    }
591    if use_parser_code() && !rc.code.is_empty() {
592        code_cells = rc.code;
593    }
594    if rotation != 0 {
595        for c in cells
596            .iter_mut()
597            .chain(word_cells.iter_mut())
598            .chain(code_cells.iter_mut())
599        {
600            let (l, t, r, b) = to_display_frame((c.l, c.t, c.r, c.b), rotation, unrot_w, unrot_h);
601            (c.l, c.t, c.r, c.b) = (l, t, r, b);
602        }
603    }
604
605    // The opt-in docling-parse renderer (#478, `dparse_render.rs`): both model
606    // inputs come from docling-parse's Blend2D canvas, requested in docling's
607    // order — the scale-1.0 layout image first (docling decodes the page at its
608    // `render_scale` of 1.0, which fixes the bitmap decode resolution), then
609    // the scale-2.0 image TableFormer/OCR crop from (`_render_image_at_scale`
610    // on the same page decoder). The canvases are `ceil`-sized where pdfium's
611    // are `round`ed; every consumer maps points through `RENDER_SCALE`, not
612    // through the image size, so the extra row/column is harmless.
613    let (mut dp_image, mut dp_layout) = (None, None);
614    if let (true, Some(dp)) = (render_image, dparse) {
615        let io_err = |e: String| PdfiumError::IoError(std::io::Error::other(e));
616        let layout =
617            crate::timing::timed("dparse.render_layout", || dp.render(index as usize, 1.0))
618                .map_err(io_err)?;
619        let full = crate::timing::timed("dparse.render", || {
620            dp.render(index as usize, f64::from(RENDER_SCALE))
621        })
622        .map_err(io_err)?;
623        dp.release_page(index as usize);
624        dp_image = Some(full);
625        dp_layout = Some(layout);
626    }
627    let image = if let Some(img) = dp_image.take() {
628        img
629    } else if render_image {
630        // docling's pypdfium2 backend renders at 1.5× the target scale and
631        // downsamples "to make it sharper" (pypdfium2 → PIL BICUBIC). Replicate
632        // exactly: the TableFormer model is pixel-sensitive, so the page bitmap
633        // must match that backend's byte-for-byte (docling 2.123+'s default
634        // docling-parse backend renders it itself — #478).
635        // `CatmullRom` is the same a=-0.5 cubic kernel as PIL's BICUBIC.
636        const SUPERSAMPLE: f32 = 1.5;
637        // The 3x supersample is the largest bitmap the pipeline renders, so the
638        // per-side cap is enforced here; the 1.5x layout render below is always
639        // smaller and needs no separate guard.
640        let (tw, th) = checked_render_dims(
641            f64::from(width * RENDER_SCALE * SUPERSAMPLE),
642            f64::from(height * RENDER_SCALE * SUPERSAMPLE),
643            (index + 1) as usize,
644        )
645        .map_err(|e| PdfiumError::IoError(std::io::Error::other(e.to_string())))?;
646        let cfg = PdfRenderConfig::new()
647            .set_target_width(tw)
648            .set_target_height(th);
649        let big = crate::timing::timed("pdfium.render", || {
650            page.render_with_config(&cfg)
651                .map(|b| b.as_image().into_rgb8())
652        })?;
653        let dw = (width * RENDER_SCALE).round().max(1.0) as u32;
654        let dh = (height * RENDER_SCALE).round().max(1.0) as u32;
655        crate::timing::timed("image.resize", || fast_downscale(&big, dw, dh))
656    } else {
657        RgbImage::new(1, 1)
658    };
659    // The layout model's input image, built exactly like docling's pypdfium2
660    // `get_page_image(scale=1.0)`: a pdfium render at 1.5× (pypdfium2 sizes
661    // with `ceil`), PIL-BICUBIC down to the point-size image (PIL `resize`'s
662    // default kernel; Python `round` = ties-to-even). Distinct from the 2×
663    // bitmap above — resampling 1224→640 and 612→640 are different regimes,
664    // and the heron model's borderline scores follow the pixels. "Exactly" is
665    // against the pypdfium2 backend: docling 2.123+'s default docling-parse
666    // backend renders this image with its own renderer (#478, see
667    // docs/PDF_CONFORMANCE.md).
668    let image_layout = if let Some(img) = dp_layout.take() {
669        Some(img)
670    } else if render_image {
671        let tw = f64::from(width * 1.5).ceil().max(1.0) as i32;
672        let th = f64::from(height * 1.5).ceil().max(1.0) as i32;
673        let cfg = PdfRenderConfig::new()
674            .set_target_width(tw)
675            .set_target_height(th);
676        let big = crate::timing::timed("pdfium.render_layout", || {
677            page.render_with_config(&cfg)
678                .map(|b| b.as_image().into_rgb8())
679        })?;
680        let dw = f64::from(width).round_ties_even().max(1.0) as u32;
681        let dh = f64::from(height).round_ties_even().max(1.0) as u32;
682        Some(crate::timing::timed("image.resize_layout", || {
683            crate::resample::pil_resize(&big, dw, dh, crate::resample::PilFilter::Bicubic)
684        }))
685    } else {
686        None
687    };
688
689    let mut links = extract_links(page, unrot_h);
690    if rotation != 0 {
691        for l in &mut links {
692            let (a, t, r, b) = to_display_frame((l.l, l.t, l.r, l.b), rotation, unrot_w, unrot_h);
693            (l.l, l.t, l.r, l.b) = (a, t, r, b);
694        }
695    }
696
697    // `/Rotate` normalization for scanned pages: pdfium renders the page as a
698    // viewer displays it — `/Rotate` applied — so a rotated scan hands layout
699    // and OCR a sideways/upside-down raster and the recognition output is
700    // garbage. A page with a text layer needs none of this (its cells carry
701    // the geometry; the models never see its pixels decide text), so the
702    // normalization is gated to pages with no cells at all — exactly the set
703    // the OCR path fires on. The bitmaps are un-rotated to upright (lossless
704    // 90° steps), `width`/`height` swap to the upright box, and the display
705    // rotation is recorded so assembly can rotate the finished geometry back
706    // into display space (docling reports rotated pages in display coords).
707    let scanned = cells.is_empty() && word_cells.is_empty() && code_cells.is_empty();
708    let mut page = PdfPage {
709        width,
710        height,
711        scale: RENDER_SCALE,
712        image_layout,
713        cells,
714        code_cells,
715        word_cells,
716        image,
717        links,
718        rotation: 0,
719    };
720    if rotation != 0 && scanned && render_image {
721        page.unrotate(rotation);
722    }
723    Ok(page)
724}
725
726#[cfg(feature = "ml")]
727/// The supersample→target downscale via `fast_image_resize` (SIMD convolution;
728/// the same a=-0.5 Catmull-Rom kernel as `image::imageops::resize(...,
729/// CatmullRom)` and PIL BICUBIC — see the render comment above). Set
730/// `DOCLING_RS_SLOW_RESIZE=1` to fall back to the `image`-crate scalar resize
731/// (byte-parity with the pre-SIMD pipeline, several times slower).
732fn fast_downscale(big: &RgbImage, dw: u32, dh: u32) -> RgbImage {
733    use fast_image_resize as fir;
734    static SLOW: std::sync::OnceLock<bool> = std::sync::OnceLock::new();
735    let slow = *SLOW.get_or_init(|| docling_core::env::flag("DOCLING_RS_SLOW_RESIZE"));
736    if !slow {
737        if let Some(out) = (|| {
738            let src = fir::images::ImageRef::new(
739                big.width(),
740                big.height(),
741                big.as_raw(),
742                fir::PixelType::U8x3,
743            )
744            .ok()?;
745            let mut dst = fir::images::Image::new(dw, dh, fir::PixelType::U8x3);
746            fir::Resizer::new()
747                .resize(
748                    &src,
749                    &mut dst,
750                    &fir::ResizeOptions::new()
751                        .resize_alg(fir::ResizeAlg::Convolution(fir::FilterType::CatmullRom)),
752                )
753                .ok()?;
754            RgbImage::from_raw(dw, dh, dst.into_vec())
755        })() {
756            return out;
757        }
758        // Unreachable in practice; fall through to the scalar path on any error.
759    }
760    image::imageops::resize(big, dw, dh, image::imageops::FilterType::CatmullRom)
761}
762
763#[cfg(feature = "ml")]
764/// Collect web/mail/tel hyperlink annotations on a page, mapping each link's
765/// rectangle into top-left page coordinates (like [`TextCell`]). `file://` and
766/// in-document destinations are skipped — only externally meaningful targets are
767/// rendered. pdfium occasionally lists a link twice; rects are kept as-is and the
768/// caller dedupes by resolved anchor text.
769fn extract_links(page: &pdfium_render::prelude::PdfPage<'_>, page_h: f32) -> Vec<LinkAnnot> {
770    let mut out = Vec::new();
771    for link in page.links().iter() {
772        let Some(uri) = link
773            .action()
774            .and_then(|a| a.as_uri_action().and_then(|u| u.uri().ok()))
775        else {
776            continue;
777        };
778        let scheme_ok = ["http://", "https://", "mailto:", "tel:"]
779            .iter()
780            .any(|s| uri.starts_with(s));
781        if !scheme_ok {
782            continue;
783        }
784        if let Ok(rect) = link.rect() {
785            out.push(LinkAnnot {
786                l: rect.left().value,
787                t: page_h - rect.top().value,
788                r: rect.right().value,
789                b: page_h - rect.bottom().value,
790                uri,
791            });
792        }
793    }
794    out
795}
796
797/// Map a top-left-origin rect from a page's unrotated (MediaBox) frame into its
798/// `/Rotate`d display frame — the counterpart of docling's pypdfium2
799/// `_rect_to_display_frame` (docling#4008) for our y-down coordinates.
800/// `unrot_w`/`unrot_h` are the unrotated page box; the display box is the same
801/// for 180° and swapped for 90°/270°.
802pub(crate) fn to_display_frame(
803    (l, t, r, b): (f32, f32, f32, f32),
804    rotation: u16,
805    unrot_w: f32,
806    unrot_h: f32,
807) -> (f32, f32, f32, f32) {
808    match rotation {
809        // Page turned 90° clockwise for display: the unrotated top edge becomes
810        // the display right edge, so x' runs from the old bottom edge up.
811        90 => (unrot_h - b, l, unrot_h - t, r),
812        180 => (unrot_w - r, unrot_h - b, unrot_w - l, unrot_h - t),
813        270 => (t, unrot_w - r, b, unrot_w - l),
814        _ => (l, t, r, b),
815    }
816}
817
818#[cfg(feature = "ml")]
819/// Fallback line cells from pdfium-render's style segments (one cell per
820/// segment). Used only when the raw-FFI text page can't be loaded.
821fn segment_cells(text: &PdfPageText, page_h: f32) -> Vec<TextCell> {
822    text.segments()
823        .iter()
824        .filter_map(|seg| {
825            let s = seg.text();
826            if s.trim().is_empty() {
827                return None;
828            }
829            let r = seg.bounds();
830            Some(TextCell {
831                text: s,
832                l: r.left().value,
833                t: page_h - r.top().value,
834                r: r.right().value,
835                b: page_h - r.bottom().value,
836            })
837        })
838        .collect()
839}
840
841#[cfg(feature = "ml")]
842/// A second, raw-FFI handle on the same PDF used to drive the character loop
843/// (`FPDFText_GetUnicode`/`GetCharBox`) that pdfium-render's safe API doesn't
844/// expose. Closes the document on drop.
845struct FfiText<'a> {
846    bindings: &'a dyn PdfiumLibraryBindings,
847    doc: FPDF_DOCUMENT,
848}
849
850/// One glyph: codepoint + native (y-up) box edges. `l/b/r/t` is pdfium's *tight*
851/// ink box (used by the legacy `lines_from_glyphs`); `ll/lb/lr/lt` is the *loose*
852/// box (font ascent/descent + advance — uniform per font/size), which the
853/// docling-parse-style sanitizer needs so adjacent glyphs share a top edge.
854pub(crate) struct Glyph {
855    pub(crate) ch: char,
856    pub(crate) l: f32,
857    pub(crate) b: f32,
858    pub(crate) r: f32,
859    pub(crate) t: f32,
860    pub(crate) ll: f32,
861    pub(crate) lb: f32,
862    pub(crate) lr: f32,
863    pub(crate) lt: f32,
864    /// Hash of the PDF font name + flags (0 when not fetched). The sanitizer uses
865    /// it for docling-parse's `enforce_same_font` (keeps a bold label and regular
866    /// value as separate line cells, e.g. `LABEL : value`).
867    pub(crate) font: u64,
868}
869
870#[cfg(feature = "ml")]
871impl<'a> FfiText<'a> {
872    fn load(bindings: &'a dyn PdfiumLibraryBindings, bytes: &[u8], password: Option<&str>) -> Self {
873        let doc = bindings.FPDF_LoadMemDocument(bytes, password);
874        FfiText { bindings, doc }
875    }
876
877    /// Reconstruct line cells for page `index` (zero-based) via the
878    /// chars→words→lines grouping. Returns `(prose_cells, code_cells)` — the same
879    /// glyphs grouped two ways (gap-heuristic for prose, space-glyph-only for
880    /// code). Both empty on any failure (caller falls back).
881    fn page_cells(&self, index: i32, page_h: f32) -> (Vec<TextCell>, Vec<TextCell>, Vec<TextCell>) {
882        let empty = || (Vec::new(), Vec::new(), Vec::new());
883        if self.doc.is_null() {
884            return empty();
885        }
886        let b = self.bindings;
887        let page = b.FPDF_LoadPage(self.doc, index);
888        if page.is_null() {
889            return empty();
890        }
891        let tp = b.FPDFText_LoadPage(page);
892        let out = if tp.is_null() {
893            empty()
894        } else {
895            let dp = use_dp_lines();
896            let g = glyphs(b, tp, dp);
897            b.FPDFText_ClosePage(tp);
898            // Prose line cells: the docling-parse-style sanitizer (behind a flag
899            // while it's validated) or the legacy gap-heuristic reconstruction.
900            let prose = if dp {
901                crate::dp_lines::line_cells(&g, page_h, false)
902            } else {
903                lines_from_glyphs(&g, page_h, Grouping::Prose)
904            };
905            (
906                prose,
907                lines_from_glyphs(&g, page_h, Grouping::CodeSpaceOnly),
908                words_from_glyphs(&g, page_h),
909            )
910        };
911        b.FPDF_ClosePage(page);
912        out
913    }
914}
915
916#[cfg(feature = "ml")]
917impl Drop for FfiText<'_> {
918    fn drop(&mut self) {
919        if !self.doc.is_null() {
920            self.bindings.FPDF_CloseDocument(self.doc);
921        }
922    }
923}
924
925#[cfg(feature = "ml")]
926/// Read every glyph (codepoint + native box) from the text page, in document
927/// order. A space glyph is kept as a word-boundary marker (NaN box, char `' '`);
928/// pdfium emits these on most lines and they pin word splits exactly. Hard line
929/// breaks are dropped (line structure comes from geometry); the gap heuristic in
930/// [`lines_from_glyphs`] is the fallback for the lines pdfium leaves space-less.
931/// Debug helper: the raw pdfium glyph stream (codepoint + native bottom-left
932/// box) for a page, in pdfium's character order. For comparing against
933/// docling-parse's char cells.
934pub fn debug_glyphs(bytes: &[u8], index: i32) -> Vec<(char, f32, f32)> {
935    let Ok(pdfium) = bind() else {
936        return Vec::new();
937    };
938    let ffi = FfiText::load(pdfium.bindings(), bytes, None);
939    if ffi.doc.is_null() {
940        return Vec::new();
941    }
942    let b = ffi.bindings;
943    let page = b.FPDF_LoadPage(ffi.doc, index);
944    if page.is_null() {
945        return Vec::new();
946    }
947    let tp = b.FPDFText_LoadPage(page);
948    let mut out = Vec::new();
949    if !tp.is_null() {
950        for g in glyphs(b, tp, true) {
951            out.push((g.ch, g.ll, g.lr));
952        }
953        b.FPDFText_ClosePage(tp);
954    }
955    b.FPDF_ClosePage(page);
956    out
957}
958
959#[cfg(feature = "ml")]
960/// One text object on a page, for the hidden-layer diagnostic.
961#[derive(Debug, Clone)]
962pub struct DebugTextObject {
963    /// True when the object is drawn invisibly (text render mode 3) — the marker of
964    /// a hidden duplicate text layer.
965    pub invisible: bool,
966    /// Bounding box in native PDF points (bottom-left origin).
967    pub l: f32,
968    pub b: f32,
969    pub r: f32,
970    pub t: f32,
971    /// The object's text (best-effort; empty if it could not be read).
972    pub text: String,
973}
974
975#[cfg(feature = "ml")]
976/// Diagnostic: every text object on page `index`, each tagged visible/invisible
977/// (via the object-level [`FPDFTextObj_GetTextRenderMode`], which — unlike the
978/// per-character render-mode API — is available on the default pdfium binding).
979/// A hidden duplicate text layer shows up as invisible objects repeating the
980/// visible text. Used by the `dump_render_modes` example.
981///
982/// [`FPDFTextObj_GetTextRenderMode`]: pdfium_render::prelude::PdfiumLibraryBindings::FPDFTextObj_GetTextRenderMode
983pub fn debug_text_objects(bytes: &[u8], index: i32) -> Vec<DebugTextObject> {
984    let Ok(pdfium) = bind() else {
985        return Vec::new();
986    };
987    let ffi = FfiText::load(pdfium.bindings(), bytes, None);
988    if ffi.doc.is_null() {
989        return Vec::new();
990    }
991    let b = ffi.bindings;
992    let page = b.FPDF_LoadPage(ffi.doc, index);
993    if page.is_null() {
994        return Vec::new();
995    }
996    let tp = b.FPDFText_LoadPage(page);
997    let mut out = Vec::new();
998    let n = b.FPDFPage_CountObjects(page);
999    for i in 0..n {
1000        let obj = b.FPDFPage_GetObject(page, i);
1001        if obj.is_null() || b.FPDFPageObj_GetType(obj) != FPDF_PAGEOBJ_TEXT as i32 {
1002            continue;
1003        }
1004        let (mut l, mut bot, mut r, mut top) = (0f32, 0f32, 0f32, 0f32);
1005        if b.FPDFPageObj_GetBounds(obj, &mut l, &mut bot, &mut r, &mut top) == 0 {
1006            continue;
1007        }
1008        let invisible = b.FPDFTextObj_GetTextRenderMode(obj) == INVISIBLE_RENDER_MODE;
1009        let text = if tp.is_null() {
1010            String::new()
1011        } else {
1012            // FPDFTextObj_GetText returns the count of UTF-16 code units, including
1013            // the trailing NUL; call once for the size, once to fill.
1014            let need = b.FPDFTextObj_GetText(obj, tp, std::ptr::null_mut(), 0);
1015            if need <= 1 {
1016                String::new()
1017            } else {
1018                let mut buf = vec![0u16; need as usize];
1019                b.FPDFTextObj_GetText(obj, tp, buf.as_mut_ptr(), need);
1020                if let Some(&0) = buf.last() {
1021                    buf.pop();
1022                }
1023                String::from_utf16_lossy(&buf)
1024            }
1025        };
1026        out.push(DebugTextObject {
1027            invisible,
1028            l,
1029            b: bot,
1030            r,
1031            t: top,
1032            text,
1033        });
1034    }
1035    if !tp.is_null() {
1036        b.FPDFText_ClosePage(tp);
1037    }
1038    b.FPDF_ClosePage(page);
1039    out
1040}
1041
1042#[cfg(feature = "ml")]
1043/// Hash a glyph's PDF font name + flags, for `enforce_same_font`. 0 if unavailable.
1044fn font_hash(b: &dyn PdfiumLibraryBindings, tp: FPDF_TEXTPAGE, i: i32) -> u64 {
1045    use std::hash::{Hash, Hasher};
1046    let mut flags: std::os::raw::c_int = 0;
1047    let len = b.FPDFText_GetFontInfo(tp, i, std::ptr::null_mut(), 0, &mut flags);
1048    if len == 0 {
1049        return 0;
1050    }
1051    let mut buf = vec![0u8; len as usize];
1052    b.FPDFText_GetFontInfo(
1053        tp,
1054        i,
1055        buf.as_mut_ptr() as *mut std::os::raw::c_void,
1056        len,
1057        &mut flags,
1058    );
1059    let mut h = std::collections::hash_map::DefaultHasher::new();
1060    buf.hash(&mut h);
1061    flags.hash(&mut h);
1062    h.finish()
1063}
1064
1065#[cfg(feature = "ml")]
1066/// A glyph's PDF font name (NUL-trimmed), or empty if unavailable.
1067fn font_name_bytes(b: &dyn PdfiumLibraryBindings, tp: FPDF_TEXTPAGE, i: i32) -> Vec<u8> {
1068    let mut flags: std::os::raw::c_int = 0;
1069    let len = b.FPDFText_GetFontInfo(tp, i, std::ptr::null_mut(), 0, &mut flags);
1070    if len == 0 {
1071        return Vec::new();
1072    }
1073    let mut buf = vec![0u8; len as usize];
1074    b.FPDFText_GetFontInfo(
1075        tp,
1076        i,
1077        buf.as_mut_ptr() as *mut std::os::raw::c_void,
1078        len,
1079        &mut flags,
1080    );
1081    while buf.last() == Some(&0) {
1082        buf.pop();
1083    }
1084    buf
1085}
1086
1087#[cfg(feature = "ml")]
1088/// Read the text layer's glyph boxes and font styles for the given **1-based**
1089/// pages — the heading-hierarchy stage's style signal (#302). A separate,
1090/// on-demand pass over the text pages (no rendering), so the extraction
1091/// pipeline itself stays byte-identical whether or not the stage runs; pages
1092/// without a text layer (scans) simply yield no glyphs and the stage falls
1093/// back to its other signals. Boxes are the *loose* char boxes (font ascent +
1094/// descent — the font-size proxy), converted to top-left origin.
1095pub(crate) fn glyph_styles(
1096    bytes: &[u8],
1097    password: Option<&str>,
1098    pages: &[usize],
1099) -> std::collections::HashMap<usize, Vec<crate::heading_hierarchy::GlyphStyle>> {
1100    use crate::heading_hierarchy::GlyphStyle;
1101    let mut out = std::collections::HashMap::new();
1102    let Ok(pdfium) = bind() else {
1103        return out;
1104    };
1105    let ffi = FfiText::load(pdfium.bindings(), bytes, password);
1106    if ffi.doc.is_null() {
1107        return out;
1108    }
1109    let b = ffi.bindings;
1110    // Each distinct font name parses once per document.
1111    let mut cache: std::collections::HashMap<Vec<u8>, crate::font_style::FontStyle> =
1112        std::collections::HashMap::new();
1113    for &page_no in pages {
1114        if page_no == 0 {
1115            continue;
1116        }
1117        let page = b.FPDF_LoadPage(ffi.doc, (page_no - 1) as i32);
1118        if page.is_null() {
1119            continue;
1120        }
1121        let page_h = b.FPDF_GetPageHeightF(page);
1122        let tp = b.FPDFText_LoadPage(page);
1123        if !tp.is_null() {
1124            let n = b.FPDFText_CountChars(tp);
1125            let mut styles = Vec::with_capacity(n.max(0) as usize);
1126            for i in 0..n {
1127                let ch = match char::from_u32(b.FPDFText_GetUnicode(tp, i)) {
1128                    Some(c) => c,
1129                    None => continue,
1130                };
1131                if ch.is_whitespace() {
1132                    continue;
1133                }
1134                let mut lr = FS_RECTF {
1135                    left: 0.0,
1136                    top: 0.0,
1137                    right: 0.0,
1138                    bottom: 0.0,
1139                };
1140                if b.FPDFText_GetLooseCharBox(tp, i, &mut lr) == 0 {
1141                    continue;
1142                }
1143                let name = font_name_bytes(b, tp, i);
1144                let style = *cache.entry(name).or_insert_with_key(|n| {
1145                    crate::font_style::parse_font_style(&String::from_utf8_lossy(n))
1146                });
1147                styles.push(GlyphStyle {
1148                    l: lr.left,
1149                    t: page_h - lr.top,
1150                    r: lr.right,
1151                    b: page_h - lr.bottom,
1152                    height: lr.top - lr.bottom,
1153                    weight_cls: crate::font_style::weight_class(style.weight),
1154                    italic: style.italic,
1155                    styled: style.known,
1156                });
1157            }
1158            b.FPDFText_ClosePage(tp);
1159            out.insert(page_no, styles);
1160        }
1161        b.FPDF_ClosePage(page);
1162    }
1163    out
1164}
1165
1166#[cfg(feature = "ml")]
1167/// pdfium text render mode 3: the glyph is drawn with neither fill nor stroke —
1168/// an invisible glyph. Web-to-PDF exporters put a hidden plain-text copy of
1169/// syntax-highlighted code (and other "copy"/accessibility layers) in this mode,
1170/// which the char-level text API then extracts as a duplicate of the visible text.
1171const INVISIBLE_RENDER_MODE: i32 = 3;
1172
1173#[cfg(feature = "ml")]
1174fn glyphs(b: &dyn PdfiumLibraryBindings, tp: FPDF_TEXTPAGE, fetch_font: bool) -> Vec<Glyph> {
1175    let n = b.FPDFText_CountChars(tp);
1176    let mut out = Vec::with_capacity(n.max(0) as usize);
1177    for i in 0..n {
1178        let ch = match char::from_u32(b.FPDFText_GetUnicode(tp, i)) {
1179            Some(c) => c,
1180            None => continue,
1181        };
1182        if ch == '\r' || ch == '\n' {
1183            continue;
1184        }
1185        // Spaces are font-neutral (0): pdfium's generated spaces carry a default
1186        // font that would otherwise block every word↔space merge under
1187        // enforce_same_font; docling-parse's spaces inherit the run's font.
1188        let font = if fetch_font && !ch.is_whitespace() {
1189            font_hash(b, tp, i)
1190        } else {
1191            0
1192        };
1193        let (mut l, mut r, mut bot, mut top) = (0f64, 0f64, 0f64, 0f64);
1194        let has_box = b.FPDFText_GetCharBox(tp, i, &mut l, &mut r, &mut bot, &mut top) != 0;
1195        // Loose box: font ascent/descent + glyph advance, uniform per font/size.
1196        let mut lr = FS_RECTF {
1197            left: 0.0,
1198            top: 0.0,
1199            right: 0.0,
1200            bottom: 0.0,
1201        };
1202        let (ll, lb, lrt, ltop) = if b.FPDFText_GetLooseCharBox(tp, i, &mut lr) != 0 {
1203            (lr.left, lr.bottom, lr.right, lr.top)
1204        } else if has_box {
1205            (l as f32, bot as f32, r as f32, top as f32)
1206        } else {
1207            (f32::NAN, 0.0, 0.0, 0.0)
1208        };
1209        if ch.is_whitespace() {
1210            // Keep the space *with its box* (the docling-parse-style line sanitizer
1211            // needs literal space glyphs); NaN `l` if pdfium reports no box (the
1212            // legacy `lines_from_glyphs` ignores the box and only flags a space).
1213            out.push(Glyph {
1214                ch: ' ',
1215                l: if has_box { l as f32 } else { f32::NAN },
1216                b: if has_box { bot as f32 } else { 0.0 },
1217                r: if has_box { r as f32 } else { 0.0 },
1218                t: if has_box { top as f32 } else { 0.0 },
1219                ll,
1220                lb,
1221                lr: lrt,
1222                lt: ltop,
1223                font,
1224            });
1225            continue;
1226        }
1227        if !has_box {
1228            continue;
1229        }
1230        out.push(Glyph {
1231            ch,
1232            l: l as f32,
1233            b: bot as f32,
1234            r: r as f32,
1235            t: top as f32,
1236            ll,
1237            lb,
1238            lr: lrt,
1239            lt: ltop,
1240            font,
1241        });
1242    }
1243    // pdfium splits the Arabic lam-alef ligature into two chars at the *same* x
1244    // (it's one glyph) in visual order — `alef-variant, lam`. docling-parse and
1245    // logical order are `lam, alef-variant`. Detect the ligature by the shared x
1246    // and swap. The shared-x test reliably distinguishes a true ligature from a
1247    // genuine `alef + lam` sequence (the article `ال`, or `فعالة`), whose two
1248    // glyphs sit at different x and must NOT be reordered.
1249    for i in 0..out.len().saturating_sub(1) {
1250        let same_x = out[i].l.is_finite()
1251            && out[i + 1].l.is_finite()
1252            && (out[i].l - out[i + 1].l).abs() < 1.0;
1253        if same_x
1254            && matches!(out[i].ch, '\u{0622}' | '\u{0623}' | '\u{0625}' | '\u{0627}')
1255            && out[i + 1].ch == '\u{0644}'
1256        {
1257            out.swap(i, i + 1);
1258        }
1259    }
1260    // Reconstruct degenerate (zero-width) loose space boxes by spanning the gap to
1261    // the next glyph on the same line, so the sanitizer keeps them as word
1262    // separators rather than dropping them (which would merge `Information systems`
1263    // → `Informationsystems`). pdfium gives generated spaces a zero-width box at a
1264    // wrong baseline; a wrap (different baseline) or a touching gap is left alone.
1265    for i in 0..out.len() {
1266        if out[i].ch != ' ' || (out[i].lr - out[i].ll).abs() >= 0.5 {
1267            continue;
1268        }
1269        let prev = out[..i]
1270            .iter()
1271            .rev()
1272            .find(|g| g.ch != ' ' && g.ll.is_finite())
1273            .map(|g| (g.lr, g.lb, g.lt));
1274        let next = out[i + 1..]
1275            .iter()
1276            .find(|g| g.ch != ' ' && g.ll.is_finite())
1277            .map(|g| (g.ll, g.lb));
1278        if let (Some((plr, plb, plt)), Some((nll, nlb))) = (prev, next) {
1279            let line_h = (plt - plb).abs().max(1.0);
1280            if (plb - nlb).abs() < line_h * 0.5 && nll > plr + 0.5 {
1281                out[i].ll = plr;
1282                out[i].lr = nll;
1283                out[i].lb = plb;
1284                out[i].lt = plt;
1285            }
1286        }
1287    }
1288    out
1289}
1290
1291/// How [`lines_from_glyphs`] splits a line into words.
1292#[derive(Clone, Copy, PartialEq)]
1293enum Grouping {
1294    /// Gap heuristic + punctuation glue (`engines,`, `[37`, `98.5`) — prose.
1295    Prose,
1296    /// Split only at literal space glyphs, never glue — pdfium code cells.
1297    /// pdfium's monospace listings carry a real space glyph at every source space,
1298    /// and its overhanging loose boxes would make the gap heuristic over-split
1299    /// (`f un c t i o n`), so honouring just the spaces reproduces the spacing.
1300    CodeSpaceOnly,
1301    /// Split on the inter-glyph **gap** (or a space glyph), but never glue — for
1302    /// the parser's code cells: the parser emits no space glyphs (a source space
1303    /// is a positioning gap), and its clean advance boxes make the gap reliable.
1304    /// Unlike [`Grouping::Prose`] there is no punctuation glue, so a real gap
1305    /// always splits (`et al. 2000`, not `et al.2000`) while genuinely touching
1306    /// tokens stay joined (`add(a,` / `b)`).
1307    CodeGap,
1308}
1309
1310/// Group glyphs (document order) into words then lines, the way docling-parse
1311/// does: a new **word** starts where the horizontal gap to the previous glyph
1312/// exceeds ~0.2 × the font height (a real space is ~0.3 × height; letter
1313/// tracking is smaller, so titles don't shatter); a new **line** starts where
1314/// the baseline drops by ~half the font height (a superscript rises without
1315/// dropping, so it stays on its line). Coordinates are flipped to top-left.
1316/// See [`Grouping`] for how each mode decides word boundaries.
1317fn lines_from_glyphs(gs: &[Glyph], page_h: f32, mode: Grouping) -> Vec<TextCell> {
1318    let mut cells: Vec<TextCell> = Vec::new();
1319    let mut words: Vec<String> = Vec::new(); // words on the current line
1320    let mut word = String::new();
1321    // current line bounding box, native
1322    let (mut ll, mut lb, mut lr, mut lt) = (
1323        f32::INFINITY,
1324        f32::INFINITY,
1325        f32::NEG_INFINITY,
1326        f32::NEG_INFINITY,
1327    );
1328    // Tallest glyph seen on the current line: the word-gap threshold is relative
1329    // to it, so a small-font run on the line (a superscript citation) isn't split
1330    // at its tight digit gaps, while a big display title isn't split at its wider
1331    // letter tracking. A real inter-word space is ~0.3× the font height.
1332    let mut line_h: f32 = 0.0;
1333    let mut prev: Option<&Glyph> = None;
1334    // A space glyph between non-space glyphs pins a word split the gap heuristic
1335    // can miss (tight justified spacing); it carries no geometry.
1336    let mut pending_space = false;
1337
1338    for g in gs {
1339        if g.ch == ' ' {
1340            pending_space = true;
1341            continue;
1342        }
1343        let h = (g.t - g.b).abs().max(1.0);
1344        let (mut new_word, mut new_line) = (false, false);
1345        if let Some(p) = prev {
1346            // A new line drops the baseline *and* resets x leftward; requiring the
1347            // x-reset avoids a descending comma/semicolon faking a line break. A
1348            // *large* drop (≥1.5× the line height — a skipped line, e.g. a centered
1349            // page-number footer below a short last word) is always a new line,
1350            // even without the x-reset.
1351            // LTR wraps reset x leftward (`g.l < p.r`); RTL (Arabic) wraps reset
1352            // rightward (the new line begins at the far right). A large drop
1353            // (≥1.5× line height) is a new line regardless of x.
1354            let x_reset = if is_arabic(g.ch) || is_arabic(p.ch) {
1355                g.l > p.r
1356            } else {
1357                g.l < p.r
1358            };
1359            new_line = (p.b - g.b > h * 0.5 && x_reset) || (p.b - g.b > line_h.max(h) * 1.5);
1360            // Don't split before closing punctuation, after opening punctuation, or
1361            // after a period that runs into a digit/lowercase letter — docling
1362            // keeps `engines,` / `[37` / `i.e.` / `98.5` together even across a
1363            // space or gap.
1364            let glued = is_close_punct(g.ch)
1365                || is_open_punct(p.ch)
1366                || (p.ch.is_ascii_digit() && g.ch.is_ascii_digit())
1367                || (p.ch == '.'
1368                    && !pending_space
1369                    && (g.ch.is_ascii_digit() || g.ch.is_ascii_lowercase()));
1370            let word_gap = line_h.max(h) * 0.25;
1371            new_word = if mode == Grouping::CodeSpaceOnly {
1372                new_line || pending_space
1373            } else if mode == Grouping::CodeGap {
1374                // Gap-based, no glue: a real gap always splits, touching tokens join.
1375                new_line || pending_space || g.l - p.r > word_gap
1376            } else if is_arabic(g.ch) || is_arabic(p.ch) {
1377                // RTL runs right-to-left, so the inter-word gap is `p.l - g.r`. A
1378                // real word space has a gap; pdfium also emits spurious zero-gap
1379                // space glyphs inside words (`التي`), so require the gap rather
1380                // than trusting a bare space glyph.
1381                new_line || (p.l - g.r > word_gap && !glued)
1382            } else {
1383                new_line || ((pending_space || g.l - p.r > word_gap) && !glued)
1384            };
1385        }
1386        pending_space = false;
1387        if new_line {
1388            push_word(&mut word, &mut words);
1389            push_line(&mut words, (ll, lb, lr, lt), page_h, &mut cells);
1390            (ll, lb, lr, lt) = (
1391                f32::INFINITY,
1392                f32::INFINITY,
1393                f32::NEG_INFINITY,
1394                f32::NEG_INFINITY,
1395            );
1396            line_h = 0.0;
1397        } else if new_word {
1398            push_word(&mut word, &mut words);
1399        }
1400        word.push(g.ch);
1401        ll = ll.min(g.l);
1402        lb = lb.min(g.b);
1403        lr = lr.max(g.r);
1404        lt = lt.max(g.t);
1405        line_h = line_h.max(h);
1406        prev = Some(g);
1407    }
1408    push_word(&mut word, &mut words);
1409    push_line(&mut words, (ll, lb, lr, lt), page_h, &mut cells);
1410    cells
1411}
1412
1413/// Code line cells from the **parser**'s glyph stream. Unlike pdfium — whose
1414/// monospace listings carry explicit space glyphs (so [`Grouping::CodeSpaceOnly`]
1415/// keeps their spacing) — the parser emits no space glyphs: a source space is a
1416/// positioning gap. So code cells use [`Grouping::CodeGap`], which splits on the
1417/// inter-glyph gap (a space wherever it exceeds ~0.25× the line height) but never
1418/// glues punctuation, so `et al. 2000` keeps its space while `add(a,` / `b)` stay
1419/// joined. The parser's clean advance boxes make the gap heuristic reliable here,
1420/// where pdfium's overhanging loose boxes would over-split (`f un c t i o n`).
1421pub(crate) fn code_cells_from_glyphs(gs: &[Glyph], page_h: f32) -> Vec<TextCell> {
1422    lines_from_glyphs(gs, page_h, Grouping::CodeGap)
1423}
1424
1425/// Per-word cells (each word's text + top-left bbox), using the same word/line
1426/// splitting as [`lines_from_glyphs`] but emitting one cell per word instead of
1427/// joining into lines — the legacy gap-heuristic word grouping, kept for the
1428/// pdfium word path (`DOCLING_PDFIUM_WORDS`). The default parser path uses
1429/// [`crate::dp_lines::word_cells`] instead.
1430pub(crate) fn words_from_glyphs(gs: &[Glyph], page_h: f32) -> Vec<TextCell> {
1431    let mut cells = Vec::new();
1432    let mut word = String::new();
1433    let inf = (
1434        f32::INFINITY,
1435        f32::INFINITY,
1436        f32::NEG_INFINITY,
1437        f32::NEG_INFINITY,
1438    );
1439    let (mut wl, mut wb, mut wr, mut wt) = inf;
1440    let mut line_h: f32 = 0.0;
1441    let mut prev: Option<&Glyph> = None;
1442    let mut pending_space = false;
1443    for g in gs {
1444        if g.ch == ' ' {
1445            pending_space = true;
1446            continue;
1447        }
1448        let h = (g.t - g.b).abs().max(1.0);
1449        let mut new_line = false;
1450        let mut new_word = false;
1451        if let Some(p) = prev {
1452            // LTR wraps reset x leftward (`g.l < p.r`); RTL (Arabic) wraps reset
1453            // rightward (the new line begins at the far right). A large drop
1454            // (≥1.5× line height) is a new line regardless of x.
1455            let x_reset = if is_arabic(g.ch) || is_arabic(p.ch) {
1456                g.l > p.r
1457            } else {
1458                g.l < p.r
1459            };
1460            new_line = (p.b - g.b > h * 0.5 && x_reset) || (p.b - g.b > line_h.max(h) * 1.5);
1461            // No digit-digit glue here (unlike the prose grouping): table cells in
1462            // adjacent columns are numeric and a column gap must still split them
1463            // (`0.965` `0.934`, not `0.9650.934`). Intra-number digits have no gap
1464            // so they stay together regardless.
1465            let glued = is_close_punct(g.ch)
1466                || is_open_punct(p.ch)
1467                || (p.ch == '.'
1468                    && !pending_space
1469                    && (g.ch.is_ascii_digit() || g.ch.is_ascii_lowercase()));
1470            let word_gap = line_h.max(h) * 0.25;
1471            new_word = new_line || ((pending_space || g.l - p.r > word_gap) && !glued);
1472        }
1473        pending_space = false;
1474        if new_word && !word.is_empty() {
1475            cells.push(TextCell {
1476                text: std::mem::take(&mut word),
1477                l: wl,
1478                t: page_h - wt,
1479                r: wr,
1480                b: page_h - wb,
1481            });
1482            (wl, wb, wr, wt) = inf;
1483        }
1484        if new_line {
1485            line_h = 0.0;
1486        }
1487        word.push(g.ch);
1488        wl = wl.min(g.l);
1489        wb = wb.min(g.b);
1490        wr = wr.max(g.r);
1491        wt = wt.max(g.t);
1492        line_h = line_h.max(h);
1493        prev = Some(g);
1494    }
1495    if !word.is_empty() {
1496        cells.push(TextCell {
1497            text: word,
1498            l: wl,
1499            t: page_h - wt,
1500            r: wr,
1501            b: page_h - wb,
1502        });
1503    }
1504    cells
1505}
1506
1507fn is_arabic(c: char) -> bool {
1508    ('\u{0600}'..='\u{06FF}').contains(&c)
1509}
1510
1511fn is_close_punct(c: char) -> bool {
1512    matches!(
1513        c,
1514        ',' | '.' | ';' | '!' | '?' | ')' | ']' | '}' | '%' | '\'' | '\u{2019}' | '\u{2018}'
1515    )
1516}
1517
1518fn is_open_punct(c: char) -> bool {
1519    // `@` glues to what follows (`mAP @0.5`, `bpf@zurich`, `@decorator`).
1520    matches!(c, '(' | '[' | '{' | '@')
1521}
1522
1523fn push_word(word: &mut String, words: &mut Vec<String>) {
1524    if !word.is_empty() {
1525        words.push(std::mem::take(word));
1526    }
1527}
1528
1529fn push_line(
1530    words: &mut Vec<String>,
1531    bbox: (f32, f32, f32, f32),
1532    page_h: f32,
1533    cells: &mut Vec<TextCell>,
1534) {
1535    if words.is_empty() {
1536        return;
1537    }
1538    let text = std::mem::take(words).join(" ");
1539    let (l, b, r, t) = bbox;
1540    cells.push(TextCell {
1541        text,
1542        l,
1543        t: page_h - t,
1544        r,
1545        b: page_h - b,
1546    });
1547}
1548
1549#[cfg(test)]
1550mod tests {
1551    use super::{checked_render_dims, max_render_side, to_display_frame};
1552
1553    /// A page whose declared size renders past the per-side cap is rejected
1554    /// with a recoverable error, before pdfium or `image` allocates the
1555    /// multi-gigabyte bitmap that would otherwise abort the process; a normal
1556    /// page passes through with its dimensions rounded to `i32`.
1557    #[test]
1558    fn oversized_render_is_rejected_not_allocated() {
1559        let cap = f64::from(max_render_side());
1560        // A 12000 pt box at the pipeline's 3x supersample is 36000 px/side.
1561        let huge = checked_render_dims(cap + 1.0, 10.0, 1);
1562        assert!(huge.is_err(), "over-cap width must be rejected");
1563        let tall = checked_render_dims(10.0, cap + 1.0, 7);
1564        assert!(tall.is_err(), "over-cap height must be rejected");
1565        assert!(
1566            tall.unwrap_err().to_string().contains("page 7"),
1567            "the error names the offending page"
1568        );
1569        // A Letter page at 2x supersample (612x792 pt -> 1836x2376 px) is fine.
1570        assert_eq!(
1571            checked_render_dims(1836.4, 2375.6, 1).unwrap(),
1572            (1836, 2376)
1573        );
1574        // Exactly at the cap is allowed; a zero-or-negative size floors to 1.
1575        assert_eq!(
1576            checked_render_dims(cap, cap, 1).unwrap(),
1577            (cap as i32, cap as i32)
1578        );
1579        assert_eq!(checked_render_dims(0.0, 0.0, 1).unwrap(), (1, 1));
1580    }
1581
1582    /// A 612×792 portrait page displayed under `/Rotate`: a rect near the
1583    /// unrotated top-left lands where a viewer shows it (docling#4008).
1584    #[test]
1585    fn display_frame_follows_the_page_rotation() {
1586        let r = (72.0, 63.0, 387.0, 74.0); // top-left origin, unrotated
1587        assert_eq!(to_display_frame(r, 0, 612.0, 792.0), r);
1588        // 90° clockwise: the page becomes 792×612; the old top edge is the
1589        // display right edge, old left edge the display top.
1590        assert_eq!(
1591            to_display_frame(r, 90, 612.0, 792.0),
1592            (718.0, 72.0, 729.0, 387.0)
1593        );
1594        // 180°: both axes mirror inside the same box.
1595        assert_eq!(
1596            to_display_frame(r, 180, 612.0, 792.0),
1597            (225.0, 718.0, 540.0, 729.0)
1598        );
1599        // 270°: the old top edge is the display left edge, old right edge the
1600        // display top.
1601        assert_eq!(
1602            to_display_frame(r, 270, 612.0, 792.0),
1603            (63.0, 225.0, 74.0, 540.0)
1604        );
1605    }
1606
1607    #[test]
1608    fn display_frame_rotations_compose_to_identity() {
1609        let r = (10.0, 20.0, 110.0, 40.0);
1610        // 90° then 270° from the intermediate (792×612) box round-trips.
1611        let once = to_display_frame(r, 90, 612.0, 792.0);
1612        assert_eq!(to_display_frame(once, 270, 792.0, 612.0), r);
1613        let twice = to_display_frame(to_display_frame(r, 180, 612.0, 792.0), 180, 612.0, 792.0);
1614        assert_eq!(twice, r);
1615    }
1616}